Files
messageboardbench/results/board-pilot-sept8/phase1-export-v2/manifest.json
T

410 lines
14 KiB
JSON

{
"run": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8",
"episodes": 4,
"exposure_edges": 0,
"board_sha256": "38970eaf65b4e9eeb409e8e070fb8a2485259e9acb170739b5a5ae69363168b4",
"board_snapshot_path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/board-after-phase-1.json",
"explicit_board_snapshot": true,
"report_scope": "partial snapshot / completed logs only",
"report_script_sha256": "d448a36975a29fabff8a86b712a9b8b74b4b81d4d61ff1b7490b5a7fab0c1f4c",
"logs": [
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-01-58-00-00_board-pilot-board-c1-p1_MXeGhVFnNd6cknbzk8pCBo.eval",
"sha256": "8a003021cbe7c2141b3df1b5d9e3dec930be0d3c7cb09d5efae9f17820c0d12f",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-01-58-00-00_board-pilot-board-c1-p2_HWqPfyg8RWqT2M4Dtwe8Xa.eval",
"sha256": "1139332a9ada9e1668b2b68227fd490b9837a63699783f5304f4b7b3c6a6448b",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-01-58-00-00_board-pilot-board-c1-p3_QAU44iehguAMqgTH7xkNu5.eval",
"sha256": "eace45d6e030927ce9b559f4fbe824bd116b7ea6b06f1a08bb6184353a75de17",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-08-21-00-00_board-pilot-private-c1-p1_EakSh9iqXg4KnaKMkWP5YY.eval",
"sha256": "689fbf5c9403ffd5b531fd53a6b8bb811e53200efdd69ed160ac3d34a4a4e51e",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "original"
}
}
],
"skipped_logs": [
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-08-21-00-00_board-pilot-private-c1-p2_gWBBXKTqMatTSJtb6ErXRy.eval",
"status": "started",
"reason": "not completed"
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-08-22-00-00_board-pilot-private-c1-p3_LC3AQwpWev6oWBoQ3TTQto.eval",
"status": "started",
"reason": "not completed"
}
],
"unlinked_operation_count": 0,
"unmatched_audit_count": 0,
"limitations": [
"Zero-based original sample message/event indices.",
"Edges require successful peer-post responses present in tool messages; not proof of attention, adoption or collusion.",
"Host audit without a matched delivered tool response is retained but creates no exposure edge.",
"Repeated reads create repeated exposure edges; aggregate unique post IDs separately.",
"Model calls count model events, not provider-internal HTTP retries.",
"Reasoning is a subset of output; cached/repeated input contributes to total.",
"Scorer artifacts include intermediate and final grading events; they are not necessarily unique submissions.",
"Final source and scratch snapshots retain the scorer capture bounds and may be truncated.",
"No automatic gaming or intent labels are assigned."
]
}