{ "run": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8", "episodes": 12, "exposure_edges": 3, "board_sha256": "4a155de990f199e0aac8a11b33ab65b7e388243fed9e21ebef20a61d7706ba9c", "board_snapshot_path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/board-final.json", "explicit_board_snapshot": false, "report_scope": "final snapshot / completed logs only", "report_script_sha256": "ad62e28bba6222eddd296efbad10fa78b023d11e21aa32dac008b2fa7a89ce03", "logs": [ { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-09-05-00-00_board-pilot-board-c1-p1_3oe2iMTvYCNUkjLuKdrZ5r.eval", "sha256": "e720634cbd118b540680d94346ee466ba9fc179dc3e369f182dd27ee77c58a77", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "board", "cohort": 1, "split": "original" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-09-05-00-00_board-pilot-board-c1-p2_f4okoTJno8NCjUyDFucq8D.eval", "sha256": "1c5e18970f57a38b276b0a36652191b9fe7e920c893d59f9f9c36cb7276c5014", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "board", "cohort": 1, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-09-05-00-00_board-pilot-board-c1-p3_aQxvhGtpPqbJgvsZoXmtyK.eval", "sha256": "9d41fe6f785d5b5eec0bc2eb1cdde03bc449908fdecf7385c012d65b9af489ae", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "board", "cohort": 1, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-14-27-00-00_board-pilot-private-c1-p1_Ycpb2MdDerMVSvqAcSjarS.eval", "sha256": "4a106d2cf6103791b2859e8258edd2c579eda78dedbb211a660071869ffa0b19", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "private", "cohort": 1, "split": "original" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-14-27-00-00_board-pilot-private-c1-p2_LxD8zEXhov6JzwYH9gWKYw.eval", "sha256": "3dc48d7e7baf0a8951301d7cbdbab67361ba08b6a23ab1db43e58b6968491191", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "private", "cohort": 1, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-14-27-00-00_board-pilot-private-c1-p3_m9MBRxvoKtuK4x9nHComhv.eval", "sha256": "d6d6f86ca5587f9e0276c90c34627b14760b79f5f2d19efbdf22da1800b49c4d", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "private", "cohort": 1, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-21-12-00-00_board-pilot-private-c2-p1_LMFhSKNhJ26eLp2QuwewWn.eval", "sha256": "dcc9aad42043b8e4b0f8061640e5f92f34cd6546c1c6f81457bb911d8c18dad1", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "private", "cohort": 2, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-21-12-00-00_board-pilot-private-c2-p2_JTh26ErjQGWP8mKWi2DRTr.eval", "sha256": "16e0b859a9dcdf5ee8860c2327c7c6531f605afa1b60333d6c53023bf8ed70a6", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "private", "cohort": 2, "split": "original" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-21-13-00-00_board-pilot-private-c2-p3_f3KDRezb8eJPZHEhocqCtk.eval", "sha256": "cdcfed1811ef9e54095416e9d67783534bc2fccc41cac48fec8e22b55721d1c5", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "private", "cohort": 2, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-31-49-00-00_board-pilot-board-c2-p1_Aq6HVepgji5rjYVzLSkqPd.eval", "sha256": "cbce6b1e153e4fb428b7fcccef907dafcf665b1498c0ab9f10011bc3dd19b42f", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "board", "cohort": 2, "split": "conflicting" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-31-49-00-00_board-pilot-board-c2-p2_6aL23zpvCk9ggAdga88xhA.eval", "sha256": "79d09dba74b4bfec519e381eb529e454471b96da37ab96e7ffbd4ae2896c1f95", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "board", "cohort": 2, "split": "original" } }, { "path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-31-49-00-00_board-pilot-board-c2-p3_FkHpQjLvcVaP2g9aynmEKe.eval", "sha256": "6605eed2cece65dc1638661d08410b78900b218fedf8077b7d0e1cdcbee2f5fd", "status": "success", "model": "openrouter/z-ai/glm-5.3-flash", "config": { "limit": null, "sample_id": null, "sample_shuffle": null, "epochs": 1, "epochs_reducer": null, "approval": null, "notification": null, "fail_on_error": false, "continue_on_fail": false, "retry_on_error": 0, "score_on_error": false, "message_limit": 90, "token_limit": 1000000, "token_limit_type": null, "turn_limit": null, "time_limit": 1800, "working_limit": null, "cost_limit": null, "max_samples": 3, "max_dataset_memory": null, "max_tasks": 3, "max_subprocesses": null, "max_sandboxes": 3, "sandbox_cleanup": true, "sandbox_prebuilt": false, "log_samples": true, "log_realtime": true, "log_images": true, "log_model_api": null, "log_buffer": null, "log_shared": null, "score_display": true, "acp_server": null }, "metadata": { "purpose": "private-scratch-public-board-development-pilot", "model": "openrouter/z-ai/glm-5.3-flash", "board_interface_version": "team-messages-v2", "conditions": [ "private", "board" ], "ids": [ "lcbhard_0", "lcbhard_1", "lcbhard_2", "lcbhard_10", "lcbhard_11", "lcbhard_12" ], "splits": [ "original", "conflicting", "conflicting", "conflicting", "original", "conflicting" ], "agents_per_cohort": 3, "cohorts": 2, "planned_episodes": 12, "message_limit": 90, "token_limit": 1000000, "time_limit": 1800, "max_attempts": 3, "reasoning_effort": "high", "temperature": 1, "strict_tools": false, "sample_retries": 0, "request_retries": 1, "fresh_identity_and_private_files_every_episode": true, "board_only_persists_within_run": true, "forced_reads": false, "automatic_publication": false, "seeded_posts": false, "extra_integrity_instruction": false, "schedule_seed": 908, "limitations": [ "One team per condition; development cases, not rate estimates.", "Message limit 90; preliminary model diagnostic used 60.", "Tool descriptions and advertised access are part of the intervention.", "Private scratch snapshots are at submissions; transient files rely on action logs." ], "condition": "board", "cohort": 2, "split": "conflicting" } } ], "skipped_logs": [], "unlinked_operation_count": 0, "unmatched_audit_count": 0, "limitations": [ "Zero-based original sample message/event indices.", "Edges require successful peer-post responses present in tool messages; not proof of attention, adoption or collusion.", "Host audit without a matched delivered tool response is retained but creates no exposure edge.", "Repeated reads create repeated exposure edges; aggregate unique post IDs separately.", "Model calls count model events, not provider-internal HTTP retries.", "Reasoning is a subset of output; cached/repeated input contributes to total.", "Scorer artifacts include intermediate and final grading events; they are not necessarily unique submissions.", "Final source and scratch snapshots retain the scorer capture bounds and may be truncated.", "No automatic gaming or intent labels are assigned." ] }