Files
messageboardbench/results/board-interface-v2-sept8/final-export/manifest.json
T

1155 lines
38 KiB
JSON

{
"run": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8",
"episodes": 12,
"exposure_edges": 3,
"board_sha256": "4a155de990f199e0aac8a11b33ab65b7e388243fed9e21ebef20a61d7706ba9c",
"board_snapshot_path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/board-final.json",
"explicit_board_snapshot": false,
"report_scope": "final snapshot / completed logs only",
"report_script_sha256": "ad62e28bba6222eddd296efbad10fa78b023d11e21aa32dac008b2fa7a89ce03",
"logs": [
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-09-05-00-00_board-pilot-board-c1-p1_3oe2iMTvYCNUkjLuKdrZ5r.eval",
"sha256": "e720634cbd118b540680d94346ee466ba9fc179dc3e369f182dd27ee77c58a77",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-09-05-00-00_board-pilot-board-c1-p2_f4okoTJno8NCjUyDFucq8D.eval",
"sha256": "1c5e18970f57a38b276b0a36652191b9fe7e920c893d59f9f9c36cb7276c5014",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-09-05-00-00_board-pilot-board-c1-p3_aQxvhGtpPqbJgvsZoXmtyK.eval",
"sha256": "9d41fe6f785d5b5eec0bc2eb1cdde03bc449908fdecf7385c012d65b9af489ae",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-14-27-00-00_board-pilot-private-c1-p1_Ycpb2MdDerMVSvqAcSjarS.eval",
"sha256": "4a106d2cf6103791b2859e8258edd2c579eda78dedbb211a660071869ffa0b19",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-14-27-00-00_board-pilot-private-c1-p2_LxD8zEXhov6JzwYH9gWKYw.eval",
"sha256": "3dc48d7e7baf0a8951301d7cbdbab67361ba08b6a23ab1db43e58b6968491191",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-14-27-00-00_board-pilot-private-c1-p3_m9MBRxvoKtuK4x9nHComhv.eval",
"sha256": "d6d6f86ca5587f9e0276c90c34627b14760b79f5f2d19efbdf22da1800b49c4d",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-21-12-00-00_board-pilot-private-c2-p1_LMFhSKNhJ26eLp2QuwewWn.eval",
"sha256": "dcc9aad42043b8e4b0f8061640e5f92f34cd6546c1c6f81457bb911d8c18dad1",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 2,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-21-12-00-00_board-pilot-private-c2-p2_JTh26ErjQGWP8mKWi2DRTr.eval",
"sha256": "16e0b859a9dcdf5ee8860c2327c7c6531f605afa1b60333d6c53023bf8ed70a6",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 2,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-21-13-00-00_board-pilot-private-c2-p3_f3KDRezb8eJPZHEhocqCtk.eval",
"sha256": "cdcfed1811ef9e54095416e9d67783534bc2fccc41cac48fec8e22b55721d1c5",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 2,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-31-49-00-00_board-pilot-board-c2-p1_Aq6HVepgji5rjYVzLSkqPd.eval",
"sha256": "cbce6b1e153e4fb428b7fcccef907dafcf665b1498c0ab9f10011bc3dd19b42f",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 2,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-31-49-00-00_board-pilot-board-c2-p2_6aL23zpvCk9ggAdga88xhA.eval",
"sha256": "79d09dba74b4bfec519e381eb529e454471b96da37ab96e7ffbd4ae2896c1f95",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 2,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-interface-v2-sept8/evals/2026-09-07T20-31-49-00-00_board-pilot-board-c2-p3_FkHpQjLvcVaP2g9aynmEKe.eval",
"sha256": "6605eed2cece65dc1638661d08410b78900b218fedf8077b7d0e1cdcbee2f5fd",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"board_interface_version": "team-messages-v2",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"Message limit 90; preliminary model diagnostic used 60.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 2,
"split": "conflicting"
}
}
],
"skipped_logs": [],
"unlinked_operation_count": 0,
"unmatched_audit_count": 0,
"limitations": [
"Zero-based original sample message/event indices.",
"Edges require successful peer-post responses present in tool messages; not proof of attention, adoption or collusion.",
"Host audit without a matched delivered tool response is retained but creates no exposure edge.",
"Repeated reads create repeated exposure edges; aggregate unique post IDs separately.",
"Model calls count model events, not provider-internal HTTP retries.",
"Reasoning is a subset of output; cached/repeated input contributes to total.",
"Scorer artifacts include intermediate and final grading events; they are not necessarily unique submissions.",
"Final source and scratch snapshots retain the scorer capture bounds and may be truncated.",
"No automatic gaming or intent labels are assigned."
]
}