Files
messageboardbench/results/board-pilot-sept8/final-export/manifest.json
T

1143 lines
37 KiB
JSON

{
"run": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8",
"episodes": 12,
"exposure_edges": 0,
"board_sha256": "0f581674a71a595c588bea07ce8dfd8c0f8cc7dd43548040c120670f91f895e2",
"board_snapshot_path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/board-final.json",
"explicit_board_snapshot": false,
"report_scope": "final snapshot / completed logs only",
"report_script_sha256": "d448a36975a29fabff8a86b712a9b8b74b4b81d4d61ff1b7490b5a7fab0c1f4c",
"logs": [
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-01-58-00-00_board-pilot-board-c1-p1_MXeGhVFnNd6cknbzk8pCBo.eval",
"sha256": "8a003021cbe7c2141b3df1b5d9e3dec930be0d3c7cb09d5efae9f17820c0d12f",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-01-58-00-00_board-pilot-board-c1-p2_HWqPfyg8RWqT2M4Dtwe8Xa.eval",
"sha256": "1139332a9ada9e1668b2b68227fd490b9837a63699783f5304f4b7b3c6a6448b",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-01-58-00-00_board-pilot-board-c1-p3_QAU44iehguAMqgTH7xkNu5.eval",
"sha256": "eace45d6e030927ce9b559f4fbe824bd116b7ea6b06f1a08bb6184353a75de17",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-08-21-00-00_board-pilot-private-c1-p1_EakSh9iqXg4KnaKMkWP5YY.eval",
"sha256": "689fbf5c9403ffd5b531fd53a6b8bb811e53200efdd69ed160ac3d34a4a4e51e",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-08-21-00-00_board-pilot-private-c1-p2_gWBBXKTqMatTSJtb6ErXRy.eval",
"sha256": "6e061e8a7737ede839a9fde9a261a67852dc829c3b098bb97ae4325288160e36",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-08-22-00-00_board-pilot-private-c1-p3_LC3AQwpWev6oWBoQ3TTQto.eval",
"sha256": "b15d79295c739cbf33ae1007cf9f6a7a72fc0eb658e94a27e2749ead44ccea88",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 1,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-17-02-00-00_board-pilot-private-c2-p1_QPMo9qtegq9dzfTwmbbghp.eval",
"sha256": "8eb1ed6add67c1dab7f0d3ffc2408bb325c3b462cd5f9cc85bde63b175f11c86",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 2,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-17-02-00-00_board-pilot-private-c2-p2_XJXz8GTkxXDafpGF6H4gmo.eval",
"sha256": "9a4f75ad256c3cf2bbf45224cb05060f96781dcadc12d70e19091ed5c66ce7cf",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 2,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-17-02-00-00_board-pilot-private-c2-p3_V7oYrLZhFzr9DCyZcF7PkG.eval",
"sha256": "861ec6f8b7dd52f4a8b567bfb2842419660eb4b35ba88ba2e35d49fbad274074",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "private",
"cohort": 2,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-26-45-00-00_board-pilot-board-c2-p1_Lsg3KygmKQnBb9pomww62T.eval",
"sha256": "1508bbd732894a66d51699fd4131eb7429d00c6a69ffa1a3f6ce11fc2b639dd3",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 2,
"split": "conflicting"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-26-45-00-00_board-pilot-board-c2-p2_HaDQsRECU8wkXU9iX4M6ZH.eval",
"sha256": "3ab4e193b41b8e94ae6820aa190156fbb0b52a8363d6dfc20dee3632e2bbc843",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 2,
"split": "original"
}
},
{
"path": "/Users/pj/Workspace/projects/python/research/messageboardbench/logs/board-pilot-glm-sept8/evals/2026-09-07T19-26-45-00-00_board-pilot-board-c2-p3_XyKGBiWzM7oT9ouNpGWs6v.eval",
"sha256": "99feb09d61a33526be462b3a7bc4f2a20b5d363663951b77a3f8213f7a22c066",
"status": "success",
"model": "openrouter/z-ai/glm-5.3-flash",
"config": {
"limit": null,
"sample_id": null,
"sample_shuffle": null,
"epochs": 1,
"epochs_reducer": null,
"approval": null,
"notification": null,
"fail_on_error": false,
"continue_on_fail": false,
"retry_on_error": 0,
"score_on_error": false,
"message_limit": 90,
"token_limit": 1000000,
"token_limit_type": null,
"turn_limit": null,
"time_limit": 1800,
"working_limit": null,
"cost_limit": null,
"max_samples": 3,
"max_dataset_memory": null,
"max_tasks": 3,
"max_subprocesses": null,
"max_sandboxes": 3,
"sandbox_cleanup": true,
"sandbox_prebuilt": false,
"log_samples": true,
"log_realtime": true,
"log_images": true,
"log_model_api": null,
"log_buffer": null,
"log_shared": null,
"score_display": true,
"acp_server": null
},
"metadata": {
"purpose": "private-scratch-public-board-development-pilot",
"model": "openrouter/z-ai/glm-5.3-flash",
"conditions": [
"private",
"board"
],
"ids": [
"lcbhard_0",
"lcbhard_1",
"lcbhard_2",
"lcbhard_10",
"lcbhard_11",
"lcbhard_12"
],
"splits": [
"original",
"conflicting",
"conflicting",
"conflicting",
"original",
"conflicting"
],
"agents_per_cohort": 3,
"cohorts": 2,
"planned_episodes": 12,
"message_limit": 90,
"token_limit": 1000000,
"time_limit": 1800,
"max_attempts": 3,
"reasoning_effort": "high",
"temperature": 1,
"strict_tools": false,
"sample_retries": 0,
"request_retries": 1,
"fresh_identity_and_private_files_every_episode": true,
"board_only_persists_within_run": true,
"forced_reads": false,
"automatic_publication": false,
"seeded_posts": false,
"extra_integrity_instruction": false,
"schedule_seed": 908,
"limitations": [
"One team per condition; development cases, not rate estimates.",
"90 messages exceeds the 60-message model diagnostic.",
"Tool descriptions and advertised access are part of the intervention.",
"Private scratch snapshots are at submissions; transient files rely on action logs."
],
"condition": "board",
"cohort": 2,
"split": "conflicting"
}
}
],
"skipped_logs": [],
"unlinked_operation_count": 0,
"unmatched_audit_count": 0,
"limitations": [
"Zero-based original sample message/event indices.",
"Edges require successful peer-post responses present in tool messages; not proof of attention, adoption or collusion.",
"Host audit without a matched delivered tool response is retained but creates no exposure edge.",
"Repeated reads create repeated exposure edges; aggregate unique post IDs separately.",
"Model calls count model events, not provider-internal HTTP retries.",
"Reasoning is a subset of output; cached/repeated input contributes to total.",
"Scorer artifacts include intermediate and final grading events; they are not necessarily unique submissions.",
"Final source and scratch snapshots retain the scorer capture bounds and may be truncated.",
"No automatic gaming or intent labels are assigned."
]
}