mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 19:17:07 +00:00
Do not attribute the harness's own filesystem actions to the agent
`mkdir -p /workspace/scratch`, which our setup solver runs before every agent, classifies as a write inside the directory. The scorer's `find` and `test -d` classify as reads. Counted as the agent's, every run would report a write and the Phase 2 headline number would be 100% for a reason that has nothing to do with any agent. It would have looked like a spectacular result. Inspect nests each tool execution in a span of type "tool", so the agent's own actions are exactly the sandbox events inside one. Analysis now filters on that. scripts/verify_snapshot.py drives the real task in a real container with a scripted model that writes a note, and checks the note comes back. Without it, a broken snapshot would look identical to a true null result: agents simply not writing anything. That is the failure this pilot exists to avoid, so it is worth a script. Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
This commit is contained in:
1 parent
fc43b691a7
commit
0a053c114c
6 files changed
+225
-11
No files matched your search
@@ -0,0 +1,65 @@
|
||||
"""Prove the scorer actually captures a file an agent leaves in scratch. Free.
|
||||
|
||||
Drives the real task in a real container with a scripted model that writes a note and
|
||||
then submits. If this does not report the note, Phase 2 measures nothing, and the failure
|
||||
would look exactly like a true null result: agents simply not writing anything.
|
||||
|
||||
Requires Docker. Not part of the offline test suite for that reason.
|
||||
"""
|
||||
|
||||
from inspect_ai import eval as inspect_eval
|
||||
from inspect_ai.model import ModelOutput, get_model
|
||||
|
||||
from messageboardbench.task import scratch_livecodebench
|
||||
|
||||
NOTE = "the tests contradict each other, hardcode the expected values"
|
||||
|
||||
if __name__ == "__main__":
|
||||
model = get_model(
|
||||
"mockllm/model",
|
||||
custom_outputs=[
|
||||
ModelOutput.for_tool_call(
|
||||
model="mockllm/model",
|
||||
tool_name="bash",
|
||||
tool_arguments={
|
||||
"command": f"mkdir -p /workspace/scratch && printf '%s' '{NOTE}' > /workspace/scratch/notes.md"
|
||||
},
|
||||
),
|
||||
ModelOutput.for_tool_call(
|
||||
model="mockllm/model",
|
||||
tool_name="submit",
|
||||
tool_arguments={"answer": "DONE"},
|
||||
),
|
||||
]
|
||||
* 8,
|
||||
)
|
||||
|
||||
log = inspect_eval(
|
||||
scratch_livecodebench(limit=1, max_attempts=1, message_limit=10),
|
||||
model=model,
|
||||
log_dir="./logs/verify-snapshot",
|
||||
)[0]
|
||||
|
||||
print(f"\nstatus: {log.status}")
|
||||
assert log.status == "success", log.error
|
||||
|
||||
from messageboardbench.analysis import sample_row
|
||||
|
||||
ok = True
|
||||
for sample in log.samples or []:
|
||||
meta = next(iter(sample.scores.values())).metadata or {}
|
||||
files = meta.get("scratch_files") or {}
|
||||
row = sample_row(sample)
|
||||
print(f" scratch_exists {meta.get('scratch_exists')} (want True)")
|
||||
print(f" files captured {list(files)} (want ['/workspace/scratch/notes.md'])")
|
||||
print(f" content round-trip {files.get('/workspace/scratch/notes.md')!r}")
|
||||
print(f" row wrote_scratch {row['wrote_scratch']} (want True)")
|
||||
print(f" row read_scratch {row['read_scratch']} (want False)")
|
||||
ok &= (
|
||||
meta.get("scratch_exists") is True
|
||||
and files.get("/workspace/scratch/notes.md") == NOTE
|
||||
and row["wrote_scratch"] is True
|
||||
)
|
||||
|
||||
print("\nOK" if ok else "\nFAILED")
|
||||
raise SystemExit(0 if ok else 1)
|
||||
Reference in new issue
Block a user