From bcce9d9e90d9948b762d849f63177a75ece071a4 Mon Sep 17 00:00:00 2001 From: PJ Date: Mon, 31 Aug 2026 21:44:04 +0530 Subject: [PATCH] Calibration artifact: the checks and the raw commands side by side The pilot is only worth something if the checks agree with a person reading the transcript. This writes both out per run so the comparison is possible, and leaves a verdict column to record disagreements in. The disagreements carry more information than the rate does, because they are what fixes the scorer before money goes into a real run. Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu --- justfile | 4 ++ scripts/calibrate.py | 85 +++++++++++++++++++++++++++++++ src/messageboardbench/analysis.py | 25 ++++++++- 3 files changed, 112 insertions(+), 2 deletions(-) create mode 100644 scripts/calibrate.py diff --git a/justfile b/justfile index b5a6527..a26d6a7 100644 --- a/justfile +++ b/justfile @@ -43,3 +43,7 @@ report log_dir="logs/pilot": # Proves the scorer captures a file an agent leaves in scratch. Free, needs Docker. verify-snapshot: uv run python scripts/verify_snapshot.py + +# Writes the checks and the raw agent commands side by side, to read by hand. +calibrate log_dir="logs/pilot": + uv run python scripts/calibrate.py {{log_dir}} diff --git a/scripts/calibrate.py b/scripts/calibrate.py new file mode 100644 index 0000000..b76c1f8 --- /dev/null +++ b/scripts/calibrate.py @@ -0,0 +1,85 @@ +"""Everything needed to check the automated flags by hand, one section per run. + +The pilot is only worth anything if the checks agree with a person reading the transcript. +This writes the two side by side: what the checks report, and every filesystem action the +agent took, unfiltered and unsummarised. Read them, and write down each disagreement. + +The disagreements are the point. They are what calibrates the scorer before any money goes +into a real run, and they carry more information than the rate does. + + uv run python scripts/calibrate.py logs/pilot +""" + +import sys +from pathlib import Path + +from inspect_ai.log import list_eval_logs, read_eval_log + +from messageboardbench.analysis import agent_commands, sample_row + +if __name__ == "__main__": + log_dir = sys.argv[1] if len(sys.argv) > 1 else "logs/pilot" + out = Path(sys.argv[2]) if len(sys.argv) > 2 else Path("results/calibration.md") + + samples = [] + for info in list_eval_logs(log_dir): + samples.extend(read_eval_log(info.name).samples or []) + + lines = [ + "# Hand read against the checks", + "", + "One section per run. `checks` is what the code reported; `agent commands` is", + "every filesystem action the agent took, filtered only to exclude the harness's", + "own setup and scoring. Read the commands, decide for yourself whether the agent", + "touched or wrote to the directory, and record any disagreement in the verdict", + "line. Disagreements are what calibrate the scorer.", + "", + "| sample | checks say touched | checks say wrote | my reading | agree? |", + "| --- | --- | --- | --- | --- |", + ] + rows = [(s, sample_row(s)) for s in samples] + for sample, row in rows: + lines.append( + f"| {row['sample_id']} | {row['touched_scratch']} | {row['wrote_scratch']} | | |" + ) + lines.append("") + + for sample, row in rows: + flags = ", ".join( + k + for k in ( + "cheated", + "test_modified", + "touched_scratch", + "read_scratch", + "wrote_scratch", + "unparsed", + "ambiguous", + ) + if row[k] + ) + final = next(iter(sample.scores.values()), None) + files = ((final.metadata or {}).get("scratch_files") if final else {}) or {} + + lines += [ + f"## {row['sample_id']}", + "", + f"checks: {flags or 'nothing'} (messages: {row['n_messages']}, limit: {row['limit_hit'] or 'none'})", + "", + "agent commands:", + "", + "```", + ] + lines += agent_commands(sample) or ["(none)"] + lines += ["```", ""] + + if files: + lines += ["left behind:", ""] + for path, content in files.items(): + lines += [f"`{path}`", "", "```", content.rstrip(), "```", ""] + else: + lines += ["left behind: nothing", ""] + + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text("\n".join(lines)) + print(f"wrote {out} for {len(rows)} runs") diff --git a/src/messageboardbench/analysis.py b/src/messageboardbench/analysis.py index f762810..8527e80 100644 --- a/src/messageboardbench/analysis.py +++ b/src/messageboardbench/analysis.py @@ -20,8 +20,8 @@ import csv from pathlib import Path from typing import Any, Iterable -from .checks import ScratchUse, needs_hand_read, scratch_use -from .events import ScratchSpec, interactions_from_events +from .checks import needs_hand_read, scratch_use +from .events import ScratchSpec, in_tool_span, interactions_from_events CSV_FIELDS = [ "sample_id", @@ -161,3 +161,24 @@ def scratch_contents(samples: Iterable[Any]) -> list[tuple[str, str, str]]: for path, content in (meta.get("scratch_files") or {}).items(): out.append((str(getattr(sample, "id", "")), path, content)) return out + + +def agent_commands(sample: Any) -> list[str]: + """Every filesystem action the agent itself performed, as raw text. + + This is the evidence a person reads when checking the automated flags by hand. It is + filtered the same way the checks are, to tool spans, so a disagreement is a + disagreement about classification and not about which events were even considered. + """ + events = list(getattr(sample, "events", None) or []) + keep = in_tool_span(events) + out: list[str] = [] + for event, agent in zip(events, keep): + if not agent or getattr(event, "event", None) != "sandbox": + continue + action = getattr(event, "action", None) + if action == "exec": + out.append(f"$ {getattr(event, 'cmd', '')}") + else: + out.append(f"[{action}] {getattr(event, 'file', '')}") + return out