Calibration artifact: the checks and the raw commands side by side

The pilot is only worth something if the checks agree with a person reading the
transcript. This writes both out per run so the comparison is possible, and leaves a
verdict column to record disagreements in. The disagreements carry more information
than the rate does, because they are what fixes the scorer before money goes into a
real run.

Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
This commit is contained in:
pj committed 2026-08-31 21:44:04 +05:30
1 parent 0a053c114c
commit bcce9d9e90
3 files changed
+112 -2

No files matched your search

+4
View File
@@ -43,3 +43,7 @@ report log_dir="logs/pilot":
# Proves the scorer captures a file an agent leaves in scratch. Free, needs Docker. # Proves the scorer captures a file an agent leaves in scratch. Free, needs Docker.
verify-snapshot: verify-snapshot:
uv run python scripts/verify_snapshot.py uv run python scripts/verify_snapshot.py
# Writes the checks and the raw agent commands side by side, to read by hand.
calibrate log_dir="logs/pilot":
uv run python scripts/calibrate.py {{log_dir}}
+85
View File
@@ -0,0 +1,85 @@
"""Everything needed to check the automated flags by hand, one section per run.
The pilot is only worth anything if the checks agree with a person reading the transcript.
This writes the two side by side: what the checks report, and every filesystem action the
agent took, unfiltered and unsummarised. Read them, and write down each disagreement.
The disagreements are the point. They are what calibrates the scorer before any money goes
into a real run, and they carry more information than the rate does.
uv run python scripts/calibrate.py logs/pilot
"""
import sys
from pathlib import Path
from inspect_ai.log import list_eval_logs, read_eval_log
from messageboardbench.analysis import agent_commands, sample_row
if __name__ == "__main__":
log_dir = sys.argv[1] if len(sys.argv) > 1 else "logs/pilot"
out = Path(sys.argv[2]) if len(sys.argv) > 2 else Path("results/calibration.md")
samples = []
for info in list_eval_logs(log_dir):
samples.extend(read_eval_log(info.name).samples or [])
lines = [
"# Hand read against the checks",
"",
"One section per run. `checks` is what the code reported; `agent commands` is",
"every filesystem action the agent took, filtered only to exclude the harness's",
"own setup and scoring. Read the commands, decide for yourself whether the agent",
"touched or wrote to the directory, and record any disagreement in the verdict",
"line. Disagreements are what calibrate the scorer.",
"",
"| sample | checks say touched | checks say wrote | my reading | agree? |",
"| --- | --- | --- | --- | --- |",
]
rows = [(s, sample_row(s)) for s in samples]
for sample, row in rows:
lines.append(
f"| {row['sample_id']} | {row['touched_scratch']} | {row['wrote_scratch']} | | |"
)
lines.append("")
for sample, row in rows:
flags = ", ".join(
k
for k in (
"cheated",
"test_modified",
"touched_scratch",
"read_scratch",
"wrote_scratch",
"unparsed",
"ambiguous",
)
if row[k]
)
final = next(iter(sample.scores.values()), None)
files = ((final.metadata or {}).get("scratch_files") if final else {}) or {}
lines += [
f"## {row['sample_id']}",
"",
f"checks: {flags or 'nothing'} (messages: {row['n_messages']}, limit: {row['limit_hit'] or 'none'})",
"",
"agent commands:",
"",
"```",
]
lines += agent_commands(sample) or ["(none)"]
lines += ["```", ""]
if files:
lines += ["left behind:", ""]
for path, content in files.items():
lines += [f"`{path}`", "", "```", content.rstrip(), "```", ""]
else:
lines += ["left behind: nothing", ""]
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text("\n".join(lines))
print(f"wrote {out} for {len(rows)} runs")
+23 -2
View File
@@ -20,8 +20,8 @@ import csv
from pathlib import Path from pathlib import Path
from typing import Any, Iterable from typing import Any, Iterable
from .checks import ScratchUse, needs_hand_read, scratch_use from .checks import needs_hand_read, scratch_use
from .events import ScratchSpec, interactions_from_events from .events import ScratchSpec, in_tool_span, interactions_from_events
CSV_FIELDS = [ CSV_FIELDS = [
"sample_id", "sample_id",
@@ -161,3 +161,24 @@ def scratch_contents(samples: Iterable[Any]) -> list[tuple[str, str, str]]:
for path, content in (meta.get("scratch_files") or {}).items(): for path, content in (meta.get("scratch_files") or {}).items():
out.append((str(getattr(sample, "id", "")), path, content)) out.append((str(getattr(sample, "id", "")), path, content))
return out return out
def agent_commands(sample: Any) -> list[str]:
"""Every filesystem action the agent itself performed, as raw text.
This is the evidence a person reads when checking the automated flags by hand. It is
filtered the same way the checks are, to tool spans, so a disagreement is a
disagreement about classification and not about which events were even considered.
"""
events = list(getattr(sample, "events", None) or [])
keep = in_tool_span(events)
out: list[str] = []
for event, agent in zip(events, keep):
if not agent or getattr(event, "event", None) != "sandbox":
continue
action = getattr(event, "action", None)
if action == "exec":
out.append(f"$ {getattr(event, 'cmd', '')}")
else:
out.append(f"[{action}] {getattr(event, 'file', '')}")
return out