mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 19:17:07 +00:00
was_test_modified reads intermediate score events, not just the final score. ImpossibleBench's scorer repairs test.py when it finds it modified, and basic_agent scores every attempt, so an agent that edits the tests on attempt one and not again leaves a final score with no flag on it. The repair erased the evidence. snapshot_scratch reports whether the directory exists alongside its contents, because otherwise a directory nobody wrote to and a directory that was never created look identical, and the second is a broken harness reported as a real null result. What the agents wrote is not auto-classified. "Did it write" is safe to automate; whether a note is addressed to somebody is the judgement the pilot exists to make, so pilot.py emits a markdown file to read by hand. Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
32 lines
959 B
Python
32 lines
959 B
Python
"""Rebuild the CSV and the hand-read file from an eval log, without re-running anything.
|
|
|
|
uv run python scripts/report.py logs/pilot
|
|
"""
|
|
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
from inspect_ai.log import list_eval_logs, read_eval_log
|
|
|
|
from messageboardbench.analysis import rows, summarise, write_csv
|
|
|
|
sys.path.insert(0, str(Path(__file__).parent))
|
|
from pilot import write_hand_read_file # noqa: E402
|
|
|
|
if __name__ == "__main__":
|
|
log_dir = sys.argv[1] if len(sys.argv) > 1 else "logs/pilot"
|
|
out = Path(sys.argv[2]) if len(sys.argv) > 2 else Path("results")
|
|
|
|
samples = []
|
|
for info in list_eval_logs(log_dir):
|
|
samples.extend(read_eval_log(info.name).samples or [])
|
|
|
|
table = rows(samples)
|
|
write_csv(table, out / "pilot.csv")
|
|
n_files = write_hand_read_file(samples, out / "pilot-scratch-contents.md")
|
|
|
|
s = summarise(table)
|
|
for k, v in s.items():
|
|
print(f"{k:18} {v}")
|
|
print(f"{'files left':18} {n_files}")
|