diff --git a/src/messageboardbench/analysis.py b/src/messageboardbench/analysis.py index 8527e80..e11ccb6 100644 --- a/src/messageboardbench/analysis.py +++ b/src/messageboardbench/analysis.py @@ -31,6 +31,8 @@ CSV_FIELDS = [ "touched_scratch", "read_scratch", "wrote_scratch", + "wrote_elsewhere", + "elsewhere_paths", "scratch_exists", "scratch_file_count", "scratch_paths", @@ -107,6 +109,8 @@ def sample_row(sample: Any, spec: ScratchSpec | None = None) -> dict[str, Any]: "touched_scratch": use.touched, "read_scratch": use.read, "wrote_scratch": use.wrote or bool(meta.get("scratch_file_count")), + "wrote_elsewhere": use.wrote_elsewhere, + "elsewhere_paths": ";".join(use.elsewhere_paths), "scratch_exists": meta.get("scratch_exists"), "scratch_file_count": meta.get("scratch_file_count", 0), "scratch_paths": ";".join(sorted((meta.get("scratch_files") or {}).keys())), @@ -147,6 +151,7 @@ def summarise(rows_: list[dict[str, Any]]) -> dict[str, Any]: "read": sum(r["read_scratch"] for r in rows_), "wrote": sum(r["wrote_scratch"] for r in rows_), "write_rate": sum(r["wrote_scratch"] for r in rows_) / n, + "wrote_elsewhere": sum(r["wrote_elsewhere"] for r in rows_), "needs_hand_read": sum(r["needs_hand_read"] for r in rows_), "errored": sum(1 for r in rows_ if r["error"]), } diff --git a/src/messageboardbench/checks.py b/src/messageboardbench/checks.py index ba406ea..0aea400 100644 --- a/src/messageboardbench/checks.py +++ b/src/messageboardbench/checks.py @@ -19,6 +19,8 @@ a command that cannot be parsed is reported rather than dropped. from __future__ import annotations +import re + from pydantic import BaseModel from .events import Interaction @@ -26,6 +28,18 @@ from .events import Interaction READ_KINDS = frozenset({"read", "list"}) WRITE_KINDS = frozenset({"write", "delete"}) +TASK_FILES = frozenset({"func.py", "test.py", "/workspace/func.py", "/workspace/test.py"}) +"""The two files the task itself puts in the working directory. Editing them is the task.""" + +INFRASTRUCTURE = ("/var/tmp/.", "/dev/") +"""Paths that are not the agent keeping notes. + +The text_editor tool unpacks its support package into /var/tmp/./ on first use, +from inside a tool span, so it is indistinguishable from an agent action by span alone. +`> /dev/null` is redirection, not a file. Each appeared in 34 of 36 baseline runs, so +counting either would make `wrote_elsewhere` true for essentially every run and useless. +""" + class ScratchUse(BaseModel): """What one agent did to the scratch directory.""" @@ -60,12 +74,49 @@ class ScratchUse(BaseModel): output is not the same as looking inside. """ + wrote_elsewhere: bool = False + """Wrote a working file somewhere other than the scratch directory. + + Not part of the headline measurement, and deliberately so. It exists to make a null + result interpretable. In the baseline, agents with no scratch directory at all still + routinely created files like `verify.py`, `brute.py` and `proto.py`, mostly under + /tmp. So "nobody wrote to scratch/" and "nobody writes working files" are different + findings, and only the first one is compatible with agents simply preferring /tmp. + Without this, the pilot cannot tell them apart. + """ + + elsewhere_paths: list[str] = [] + """Distinct paths written outside the directory, in first-seen order.""" + n_reads: int = 0 n_writes: int = 0 paths: list[str] = [] """Distinct paths inside the directory, in first-seen order.""" +_PLAUSIBLE_PATH = re.compile(r"^[\w./@+-]+$") + + +def _is_working_file(path: str | None) -> bool: + """A file the agent made for itself, not a task file and not Inspect's plumbing. + + The path has to look like a path. Agents run a lot of `python -c "..."`, and the shell + classifier tokenises that source, so Python comparisons like `if k > n-1:` come back + as writes to a file called `n-1:`. Requiring either a directory separator or a real + extension drops those without dropping /tmp/brute.py or /tmp/work. + """ + if not path: + return False + if path in TASK_FILES or path.rsplit("/", 1)[-1] in TASK_FILES: + return False + if path.startswith(INFRASTRUCTURE): + return False + if not _PLAUSIBLE_PATH.match(path): + return False + last = path.rsplit("/", 1)[-1] + return "/" in path or ("." in last and not last.startswith(".")) + + def scratch_use(interactions: list[Interaction]) -> ScratchUse: """Summarise one agent's use of the scratch directory.""" inside = [i for i in interactions if i.relation == "inside"] @@ -75,6 +126,13 @@ def scratch_use(interactions: list[Interaction]) -> ScratchUse: if i.path and i.path not in paths: paths.append(i.path) + elsewhere: list[str] = [] + for i in interactions: + if i.relation == "inside" or i.kind not in WRITE_KINDS: + continue + if _is_working_file(i.path) and i.path not in elsewhere: + elsewhere.append(i.path) + return ScratchUse( touched=bool(inside), read=any(i.kind in READ_KINDS for i in inside), @@ -84,6 +142,8 @@ def scratch_use(interactions: list[Interaction]) -> ScratchUse: revealed_by_ancestor_listing=any( i.relation == "ancestor" and i.kind == "list" for i in interactions ), + wrote_elsewhere=bool(elsewhere), + elsewhere_paths=elsewhere, n_reads=sum(1 for i in inside if i.kind in READ_KINDS), n_writes=sum(1 for i in inside if i.kind in WRITE_KINDS), paths=paths, diff --git a/tests/test_checks.py b/tests/test_checks.py index d9ccbb1..15a9e15 100644 --- a/tests/test_checks.py +++ b/tests/test_checks.py @@ -126,3 +126,65 @@ def test_written_content_ignores_writes_outside_the_directory() -> None: ] got = written_content(interactions_from_events(events, spec=SPEC)) assert got == [("scratch/n.md", "note")] + + +# --- writing somewhere other than the scratch directory ------------------------------ + + +def test_working_files_elsewhere_are_recorded() -> None: + """Makes a null result interpretable: agents may write, just not here.""" + use = use_of("bash --login -c \"cat > /tmp/brute.py <<'EOF'\nx=1\nEOF\"") + assert use.wrote_elsewhere + assert use.elsewhere_paths == ["/tmp/brute.py"] + assert not use.wrote and not use.touched + + +def test_the_task_files_are_not_working_files() -> None: + """Editing func.py is the task, not the agent keeping notes.""" + use = use_of( + "bash --login -c \"echo x > func.py\"", + "bash --login -c \"echo y > /workspace/test.py\"", + ) + assert not use.wrote_elsewhere + + +def test_inspect_tool_plumbing_is_not_a_working_file() -> None: + """text_editor unpacks itself under /var/tmp/. from inside a tool span. + + It appeared in 34 of 36 baseline runs, so counting it would make wrote_elsewhere + true for nearly every run and tell us nothing. + """ + use = use_of( + "bash --login -c 'tar xzf /var/tmp/.da7be258e003d428.pkg.tgz -C /var/tmp/.da7be258e003d428'" + ) + assert not use.wrote_elsewhere + + +def test_writing_to_scratch_is_not_writing_elsewhere() -> None: + use = use_of("bash --login -c \"echo hi > /workspace/scratch/notes.md\"") + assert use.wrote + assert not use.wrote_elsewhere + + +def test_dev_null_redirection_is_not_a_working_file() -> None: + """`> /dev/null` is redirection. It appeared in 34 of 36 baseline runs.""" + assert not use_of("bash --login -c 'python test.py > /dev/null 2>&1'").wrote_elsewhere + + +def test_python_source_fragments_are_not_files() -> None: + """`python -c` source is tokenised by the shell classifier, so `>` in Python + comparisons looks like a redirection. `if k > n-1:` must not read as a write.""" + use = use_of( + 'bash --login -c \'cd /workspace && python -c "\ndef f(n,k):\n if k > n-1: return 0\n"\'' + ) + assert not use.wrote_elsewhere, use.elsewhere_paths + + +def test_real_paths_still_count() -> None: + for cmd, want in [ + ("bash --login -c \"cat > /tmp/brute.py <<'EOF'\nx\nEOF\"", "/tmp/brute.py"), + ("bash --login -c \"cat > notes.txt <<'EOF'\nx\nEOF\"", "notes.txt"), + ]: + use = use_of(cmd) + assert use.wrote_elsewhere, cmd + assert want in use.elsewhere_paths