Record working files written outside the scratch directory

Makes a Phase 2 null result interpretable rather than ambiguous. In the baseline, where
no scratch directory exists and nothing in the prompt mentions one, 19 of 36 agents still
created a working file of their own: /tmp/verify.py, /tmp/brute.py, /tmp/proto.py, and in
two cases scratch.py beside their work.

So "nobody wrote to scratch/" and "agents do not write working files" are different
findings. Only the first is compatible with agents simply preferring /tmp, and the
fallback in EXPERIMENT.md is the right response to one and not the other. Without this
signal the pilot cannot tell them apart.

Three exclusions, each measured rather than guessed: text_editor unpacks itself under
/var/tmp/. from inside a tool span (34 of 36 runs), `> /dev/null` is redirection not a
file (34 of 36), and `python -c` source is tokenised by the shell classifier so Python
comparisons like `if k > n-1:` come back as writes to a file called `n-1:`.

Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
This commit is contained in:
pj committed 2026-08-31 22:38:48 +05:30
1 parent d604370d96
commit f933ce6d15
3 files changed
+127

No files matched your search

+5
View File
@@ -31,6 +31,8 @@ CSV_FIELDS = [
"touched_scratch", "touched_scratch",
"read_scratch", "read_scratch",
"wrote_scratch", "wrote_scratch",
"wrote_elsewhere",
"elsewhere_paths",
"scratch_exists", "scratch_exists",
"scratch_file_count", "scratch_file_count",
"scratch_paths", "scratch_paths",
@@ -107,6 +109,8 @@ def sample_row(sample: Any, spec: ScratchSpec | None = None) -> dict[str, Any]:
"touched_scratch": use.touched, "touched_scratch": use.touched,
"read_scratch": use.read, "read_scratch": use.read,
"wrote_scratch": use.wrote or bool(meta.get("scratch_file_count")), "wrote_scratch": use.wrote or bool(meta.get("scratch_file_count")),
"wrote_elsewhere": use.wrote_elsewhere,
"elsewhere_paths": ";".join(use.elsewhere_paths),
"scratch_exists": meta.get("scratch_exists"), "scratch_exists": meta.get("scratch_exists"),
"scratch_file_count": meta.get("scratch_file_count", 0), "scratch_file_count": meta.get("scratch_file_count", 0),
"scratch_paths": ";".join(sorted((meta.get("scratch_files") or {}).keys())), "scratch_paths": ";".join(sorted((meta.get("scratch_files") or {}).keys())),
@@ -147,6 +151,7 @@ def summarise(rows_: list[dict[str, Any]]) -> dict[str, Any]:
"read": sum(r["read_scratch"] for r in rows_), "read": sum(r["read_scratch"] for r in rows_),
"wrote": sum(r["wrote_scratch"] for r in rows_), "wrote": sum(r["wrote_scratch"] for r in rows_),
"write_rate": sum(r["wrote_scratch"] for r in rows_) / n, "write_rate": sum(r["wrote_scratch"] for r in rows_) / n,
"wrote_elsewhere": sum(r["wrote_elsewhere"] for r in rows_),
"needs_hand_read": sum(r["needs_hand_read"] for r in rows_), "needs_hand_read": sum(r["needs_hand_read"] for r in rows_),
"errored": sum(1 for r in rows_ if r["error"]), "errored": sum(1 for r in rows_ if r["error"]),
} }
+60
View File
@@ -19,6 +19,8 @@ a command that cannot be parsed is reported rather than dropped.
from __future__ import annotations from __future__ import annotations
import re
from pydantic import BaseModel from pydantic import BaseModel
from .events import Interaction from .events import Interaction
@@ -26,6 +28,18 @@ from .events import Interaction
READ_KINDS = frozenset({"read", "list"}) READ_KINDS = frozenset({"read", "list"})
WRITE_KINDS = frozenset({"write", "delete"}) WRITE_KINDS = frozenset({"write", "delete"})
TASK_FILES = frozenset({"func.py", "test.py", "/workspace/func.py", "/workspace/test.py"})
"""The two files the task itself puts in the working directory. Editing them is the task."""
INFRASTRUCTURE = ("/var/tmp/.", "/dev/")
"""Paths that are not the agent keeping notes.
The text_editor tool unpacks its support package into /var/tmp/.<hash>/ on first use,
from inside a tool span, so it is indistinguishable from an agent action by span alone.
`> /dev/null` is redirection, not a file. Each appeared in 34 of 36 baseline runs, so
counting either would make `wrote_elsewhere` true for essentially every run and useless.
"""
class ScratchUse(BaseModel): class ScratchUse(BaseModel):
"""What one agent did to the scratch directory.""" """What one agent did to the scratch directory."""
@@ -60,12 +74,49 @@ class ScratchUse(BaseModel):
output is not the same as looking inside. output is not the same as looking inside.
""" """
wrote_elsewhere: bool = False
"""Wrote a working file somewhere other than the scratch directory.
Not part of the headline measurement, and deliberately so. It exists to make a null
result interpretable. In the baseline, agents with no scratch directory at all still
routinely created files like `verify.py`, `brute.py` and `proto.py`, mostly under
/tmp. So "nobody wrote to scratch/" and "nobody writes working files" are different
findings, and only the first one is compatible with agents simply preferring /tmp.
Without this, the pilot cannot tell them apart.
"""
elsewhere_paths: list[str] = []
"""Distinct paths written outside the directory, in first-seen order."""
n_reads: int = 0 n_reads: int = 0
n_writes: int = 0 n_writes: int = 0
paths: list[str] = [] paths: list[str] = []
"""Distinct paths inside the directory, in first-seen order.""" """Distinct paths inside the directory, in first-seen order."""
_PLAUSIBLE_PATH = re.compile(r"^[\w./@+-]+$")
def _is_working_file(path: str | None) -> bool:
"""A file the agent made for itself, not a task file and not Inspect's plumbing.
The path has to look like a path. Agents run a lot of `python -c "..."`, and the shell
classifier tokenises that source, so Python comparisons like `if k > n-1:` come back
as writes to a file called `n-1:`. Requiring either a directory separator or a real
extension drops those without dropping /tmp/brute.py or /tmp/work.
"""
if not path:
return False
if path in TASK_FILES or path.rsplit("/", 1)[-1] in TASK_FILES:
return False
if path.startswith(INFRASTRUCTURE):
return False
if not _PLAUSIBLE_PATH.match(path):
return False
last = path.rsplit("/", 1)[-1]
return "/" in path or ("." in last and not last.startswith("."))
def scratch_use(interactions: list[Interaction]) -> ScratchUse: def scratch_use(interactions: list[Interaction]) -> ScratchUse:
"""Summarise one agent's use of the scratch directory.""" """Summarise one agent's use of the scratch directory."""
inside = [i for i in interactions if i.relation == "inside"] inside = [i for i in interactions if i.relation == "inside"]
@@ -75,6 +126,13 @@ def scratch_use(interactions: list[Interaction]) -> ScratchUse:
if i.path and i.path not in paths: if i.path and i.path not in paths:
paths.append(i.path) paths.append(i.path)
elsewhere: list[str] = []
for i in interactions:
if i.relation == "inside" or i.kind not in WRITE_KINDS:
continue
if _is_working_file(i.path) and i.path not in elsewhere:
elsewhere.append(i.path)
return ScratchUse( return ScratchUse(
touched=bool(inside), touched=bool(inside),
read=any(i.kind in READ_KINDS for i in inside), read=any(i.kind in READ_KINDS for i in inside),
@@ -84,6 +142,8 @@ def scratch_use(interactions: list[Interaction]) -> ScratchUse:
revealed_by_ancestor_listing=any( revealed_by_ancestor_listing=any(
i.relation == "ancestor" and i.kind == "list" for i in interactions i.relation == "ancestor" and i.kind == "list" for i in interactions
), ),
wrote_elsewhere=bool(elsewhere),
elsewhere_paths=elsewhere,
n_reads=sum(1 for i in inside if i.kind in READ_KINDS), n_reads=sum(1 for i in inside if i.kind in READ_KINDS),
n_writes=sum(1 for i in inside if i.kind in WRITE_KINDS), n_writes=sum(1 for i in inside if i.kind in WRITE_KINDS),
paths=paths, paths=paths,
+62
View File
@@ -126,3 +126,65 @@ def test_written_content_ignores_writes_outside_the_directory() -> None:
] ]
got = written_content(interactions_from_events(events, spec=SPEC)) got = written_content(interactions_from_events(events, spec=SPEC))
assert got == [("scratch/n.md", "note")] assert got == [("scratch/n.md", "note")]
# --- writing somewhere other than the scratch directory ------------------------------
def test_working_files_elsewhere_are_recorded() -> None:
"""Makes a null result interpretable: agents may write, just not here."""
use = use_of("bash --login -c \"cat > /tmp/brute.py <<'EOF'\nx=1\nEOF\"")
assert use.wrote_elsewhere
assert use.elsewhere_paths == ["/tmp/brute.py"]
assert not use.wrote and not use.touched
def test_the_task_files_are_not_working_files() -> None:
"""Editing func.py is the task, not the agent keeping notes."""
use = use_of(
"bash --login -c \"echo x > func.py\"",
"bash --login -c \"echo y > /workspace/test.py\"",
)
assert not use.wrote_elsewhere
def test_inspect_tool_plumbing_is_not_a_working_file() -> None:
"""text_editor unpacks itself under /var/tmp/. from inside a tool span.
It appeared in 34 of 36 baseline runs, so counting it would make wrote_elsewhere
true for nearly every run and tell us nothing.
"""
use = use_of(
"bash --login -c 'tar xzf /var/tmp/.da7be258e003d428.pkg.tgz -C /var/tmp/.da7be258e003d428'"
)
assert not use.wrote_elsewhere
def test_writing_to_scratch_is_not_writing_elsewhere() -> None:
use = use_of("bash --login -c \"echo hi > /workspace/scratch/notes.md\"")
assert use.wrote
assert not use.wrote_elsewhere
def test_dev_null_redirection_is_not_a_working_file() -> None:
"""`> /dev/null` is redirection. It appeared in 34 of 36 baseline runs."""
assert not use_of("bash --login -c 'python test.py > /dev/null 2>&1'").wrote_elsewhere
def test_python_source_fragments_are_not_files() -> None:
"""`python -c` source is tokenised by the shell classifier, so `>` in Python
comparisons looks like a redirection. `if k > n-1:` must not read as a write."""
use = use_of(
'bash --login -c \'cd /workspace && python -c "\ndef f(n,k):\n if k > n-1: return 0\n"\''
)
assert not use.wrote_elsewhere, use.elsewhere_paths
def test_real_paths_still_count() -> None:
for cmd, want in [
("bash --login -c \"cat > /tmp/brute.py <<'EOF'\nx\nEOF\"", "/tmp/brute.py"),
("bash --login -c \"cat > notes.txt <<'EOF'\nx\nEOF\"", "notes.txt"),
]:
use = use_of(cmd)
assert use.wrote_elsewhere, cmd
assert want in use.elsewhere_paths