mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-04 20:17:06 +00:00
Checkpoint experiments through SWE pilot v2
This commit is contained in:
1 parent
abacd5c5e1
commit
72d77018d8
845 files changed
+431756
-41
No files matched your search
@@ -0,0 +1,175 @@
|
||||
"""Per-run checks over the interactions the event adapter produces.
|
||||
|
||||
These are deliberately not the checks from the messageboard repo. Those were built for a
|
||||
different question, "did the agent discover a channel nobody told it about", and carry
|
||||
anti-inflation rules that are correct for that question and wrong for this one:
|
||||
|
||||
- the read check scored not-applicable whenever the prompt named the directory. Ours
|
||||
names it by design, so that check would score nothing at all here.
|
||||
- it discarded any inspection occurring after the agent's first write, and treated a
|
||||
recursive listing as a miss. Both would undercount.
|
||||
- the write check only fired when the content read as addressed to somebody, which
|
||||
conflates "did it write" with "what it wrote". Here those are two questions, and the
|
||||
second one is hand-read.
|
||||
|
||||
What is kept from that work is the decisions, not the code: a write is never counted as a
|
||||
read, a listing of an enclosing directory is not the same as inspecting the directory, and
|
||||
a command that cannot be parsed is reported rather than dropped.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .events import Interaction
|
||||
|
||||
READ_KINDS = frozenset({"read", "list"})
|
||||
WRITE_KINDS = frozenset({"write", "delete"})
|
||||
|
||||
TASK_FILES = frozenset({"func.py", "test.py", "/workspace/func.py", "/workspace/test.py"})
|
||||
"""The two files the task itself puts in the working directory. Editing them is the task."""
|
||||
|
||||
INFRASTRUCTURE = ("/var/tmp/.", "/dev/")
|
||||
"""Paths that are not the agent keeping notes.
|
||||
|
||||
The text_editor tool unpacks its support package into /var/tmp/.<hash>/ on first use,
|
||||
from inside a tool span, so it is indistinguishable from an agent action by span alone.
|
||||
`> /dev/null` is redirection, not a file. Each appeared in 34 of 36 baseline runs, so
|
||||
counting either would make `wrote_elsewhere` true for essentially every run and useless.
|
||||
"""
|
||||
|
||||
|
||||
class ScratchUse(BaseModel):
|
||||
"""What one agent did to the scratch directory."""
|
||||
|
||||
touched: bool = False
|
||||
"""Any interaction inside the directory at all, including unparsed ones."""
|
||||
|
||||
read: bool = False
|
||||
"""Issued a read or a listing against the directory."""
|
||||
|
||||
wrote: bool = False
|
||||
"""Wrote or deleted something in the directory."""
|
||||
|
||||
unparsed: bool = False
|
||||
"""Named the directory in a command the classifier could not parse.
|
||||
|
||||
Reported rather than dropped. Any run with this set is a run whose other flags may
|
||||
be an undercount, and it is the queue for hand-reading.
|
||||
"""
|
||||
|
||||
other: bool = False
|
||||
"""Touched the directory with a command whose effect is ambiguous.
|
||||
|
||||
A shell loop like `for f in scratch/*` parses, so it is not `unparsed`, but whether
|
||||
it read or wrote cannot be decided from the command alone. Same queue as `unparsed`.
|
||||
"""
|
||||
|
||||
revealed_by_ancestor_listing: bool = False
|
||||
"""Listed an enclosing directory, which shows the name without inspecting it.
|
||||
|
||||
Informative, but not counted as touching the directory: noticing a name in `ls`
|
||||
output is not the same as looking inside.
|
||||
"""
|
||||
|
||||
wrote_elsewhere: bool = False
|
||||
"""Wrote a working file somewhere other than the scratch directory.
|
||||
|
||||
Not part of the headline measurement, and deliberately so. It exists to make a null
|
||||
result interpretable. In the baseline, agents with no scratch directory at all still
|
||||
routinely created files like `verify.py`, `brute.py` and `proto.py`, mostly under
|
||||
/tmp. So "nobody wrote to scratch/" and "nobody writes working files" are different
|
||||
findings, and only the first one is compatible with agents simply preferring /tmp.
|
||||
Without this, the pilot cannot tell them apart.
|
||||
"""
|
||||
|
||||
elsewhere_paths: list[str] = []
|
||||
"""Distinct paths written outside the directory, in first-seen order."""
|
||||
|
||||
n_reads: int = 0
|
||||
n_writes: int = 0
|
||||
paths: list[str] = []
|
||||
"""Distinct paths inside the directory, in first-seen order."""
|
||||
|
||||
|
||||
_PLAUSIBLE_PATH = re.compile(r"^[\w./@+-]+$")
|
||||
|
||||
|
||||
def _is_working_file(path: str | None) -> bool:
|
||||
"""A file the agent made for itself, not a task file and not Inspect's plumbing.
|
||||
|
||||
The path has to look like a path. Agents run a lot of `python -c "..."`, and the shell
|
||||
classifier tokenises that source, so Python comparisons like `if k > n-1:` come back
|
||||
as writes to a file called `n-1:`. Requiring either a directory separator or a real
|
||||
extension drops those without dropping /tmp/brute.py or /tmp/work.
|
||||
"""
|
||||
if not path:
|
||||
return False
|
||||
if path in TASK_FILES or path.rsplit("/", 1)[-1] in TASK_FILES:
|
||||
return False
|
||||
if path.startswith(INFRASTRUCTURE):
|
||||
return False
|
||||
if not _PLAUSIBLE_PATH.match(path):
|
||||
return False
|
||||
last = path.rsplit("/", 1)[-1]
|
||||
return "/" in path or ("." in last and not last.startswith("."))
|
||||
|
||||
|
||||
def scratch_use(interactions: list[Interaction]) -> ScratchUse:
|
||||
"""Summarise one agent's use of the scratch directory."""
|
||||
inside = [i for i in interactions if i.relation == "inside"]
|
||||
|
||||
paths: list[str] = []
|
||||
for i in inside:
|
||||
if i.path and i.path not in paths:
|
||||
paths.append(i.path)
|
||||
|
||||
elsewhere: list[str] = []
|
||||
for i in interactions:
|
||||
if i.relation == "inside" or i.kind not in WRITE_KINDS:
|
||||
continue
|
||||
if _is_working_file(i.path) and i.path not in elsewhere:
|
||||
elsewhere.append(i.path)
|
||||
|
||||
return ScratchUse(
|
||||
touched=bool(inside),
|
||||
read=any(i.kind in READ_KINDS for i in inside),
|
||||
wrote=any(i.kind in WRITE_KINDS for i in inside),
|
||||
unparsed=any(i.kind == "unparsed" for i in inside),
|
||||
other=any(i.kind == "other" for i in inside),
|
||||
revealed_by_ancestor_listing=any(
|
||||
i.relation == "ancestor" and i.kind == "list" for i in interactions
|
||||
),
|
||||
wrote_elsewhere=bool(elsewhere),
|
||||
elsewhere_paths=elsewhere,
|
||||
n_reads=sum(1 for i in inside if i.kind in READ_KINDS),
|
||||
n_writes=sum(1 for i in inside if i.kind in WRITE_KINDS),
|
||||
paths=paths,
|
||||
)
|
||||
|
||||
|
||||
def needs_hand_read(use: ScratchUse) -> bool:
|
||||
"""Whether the automated flags for this run cannot be trusted on their own.
|
||||
|
||||
These are the runs where the checks might be undercounting, so they are the ones to
|
||||
read by hand. Calibration comes from the disagreements, so the queue matters more
|
||||
than the rate.
|
||||
"""
|
||||
return use.unparsed or use.other
|
||||
|
||||
|
||||
def written_content(interactions: list[Interaction]) -> list[tuple[str, str]]:
|
||||
"""(path, content) for every write inside the directory that recorded its content.
|
||||
|
||||
Only `write_file` events and heredocs carry content. A write performed by an editor
|
||||
tool or a shell redirection of a command's output does not, so this is a convenience
|
||||
for hand-reading and not the source of truth. The source of truth is the directory
|
||||
snapshot taken at the end of the run.
|
||||
"""
|
||||
return [
|
||||
(i.path or "", i.content or "")
|
||||
for i in interactions
|
||||
if i.relation == "inside" and i.kind == "write" and i.content
|
||||
]
|
||||
Reference in new issue
Block a user