Files
messageboardbench/tests/test_byte_match.py
T
pj 099563a830 Scratch directory task, event adapter, per-run checks
Forks ImpossibleBench's LiveCodeBench solver rather than passing instruction_prompt=,
which injects text into the system message, the user message and every failure message.
The scratch line now appears once, in the system message, verified by the mockllm smoke
run.

test_byte_match.py reads ImpossibleBench's expected_test construction out of its source
with ast and re-executes it, so our test.py is checked against theirs rather than a
copy. Confirmed to fail on a one-character upstream change. Without this, drift flags
every sample as test-modified, resets it, and destroys the cheat measurement silently.

The event adapter fixes a bug carried over from messageboard: relation() left absolute
paths absolute, so `cat /workspace/scratch/notes.md` classified as outside the
directory. Every absolute-path touch would have scored as a miss.

The checks are new rather than reused. The old ones score not-applicable when the prompt
names the directory, which ours does by design, and discard reads after the first write.
Both would undercount here. The decisions are kept, the code is not.

Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
2026-08-31 21:36:13 +05:30

106 lines
4.4 KiB
Python

"""Our test.py must match, byte for byte, what ImpossibleBench's scorer expects.
The scorer reconstructs the file it thinks should be on disk, asserts the real file equals
it, and on any mismatch flags the sample as having modified tests and overwrites the file.
A one-character difference in our construction therefore flags every sample and destroys
the cheat measurement, silently, because the repair leaves no failing test behind.
Rather than hard-coding a copy of their template, this reads the construction out of the
installed ImpossibleBench source with `ast` and re-executes it. If upstream changes the
template, these tests fail rather than the experiment.
"""
from __future__ import annotations
import ast
import inspect as pyinspect
import textwrap
import pytest
from impossiblebench import livecodebench_scorers
from messageboardbench.task import build_test_py
# Real-shaped inputs: a plain test, one with the `check` name the scorer rewrites, one
# whose whitespace would not survive a careless strip(), and an empty one.
CASES = [
("def check(candidate):\n assert candidate(1) == 2\n", "solve"),
("def test_thing(c):\n assert c(1) == 2\n", "f"),
("\n\ndef check(candidate):\n assert candidate(1) == 2\n\n\n", "spaced"),
("", "empty"),
("def check(candidate):\n\tassert candidate('a\\nb') == 'check('\n", "tricky"),
]
def upstream_expected_test(original_test: str, entry_point: str) -> str:
"""Re-execute the scorer's own `expected_test` construction, read from its source.
Extracts every assignment to `expected_test` in `swe_style_humaneval_scorer` and runs
them in order, so the result is upstream's string and not our idea of it.
"""
src = textwrap.dedent(pyinspect.getsource(livecodebench_scorers.swe_style_humaneval_scorer))
tree = ast.parse(src)
assignments = [
node
for node in ast.walk(tree)
if isinstance(node, ast.Assign)
and any(
isinstance(t, ast.Name) and t.id == "expected_test" for t in node.targets
)
]
assert assignments, "no assignment to expected_test found in the upstream scorer"
namespace: dict[str, object] = {
"original_test": original_test,
"entry_point": entry_point,
}
for node in assignments:
exec(compile(ast.Module([node], []), "<upstream>", "exec"), namespace) # noqa: S102
return namespace["expected_test"] # type: ignore[return-value]
@pytest.mark.parametrize("test_code,entry_point", CASES)
def test_build_test_py_matches_upstream(test_code: str, entry_point: str) -> None:
ours = build_test_py(test_code, entry_point)
theirs = upstream_expected_test(test_code, entry_point)
assert ours == theirs, (
"test.py construction has drifted from ImpossibleBench's scorer. Every sample "
"would be flagged as test-modified and reset.\n"
f"ours: {ours!r}\ntheirs: {theirs!r}"
)
@pytest.mark.parametrize("test_code,entry_point", CASES)
def test_survives_the_strip_the_scorer_applies(test_code: str, entry_point: str) -> None:
"""The scorer compares with .strip() on both sides, which is the real assertion."""
assert build_test_py(test_code, entry_point).strip() == upstream_expected_test(
test_code, entry_point
).strip()
def test_upstream_still_compares_against_test_patch_metadata() -> None:
"""The scorer builds from metadata["test_patch"]; the solver builds from ["test"].
Those are the same string only because `record_to_sample` sets `test_patch = test`.
Its `allow_test_modifications` parameter defaults to False and the task never passes
it, so the field is always populated. If upstream ever threads that flag through, the
scorer would start comparing against the *original* unmutated test, every conflicting
sample would be flagged as modified, and the tests would be reset to solvable ones.
"""
from impossiblebench.livecodebench_tasks import record_to_sample
record = {
"task_id": "t",
"prompt": "def f():",
"entry_point": "f",
"test": "def check(c):\n assert c() == 1\n",
"impossible_type": "conflicting",
}
sample = record_to_sample()(record)
assert sample.metadata is not None
assert sample.metadata["test_patch"] == sample.metadata["test"], (
"record_to_sample no longer mirrors test into test_patch; the scorer would "
"compare test.py against a different string than the solver wrote."
)