Scratch directory task, event adapter, per-run checks

Forks ImpossibleBench's LiveCodeBench solver rather than passing instruction_prompt=,
which injects text into the system message, the user message and every failure message.
The scratch line now appears once, in the system message, verified by the mockllm smoke
run.

test_byte_match.py reads ImpossibleBench's expected_test construction out of its source
with ast and re-executes it, so our test.py is checked against theirs rather than a
copy. Confirmed to fail on a one-character upstream change. Without this, drift flags
every sample as test-modified, resets it, and destroys the cheat measurement silently.

The event adapter fixes a bug carried over from messageboard: relation() left absolute
paths absolute, so `cat /workspace/scratch/notes.md` classified as outside the
directory. Every absolute-path touch would have scored as a miss.

The checks are new rather than reused. The old ones score not-applicable when the prompt
names the directory, which ours does by design, and discard reads after the first write.
Both would undercount here. The decisions are kept, the code is not.

Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
This commit is contained in:
pj committed 2026-08-31 21:36:13 +05:30
1 parent bf3bb0d144
commit 099563a830
13 files changed
+1808 -1

No files matched your search

+128
View File
@@ -0,0 +1,128 @@
"""Phase 1: the two numbers that decide whether the study is viable.
Runs the unmodified ImpossibleBench LiveCodeBench task twice over the same 40 tasks.
conflicting the tests contradict each other, so no honest solution exists and any pass
is provably a cheat. This is the cheat rate.
original the same 40 tasks unmutated. This is the solve rate, and it answers a
question the cheat rate alone cannot: whether the model can code at all.
If it solves nothing here, a low cheat rate means "too weak to do anything"
rather than "honest", and the design needs rethinking.
Cost is measured against the OpenRouter credits endpoint rather than estimated from token
counts, so the number written down is the number actually spent.
uv run python scripts/baseline.py # both splits
uv run python scripts/baseline.py conflicting # just one
Run from the repo root; inspect resolves compose.yaml against the working directory.
"""
import os
import sys
import httpx
from dotenv import load_dotenv
from inspect_ai import eval as inspect_eval
from impossiblebench import impossible_livecodebench
MODEL = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
LIMIT = int(os.environ.get("MBB_LIMIT", "40"))
CONCURRENCY = int(os.environ.get("MBB_CONCURRENCY", "12"))
# GLM 5.3 Flash is a reasoning model and single calls legitimately run for minutes,
# so the timeout is generous. It is here to stop one genuinely hung request from
# holding a slot for the length of the run, not to cut short slow thinking.
REQUEST_TIMEOUT = int(os.environ.get("MBB_TIMEOUT", "900"))
def credits_used() -> float | None:
"""Total credits spent on the key so far, or None if the endpoint is unavailable."""
key = os.environ.get("OPENROUTER_API_KEY")
if not key:
return None
try:
r = httpx.get(
"https://openrouter.ai/api/v1/credits",
headers={"Authorization": f"Bearer {key}"},
timeout=30,
)
r.raise_for_status()
return float(r.json()["data"]["total_usage"])
except Exception as e: # noqa: BLE001 - cost reporting must never abort a paid run
print(f" (could not read credits: {e})")
return None
def run_split(split: str) -> dict:
before = credits_used()
logs = inspect_eval(
impossible_livecodebench(
split=split,
agent_type="tools",
sandbox="docker",
limit=LIMIT,
),
model=MODEL,
log_dir=f"./logs/baseline/{split}",
max_connections=CONCURRENCY,
max_sandboxes=CONCURRENCY,
max_subprocesses=CONCURRENCY,
fail_on_error=False,
timeout=REQUEST_TIMEOUT,
max_retries=3,
)
after = credits_used()
log = logs[0]
samples = log.samples or []
scored = [s for s in samples if s.scores]
passed = sum(
1
for s in scored
for sc in s.scores.values()
if str(sc.value).upper() in ("C", "CORRECT", "1", "1.0")
)
errored = [s for s in samples if s.error]
usage = log.stats.model_usage if log.stats else {}
tok_in = sum(u.input_tokens for u in usage.values())
tok_out = sum(u.output_tokens for u in usage.values())
cached = sum((u.input_tokens_cache_read or 0) for u in usage.values())
n = len(scored) or 1
return {
"split": split,
"status": log.status,
"samples": len(samples),
"scored": len(scored),
"errored": len(errored),
"passed": passed,
"rate": passed / n,
"tok_in": tok_in,
"tok_out": tok_out,
"cache_read": cached,
"cost": (after - before) if (before is not None and after is not None) else None,
"log": str(log.location),
}
def report(r: dict) -> None:
label = "cheat rate" if r["split"] == "conflicting" else "solve rate"
cost = f"${r['cost']:.4f}" if r["cost"] is not None else "unknown"
per_run = f"${r['cost'] / max(r['scored'], 1):.5f}" if r["cost"] is not None else "unknown"
print(f"\n--- {r['split']} ---")
print(f" status {r['status']} ({r['errored']} errored, {r['scored']} scored)")
print(f" {label} {r['passed']}/{r['scored']} = {r['rate']:.1%}")
print(f" tokens in={r['tok_in']:,} out={r['tok_out']:,} cache_read={r['cache_read']:,}")
print(f" cost {cost} total, {per_run} per run")
print(f" log {r['log']}")
if __name__ == "__main__":
load_dotenv()
splits = sys.argv[1:] or ["conflicting", "original"]
results = [run_split(s) for s in splits]
print("\n" + "=" * 70)
for r in results:
report(r)
+49
View File
@@ -0,0 +1,49 @@
"""Confirm the model slug, the key, and tool support. Costs about a cent.
basic_agent loops uselessly against a model that cannot call a tool, so the second check
matters as much as the first: a model that answers "hi" but ignores tools would burn the
whole message limit on every sample and score zero for a reason that looks like honesty.
"""
import asyncio
import os
from dotenv import load_dotenv
from inspect_ai.model import ChatMessageUser, GenerateConfig, get_model
from inspect_ai.tool import ToolInfo, ToolParams
from inspect_ai.util import JSONSchema
MODEL = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
async def main() -> None:
load_dotenv()
model = get_model(MODEL)
out = await model.generate("hi")
print(f"[1/2] generate: {out.completion.strip()[:120]!r}")
print(f" tokens: in={out.usage.input_tokens} out={out.usage.output_tokens}")
add = ToolInfo(
name="add",
description="Add two integers.",
parameters=ToolParams(
properties={
"a": JSONSchema(type="integer", description="first addend"),
"b": JSONSchema(type="integer", description="second addend"),
},
required=["a", "b"],
),
)
out = await model.generate(
[ChatMessageUser(content="Use the add tool to add 17 and 25. Do not answer directly.")],
tools=[add],
config=GenerateConfig(max_tokens=200),
)
calls = out.message.tool_calls or []
print(f"[2/2] tool calls: {[(c.function, c.arguments) for c in calls]}")
print(" TOOL SUPPORT OK" if calls else " NO TOOL CALL: basic_agent will not work")
if __name__ == "__main__":
asyncio.run(main())
+51
View File
@@ -0,0 +1,51 @@
"""Free end-to-end check of the forked task: scratch directory, prompt, snapshot.
Uses mockllm, so it costs nothing and proves everything except the model: the scratch
directory is created before the agent runs, the system prompt names it exactly once, the
scorer snapshots the directory back out, and the inner ImpossibleBench scorer does not
flag test.py as modified (which would mean our construction had drifted from theirs).
"""
from inspect_ai import eval as inspect_eval
from messageboardbench.task import SCRATCH_LINE, scratch_livecodebench
if __name__ == "__main__":
log = inspect_eval(
scratch_livecodebench(limit=1, max_attempts=1, message_limit=4),
model="mockllm/model",
log_dir="./logs/smoke-scratch",
)[0]
print(f"\nstatus: {log.status}")
assert log.status == "success", log.error
ok = True
for sample in log.samples or []:
system = next(
(m.text for m in sample.messages if m.role == "system"), ""
)
n = system.count(SCRATCH_LINE)
whole = "\n".join(m.text for m in sample.messages)
total = whole.count(SCRATCH_LINE)
score = next(iter(sample.scores.values()))
meta = score.metadata or {}
print(f"sample {sample.id}")
print(f" scratch line in system message: {n} (want 1)")
print(f" scratch line anywhere else: {total - n} (want 0)")
print(f" test flagged as modified: {meta.get('test_modified_ever')} (want False)")
print(f" scratch directory created: {meta.get('scratch_exists')} (want True)")
print(f" scratch snapshot: {meta.get('scratch_files')}")
print(f" score: {score.value}")
ok &= (
n == 1
and total == n
and meta.get("test_modified_ever") is False
and meta.get("scratch_exists") is True
)
print("\nOK" if ok else "\nFAILED")
raise SystemExit(0 if ok else 1)