mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 11:07:07 +00:00
Repo skeleton, pinned environment, free smoke test
Public from commit one, so no key ever enters this history. compose.yaml is ImpossibleBench's, plus working_dir: /workspace. The image has no WORKDIR, so inspect resolves it to "/" and the task files land at the filesystem root among twenty-odd entries. This experiment turns on whether an agent notices a scratch directory, so that is a bad place to put one. ImpossibleBench installs with --no-deps to keep the swebench tree out; datasets is declared here instead because hf_dataset genuinely needs it. Verified: docker run prints "/", impossiblebench imports, and the real task against mockllm/model completes with a real score and tracebacks rooted at /workspace. Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
This commit is contained in:
commit
bf3bb0d144
12 files changed
+2495
No files matched your search
@@ -0,0 +1,35 @@
|
||||
"""Free end-to-end smoke test: the real task, a fake model, no money.
|
||||
|
||||
Runs the unmodified ImpossibleBench LiveCodeBench task against mockllm/model. The mock
|
||||
never calls a tool, so it burns the message limit and falls through to the scorer. That
|
||||
still exercises everything except the model: container start under our compose.yaml, the
|
||||
writes of func.py and test.py, the scorer's test-file comparison, and `python test.py`.
|
||||
|
||||
Run from the repo root so inspect finds compose.yaml (it looks in the process working
|
||||
directory). scripts/ recipes in the justfile do that for you.
|
||||
"""
|
||||
|
||||
from inspect_ai import eval as inspect_eval
|
||||
from impossiblebench import impossible_livecodebench
|
||||
|
||||
if __name__ == "__main__":
|
||||
logs = inspect_eval(
|
||||
impossible_livecodebench(
|
||||
split="conflicting",
|
||||
agent_type="tools",
|
||||
sandbox="docker",
|
||||
limit=1,
|
||||
max_attempts=1,
|
||||
message_limit=4,
|
||||
),
|
||||
model="mockllm/model",
|
||||
log_dir="./logs/smoke",
|
||||
)
|
||||
|
||||
log = logs[0]
|
||||
print(f"\nstatus: {log.status}")
|
||||
if log.error:
|
||||
print(f"error: {log.error}")
|
||||
for sample in log.samples or []:
|
||||
print(f"sample {sample.id}: score={sample.scores}")
|
||||
print(f" working dir sanity: {sample.metadata.get('task_id')}")
|
||||
Reference in new issue
Block a user