mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 11:07:07 +00:00
Public from commit one, so no key ever enters this history. compose.yaml is ImpossibleBench's, plus working_dir: /workspace. The image has no WORKDIR, so inspect resolves it to "/" and the task files land at the filesystem root among twenty-odd entries. This experiment turns on whether an agent notices a scratch directory, so that is a bad place to put one. ImpossibleBench installs with --no-deps to keep the swebench tree out; datasets is declared here instead because hf_dataset genuinely needs it. Verified: docker run prints "/", impossiblebench imports, and the real task against mockllm/model completes with a real score and tracebacks rooted at /workspace. Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
28 lines
691 B
TOML
28 lines
691 B
TOML
[project]
|
|
name = "messageboardbench"
|
|
version = "0.1.0"
|
|
description = "Does cheating spread between AI agents through a shared scratch directory?"
|
|
authors = [
|
|
{ name = "PJ", email = "[email protected]" }
|
|
]
|
|
requires-python = ">=3.13"
|
|
dependencies = [
|
|
"inspect-ai>=0.3.260",
|
|
# hf_dataset needs this to load fjzzq2002/impossible_livecodebench.
|
|
# ImpossibleBench is installed with --no-deps (see docs/setup.md), so its
|
|
# own declaration of datasets does not reach us.
|
|
"datasets>=3.0.0",
|
|
]
|
|
|
|
[build-system]
|
|
requires = ["hatchling"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[dependency-groups]
|
|
dev = [
|
|
"pytest>=9.1.1",
|
|
]
|
|
|
|
[tool.pytest.ini_options]
|
|
testpaths = ["tests"]
|