archive old experiments and publish swe series

This commit is contained in:
pj committed 2026-09-25 12:34:44 +05:30
1 parent 480100587e
commit 638e978227
1522 files changed
+220002 -4900

No files matched your search

+6 -1
View File
@@ -37,6 +37,11 @@ def main():
if feedback_operations_path.is_file() else [])
before = json.loads((args.run / 'budget-before.json').read_text())
after = json.loads((args.run / 'budget-after.json').read_text())
if before.get('provider') == 'clinepass':
account_limitation = ('ClinePass account quota is unavailable from the '
'documented API; use episode usage and the Cline dashboard.')
else:
account_limitation = 'Account usage changes may include billing delay or other account activity.'
result = {
'all': summarize(rows),
'by_condition': {c: summarize([r for r in rows if r['condition'] == c])
@@ -56,7 +61,7 @@ def main():
'Reasoning tokens are a subset of output, not an additional cost.',
'Input is uncached; cached input is reported separately and contributes to total.',
'Working seconds are summed episode time, not experiment wall time.',
'Account usage changes may include billing delay or other account activity.',
account_limitation,
'Small dependent development-task samples; descriptive comparisons only.',
],
}
+1 -1
View File
@@ -13,7 +13,7 @@ from typing import Any
from inspect_ai.log import read_eval_log
BOARD_POST_OPERATIONS = {'board_post', 'post_team_message', 'send_message'}
BOARD_POST_OPERATIONS = {'board_post', 'post_team_message', 'send_message', 'post_message'}
BOARD_READ_OPERATIONS = {'board_read', 'read_team_messages', 'read_messages'}
+10 -4
View File
@@ -13,12 +13,18 @@ from inspect_ai.model import ChatMessageUser, GenerateConfig, get_model
from inspect_ai.tool import ToolInfo, ToolParams
from inspect_ai.util import JSONSchema
MODEL = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
async def main() -> None:
load_dotenv()
model = get_model(MODEL)
model_name = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
model_args = {}
if model_name.startswith("openai-api/cline/cline-pass/"):
model_args = {
"base_url": "https://api.cline.bot/api/v1",
"responses_api": False,
"stream": True,
"strict_tools": False,
}
model = get_model(model_name, **model_args)
out = await model.generate("hi")
print(f"[1/2] generate: {out.completion.strip()[:120]!r}")
+150
View File
@@ -0,0 +1,150 @@
"""Write an automatic, unreviewed report for separate SWE control runs."""
from __future__ import annotations
import argparse
import json
from pathlib import Path
from inspect_ai.log import read_eval_log
def write_json(path: Path, value: object) -> None:
path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n")
def score_label(row: dict | None) -> str:
if row is None:
return "missing"
value = row.get("score")
if value is None:
return "null"
if isinstance(value, (int, float)) and not isinstance(value, bool):
if value == 1:
return "1"
if value == 0:
return "0"
return str(value)
def run_report(label: str, run_dir: Path) -> tuple[dict, dict[str, dict], set[str]]:
manifest = json.loads((run_dir / "manifest.json").read_text())
if manifest.get("conditions") != ["control"]:
raise ValueError(f"{label}: expected a control-only run")
rows = json.loads((run_dir / "results.json").read_text())
task_ids = list(manifest["records_sha256"])
by_task = {row["sample_id"]: row for row in rows}
if len(by_task) != len(rows):
raise ValueError(f"{label}: duplicate terminal task rows")
if set(by_task) - set(task_ids):
raise ValueError(f"{label}: terminal task is absent from the plan")
tokens = {name: 0 for name in (
"input", "output", "cache_read", "reasoning", "total", "limit_usage"
)}
limits = {name: 0 for name in ("token", "message", "time", "other")}
for row in rows:
log_path = Path(row["log"])
if not log_path.exists():
log_path = run_dir / "evals" / log_path.name
log = read_eval_log(log_path)
sample = next(sample for sample in log.samples or []
if str(sample.id) == row["sample_id"])
for usage in sample.model_usage.values():
for source, target in (
("input_tokens", "input"), ("output_tokens", "output"),
("input_tokens_cache_read", "cache_read"),
("reasoning_tokens", "reasoning"), ("total_tokens", "total"),
):
tokens[target] += getattr(usage, source, None) or 0
tokens["limit_usage"] += sample.token_limit_usage or 0
if sample.limit:
kind = sample.limit.type
limits[kind if kind in limits else "other"] += 1
scores = [score_label(row) for row in rows]
report = {
"model": manifest.get("model") or next(iter(manifest["models_by_team"].values())),
"run_dir": str(run_dir.resolve()),
"planned": manifest.get("planned_episodes", len(task_ids)),
"terminal": len(rows),
"score_1": scores.count("1"),
"score_0": scores.count("0"),
"score_null": scores.count("null"),
"errors": sum(bool(row.get("error")) or row.get("log_status") == "error"
for row in rows),
"test_modified_ever": sum(row.get("test_modified_ever") is True for row in rows),
"test_modified_unknown": sum(row.get("test_modified_ever") is None for row in rows),
"tokens": tokens,
"limits": limits,
}
return report, by_task, set(task_ids)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--run", action="append", required=True, metavar="LABEL=RUN_DIR")
parser.add_argument("--out", type=Path, required=True)
args = parser.parse_args()
runs = {}
for item in args.run:
if "=" not in item:
parser.error("--run must have the form LABEL=RUN_DIR")
label, path = item.split("=", 1)
if not label or label in runs:
parser.error("--run labels must be nonempty and unique")
runs[label] = Path(path)
summaries = {}
rows_by_model = {}
planned_tasks = set()
for label, path in runs.items():
summaries[label], rows_by_model[label], task_ids = run_report(label, path)
planned_tasks.update(task_ids)
task_ids = sorted(planned_tasks)
paired = [
{"task_id": task_id,
"scores": {label: score_label(rows.get(task_id))
for label, rows in rows_by_model.items()}}
for task_id in task_ids
]
report = {
"automatic_unreviewed": True,
"scope": "control-only ImpossibleBench SWE baseline; no communication tools",
"models": summaries,
"per_task": paired,
"limitations": [
"A numeric scorer pass is not an automatic cheating or behavioral label.",
"Null and missing outcomes are shown separately and excluded from numeric score counts.",
"This baseline does not estimate any communication effect.",
],
}
args.out.mkdir(parents=True, exist_ok=True)
write_json(args.out / "report.json", report)
lines = [
"# Automatic SWE baseline report", "",
"Deterministic and unreviewed. Scores are evaluator outputs, not behavioral labels.", "",
"| Model | Terminal / planned | Score 1 | Score 0 | Null | Errors | Test modified | Token limit | Message limit | Total metered tokens |",
"|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|",
]
for label, row in summaries.items():
lines.append(
f"| {label} | {row['terminal']} / {row['planned']} | {row['score_1']} | "
f"{row['score_0']} | {row['score_null']} | {row['errors']} | "
f"{row['test_modified_ever']} | {row['limits']['token']} | "
f"{row['limits']['message']} | {row['tokens']['total']} |"
)
lines += ["", "## Paired task scores", "",
"| Task | " + " | ".join(runs) + " |",
"|---|" + "---:|" * len(runs)]
for item in paired:
lines.append("| " + item["task_id"] + " | " +
" | ".join(item["scores"][label] for label in runs) + " |")
lines += ["", "Missing means no terminal row; null means a terminal row without a numeric score.",
"Raw logs remain in each run directory. This control-only baseline makes no communication claim.", ""]
(args.out / "REPORT.md").write_text("\n".join(lines))
print(f"Wrote {args.out / 'report.json'} and {args.out / 'REPORT.md'}", flush=True)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+339 -38
View File
@@ -19,15 +19,57 @@ from messageboardbench.swe_validation import REMOTE_DOCKER_HOST
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CONDITIONS = ("control", "board")
CLINEPASS_MODELS = {
"openai-api/cline/cline-pass/kimi-k3",
"openai-api/cline/cline-pass/deepseek-v4-pro",
"openai-api/cline/cline-pass/deepseek-v4-flash",
}
CLINE_API_BASE_URL = "https://api.cline.bot/api/v1"
def model_provider(model: str) -> str:
if model.startswith("openrouter/"):
return "openrouter"
if model in CLINEPASS_MODELS:
return "clinepass"
raise ValueError(f"unsupported SWE experiment model: {model}")
def plan_models(plan: dict) -> list[str]:
models = plan.get("models_by_team")
if models is not None:
return [models.get(str(team["team"]), plan.get("model"))
for team in plan["team_plans"]]
return [plan["model"]]
def uses_engineering_sentinel(plan: dict) -> bool:
"""Keep the legacy paired-pilot stop rule out of completed-validation runs."""
return not str(plan.get("purpose", "")).startswith("swe-board-activation-")
purpose = str(plan.get("purpose", ""))
return (purpose != "baseline-swebench"
and not purpose.startswith("swe-board-activation-")
and not purpose.startswith("swe-scenario-"))
def treatment_metadata(plan: dict) -> dict:
"""Describe the actual model-visible intervention without legacy-arm claims."""
if str(plan.get("purpose", "")).startswith("swe-scenario-"):
return {
"conditions": plan["conditions"],
"agent_network_mode": plan["agent_network_mode"],
"shared_scratch": bool(plan.get("shared_scratch")),
"board_interface": plan.get("tool_interface"),
"organizer_feedback": plan.get("organizer_feedback_interface"),
"token_status_tool": bool(plan.get("token_status_tool")),
"system_prompt_change": plan["custom_prompt"],
}
if plan.get("purpose") == "baseline-swebench":
return {
"conditions": ["control"],
"control": "upstream ImpossibleBench SWE tools with no communication tools",
"system_prompt_change": plan["custom_prompt"],
"organizer_feedback": None,
}
if str(plan.get("purpose", "")).startswith("swe-board-activation-"):
return {
"conditions": ["board"],
@@ -56,6 +98,11 @@ def parser() -> argparse.ArgumentParser:
p.add_argument("--out", type=Path, required=True)
p.add_argument("--plan", type=Path, required=True)
p.add_argument("--execute", action="store_true")
p.add_argument(
"--resume-runner-upgrade",
choices=("terminal-grader-null-v1",),
help="explicit runner-only upgrade for an existing interrupted run",
)
return p
@@ -63,22 +110,101 @@ def dump(path: Path, value) -> None:
path.write_text(json.dumps(value, indent=2, ensure_ascii=False, default=str) + "\n")
def validate_resume_sources(archive: Path, sources: list[Path]) -> None:
def snapshot_scratch(volume: str, destination: Path) -> bool:
"""Save a shared Docker volume to the local run directory after a phase."""
with destination.open("wb") as output:
result = subprocess.run(
["docker", "run", "--rm", "--network", "none",
"--volume", f"{volume}:/scratch:ro",
"aisiuk/inspect-tool-support", "tar", "-C", "/scratch", "-cf", "-", "."],
stdout=output, stderr=subprocess.PIPE,
)
if result.returncode:
destination.with_suffix(".error.txt").write_text(
f"docker snapshot exited {result.returncode}\n"
+ result.stderr.decode(errors="replace")
)
elif destination.with_suffix(".error.txt").exists():
destination.with_suffix(".error.txt").unlink()
return result.returncode == 0
def validate_resume_sources(
archive: Path,
sources: list[Path],
*,
runner_upgrade: str | None = None,
recovery_archive: Path | None = None,
) -> None:
"""Reject resume when archived or current behavioral source bytes changed."""
index = json.loads((archive / "index.json").read_text())
current = {str(path.resolve()): path for path in sources}
if set(current) != {item["source"] for item in index}:
raise RuntimeError("resume source set differs from frozen source snapshot")
mismatches = []
for item in index:
archived = archive / item["archived"]
if hashlib.sha256(archived.read_bytes()).hexdigest() != item["sha256"]:
raise RuntimeError("resume source snapshot hash mismatch")
if hashlib.sha256(current[item["source"]].read_bytes()).hexdigest() != item["sha256"]:
raise RuntimeError("current behavioral source differs from frozen resume snapshot")
mismatches.append(item)
if not mismatches:
return
runner = str(Path(__file__).resolve())
if (runner_upgrade != "terminal-grader-null-v1"
or len(mismatches) != 1
or mismatches[0]["source"] != runner
or recovery_archive is None):
raise RuntimeError("current behavioral source differs from frozen resume snapshot")
raw = current[runner].read_bytes()
static = {
"upgrade": runner_upgrade,
"source": runner,
"base_sha256": mismatches[0]["sha256"],
"recovery_sha256": hashlib.sha256(raw).hexdigest(),
"archived": "swe_board_experiment.py",
"base_index_sha256": hashlib.sha256(
(archive / "index.json").read_bytes()
).hexdigest(),
"reason": "retain completed model trajectories with grader-only null outcomes",
}
if recovery_archive.exists():
recovery = json.loads((recovery_archive / "index.json").read_text())
if any(recovery.get(key) != value for key, value in static.items()):
raise RuntimeError("recovery source snapshot differs from current runner")
if hashlib.sha256(
(recovery_archive / static["archived"]).read_bytes()
).hexdigest() != static["recovery_sha256"]:
raise RuntimeError("recovery source snapshot hash mismatch")
for item in recovery["preexisting_evals"]:
path = archive.parent / item["path"]
if hashlib.sha256(path.read_bytes()).hexdigest() != item["sha256"]:
raise RuntimeError("pre-recovery eval changed after runner upgrade")
else:
recovery = {
**static,
"created_at": datetime.now(timezone.utc).isoformat(),
"preexisting_evals": [
{
"path": str(path.relative_to(archive.parent)),
"sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
}
for path in sorted((archive.parent / "evals").glob("*.eval"))
],
}
recovery_archive.mkdir()
(recovery_archive / static["archived"]).write_bytes(raw)
dump(recovery_archive / "index.json", recovery)
def terminal_row(log, evaluated) -> dict:
final_score = next(iter(evaluated.scores.values())) if evaluated.scores else None
score_metadata = (final_score.metadata or {}) if final_score else {}
test_modified_ever = score_metadata.get("test_modified_ever")
if test_modified_ever is None:
test_modified_ever = evaluated.metadata.get(
"_messageboardbench_test_modified_ever"
)
return {
"log": log.location, "log_status": log.status,
"sample_id": str(evaluated.id),
@@ -89,14 +215,46 @@ def terminal_row(log, evaluated) -> dict:
"messages": len(evaluated.messages),
"error": evaluated.error.message if evaluated.error else None,
"model_patch_captured": (
isinstance((final_score.metadata or {}).get("model_patch"), str)
isinstance(score_metadata.get("model_patch"), str)
if final_score else False
),
"test_modified_ever": (final_score.metadata or {}).get("test_modified_ever") if final_score else None,
"test_modified_ever": test_modified_ever,
"grader_diagnostic": evaluated.metadata.get(
"_messageboardbench_grader_diagnostic"
),
"manual_behavior_review": "pending",
}
def grader_null_row(row: dict) -> bool:
diagnostic = row.get("grader_diagnostic")
if not (
row.get("log_status") == "success"
and row.get("error") is not None
and row.get("score") is None
and isinstance(diagnostic, dict)
and isinstance(diagnostic.get("exit_code"), int)
and isinstance(diagnostic.get("target_statuses"), dict)
and isinstance(diagnostic.get("eval_script_sha256"), str)
and len(diagnostic["eval_script_sha256"]) == 64
and isinstance(diagnostic.get("output_tail"), str)
):
return False
statuses = set(diagnostic["target_statuses"].values())
return not statuses or bool(statuses & {"MISSING", "ERROR"})
def completed_row(row: dict) -> bool:
"""Return whether model work ended in a score or a grader-only null outcome."""
return (
row.get("log_status") == "success"
and (
(row.get("error") is None and row.get("score") is not None)
or grader_null_row(row)
)
)
def sentinel_failed(
rows: list[dict], *, team: int, condition: str, instance_ids: list[str]
) -> bool:
@@ -117,20 +275,42 @@ def sentinel_failed(
def recover_terminal_rows(out: Path) -> list[dict]:
"""Recover completed/error samples written before a process interruption."""
"""Recover scored and grader-null outcomes, leaving other failures retryable."""
from inspect_ai.log import read_eval_log
rows = []
by_assignment = {}
for path in sorted((out / "evals").rglob("*.eval")) if (out / "evals").exists() else []:
log = read_eval_log(path, resolve_attachments=True)
if log.status not in {"success", "error", "cancelled"}:
continue
for sample in log.samples or []:
if sample.metadata and sample.metadata.get("episode_id"):
rows.append(terminal_row(log, sample))
keys = [(row["team"], row["condition"], row["sample_id"]) for row in rows]
if len(keys) != len(set(keys)):
raise RuntimeError("duplicate terminal SWE assignments found during resume")
row = terminal_row(log, sample)
if completed_row(row):
key = (row["team"], row["condition"], row["sample_id"])
by_assignment.setdefault(key, []).append(row)
rows = []
for key, attempts in by_assignment.items():
if len(attempts) == 1:
rows.append(attempts[0])
continue
if not all(grader_null_row(row) for row in attempts):
raise RuntimeError(f"ambiguous duplicate terminal SWE assignment: {key}")
signatures = {
(
row["grader_diagnostic"]["eval_script_sha256"],
json.dumps(
row["grader_diagnostic"]["target_statuses"], sort_keys=True
),
)
for row in attempts
}
if len(signatures) != 1:
raise RuntimeError(f"grader-null retries disagree for assignment: {key}")
selected = dict(attempts[0])
selected["attempt_count"] = len(attempts)
selected["redundant_attempt_logs"] = [row["log"] for row in attempts[1:]]
rows.append(selected)
return rows
@@ -174,7 +354,7 @@ def cleanup_matched_images(
dump(path, lifecycle)
def account_budget() -> dict:
def openrouter_budget() -> dict:
import httpx
response = httpx.get(
@@ -190,6 +370,18 @@ def account_budget() -> dict:
}
def account_budget(provider: str) -> dict:
if provider == "openrouter":
return openrouter_budget()
return {
"checked_at": datetime.now(timezone.utc).isoformat(),
"provider": "clinepass",
"account_usage": None,
"account_quota": None,
"status": "unavailable_from_documented_api",
}
def main(argv: list[str] | None = None) -> int:
args = parser().parse_args(argv)
if args.execute and os.environ.get("DOCKER_HOST") != REMOTE_DOCKER_HOST:
@@ -199,6 +391,10 @@ def main(argv: list[str] | None = None) -> int:
from messageboardbench.swe_board import load_records
plan_bytes = args.plan.read_bytes()
plan = json.loads(plan_bytes)
providers = {model_provider(model) for model in plan_models(plan)}
if len(providers) != 1:
raise ValueError("use separate experiment plans for OpenRouter and ClinePass populations")
provider = next(iter(providers))
conditions = tuple(plan.get("conditions", DEFAULT_CONDITIONS))
split = plan["dataset"]["split"]
records = load_records(plan["dataset"]["revision"], split)
@@ -211,6 +407,8 @@ def main(argv: list[str] | None = None) -> int:
"remote_docker_host": REMOTE_DOCKER_HOST,
"host_mounts": [],
}
if provider == "clinepass":
config["clinepass_base_url"] = CLINE_API_BASE_URL
print(json.dumps(config, indent=2), flush=True)
if not args.execute:
print("Preview only; no Docker container or model request was started.", flush=True)
@@ -230,22 +428,33 @@ def main(argv: list[str] | None = None) -> int:
load_dotenv(ROOT / ".env")
out = args.out.resolve()
fresh = not out.exists()
if fresh and args.resume_runner_upgrade is not None:
raise RuntimeError("runner upgrades apply only to an existing interrupted run")
current_budget = account_budget(provider)
if (provider == "openrouter"
and (current_budget["limit_remaining"] is None
or current_budget["limit_remaining"] <= 0)):
raise RuntimeError("OpenRouter key has no remaining budget")
if fresh:
before = current_budget
out.mkdir(parents=True)
elif not (out / "manifest.json").is_file():
raise RuntimeError("resume output lacks a manifest")
parameters = plan["parameters"]
schedule = plan["schedule"]
team_plans = plan["team_plans"]
configs = out / "compose"
compose_by_assignment = {
instance_id: write_compose(
records[instance_id], configs, parameters["memory"],
)
for instance_id in records
}
scratch_volume = None
scratch_identity = None
if plan.get("shared_scratch"):
identity_path = out / "scratch-volume.json"
if fresh:
scratch_volume = "messageboardbench-scratch-" + uuid.uuid4().hex
scratch_identity = {"name": scratch_volume, "created": False, "removed": False}
dump(identity_path, scratch_identity)
else:
scratch_identity = json.loads(identity_path.read_text())
scratch_volume = scratch_identity["name"]
if fresh:
before = account_budget()
dump(out / "manifest.json", config)
dump(out / "budget-before.json", before)
else:
@@ -272,6 +481,8 @@ def main(argv: list[str] | None = None) -> int:
Path(upstream_scorer.__file__),
Path(upstream_tasks.__file__),
]
if str(plan.get("purpose", "")).startswith("swe-scenario-"):
sources.append(ROOT / "experiments/baseline-swebench/assets/flit_core-3.7.1-py3-none-any.whl")
if str(plan.get("purpose", "")).startswith("swe-board-activation-"):
sources.extend([
ROOT / "scripts/swe_activation_report.py",
@@ -291,17 +502,43 @@ def main(argv: list[str] | None = None) -> int:
)
dump(archive / "index.json", index)
else:
validate_resume_sources(archive, [*sources, args.plan])
validate_resume_sources(
archive,
[*sources, args.plan],
runner_upgrade=args.resume_runner_upgrade,
recovery_archive=out / "resume-source-snapshot",
)
configs = out / "compose"
compose_by_assignment = {
instance_id: write_compose(
records[instance_id], configs, parameters["memory"],
agent_network_mode=plan.get("agent_network_mode"),
scratch_volume=scratch_volume,
)
for instance_id in records
}
if scratch_identity and not scratch_identity["removed"]:
if scratch_identity["created"]:
subprocess.run(["docker", "volume", "inspect", scratch_volume], check=True,
capture_output=True, text=True)
else:
subprocess.run(["docker", "volume", "create", scratch_volume], check=True,
capture_output=True, text=True)
scratch_identity["created"] = True
dump(identity_path, scratch_identity)
boards = {}
has_board = "board" in conditions
feedback = None
if fresh:
identities = []
for team_plan in team_plans:
team = team_plan["team"]
run_id = uuid.uuid4().hex
path = out / f"board-team-{team}.sqlite"
initialize_board(path, run_id)
run_id = uuid.uuid4().hex if has_board else None
if run_id is not None:
initialize_board(out / f"board-team-{team}.sqlite", run_id)
episodes = {condition: {instance_id: "worker-" + uuid.uuid4().hex[:12]
for instance_id in team_plan["instance_ids"]}
for condition in conditions}
@@ -317,10 +554,11 @@ def main(argv: list[str] | None = None) -> int:
if json.loads((out / "schedule.json").read_text()) != schedule:
raise RuntimeError("resume schedule differs")
for identity in identities:
boards[identity["team"]] = {
"path": out / f"board-team-{identity['team']}.sqlite",
"run_id": identity["board_run_id"],
}
if identity["board_run_id"] is not None:
boards[identity["team"]] = {
"path": out / f"board-team-{identity['team']}.sqlite",
"run_id": identity["board_run_id"],
}
if plan.get("organizer_feedback_interface"):
feedback_identity = json.loads((out / "feedback-identity.json").read_text())
feedback = {
@@ -341,6 +579,12 @@ def main(argv: list[str] | None = None) -> int:
pending = [instance_id for instance_id in selected
if (team, condition, instance_id) not in terminal]
if not pending:
if scratch_volume and not scratch_identity["removed"]:
snapshot = out / f"scratch-after-phase-{phase}.tar"
if (not snapshot.exists()
or snapshot.with_suffix(".error.txt").exists()):
if not snapshot_scratch(scratch_volume, snapshot):
raise RuntimeError(f"shared scratch snapshot failed: phase {phase}")
if uses_engineering_sentinel(plan) and phase <= 2 and sentinel_failed(
results, team=team, condition=condition, instance_ids=selected
):
@@ -359,7 +603,7 @@ def main(argv: list[str] | None = None) -> int:
offset = selected.index(instance_id) + 1
slot = team_plans[team - 1]["instance_ids"].index(instance_id) + 1
episode_id = identity["episodes"][condition][instance_id]
board = boards[team]
board = boards.get(team)
sample = sample_from_record(
records[instance_id], compose_by_assignment[instance_id],
)
@@ -383,10 +627,18 @@ def main(argv: list[str] | None = None) -> int:
tool_interface=plan.get("tool_interface"),
feedback_path=feedback["path"] if feedback else None,
feedback_run_id=feedback["run_id"] if feedback else None,
enable_feedback=bool(plan.get("organizer_feedback_interface")),
enable_token_checker=bool(plan.get("token_status_tool")),
),
scorer=swe_board_scorer(
memory=parameters["memory"],
timeout_seconds=parameters["scorer_timeout_seconds"],
pin_grader_image=str(plan.get("purpose", "")).startswith(
"swe-scenario-"
),
strict_grader_statuses=str(plan.get("purpose", "")).startswith(
"swe-scenario-"
),
),
message_limit=parameters["message_limit"],
metadata={**config, "condition": condition, "team": team,
@@ -402,13 +654,28 @@ def main(argv: list[str] | None = None) -> int:
else:
dump(input_path, inputs)
print(f"Starting phase {phase}: team {team} {condition} cohort {cohort}", flush=True)
model = plan.get("models_by_team", {}).get(str(team), plan.get("model"))
model_args = {"strict_tools": False}
model_base_url = None
if model_provider(model) == "clinepass":
if (parameters["reasoning_effort"] is not None
or parameters["reasoning_tokens"] is not None):
raise ValueError(
"ClinePass plans must set reasoning_effort and reasoning_tokens "
"to null; their existing OpenRouter settings are not portable"
)
model_base_url = CLINE_API_BASE_URL
model_args.update(responses_api=False, stream=True)
logs = inspect_eval(
tasks,
model=plan.get("models_by_team", {}).get(str(team), plan.get("model")),
model_args={"strict_tools": False},
model=model,
model_base_url=model_base_url,
model_args=model_args,
log_dir=str(out / "evals"),
max_tasks=len(tasks), max_samples=len(tasks), max_sandboxes=len(tasks),
max_connections=len(tasks), max_retries=1, retry_on_error=0,
max_connections=parameters.get("max_connections", len(tasks)),
max_retries=parameters.get("request_retries", 1),
retry_on_error=parameters.get("sample_retries", 0),
fail_on_error=False, time_limit=parameters["time_limit_seconds"],
token_limit=parameters["token_limit"],
reasoning_effort=parameters["reasoning_effort"],
@@ -419,17 +686,32 @@ def main(argv: list[str] | None = None) -> int:
for log in logs:
for evaluated in log.samples or []:
new.append(terminal_row(log, evaluated))
results.extend(new)
terminal.update((row["team"], row["condition"], row["sample_id"]) for row in new)
dump(out / "results.json", results)
completed = [row for row in new if completed_row(row)]
results.extend(completed)
terminal.update(
(row["team"], row["condition"], row["sample_id"])
for row in completed
)
dump(out / "results.json", [
*results,
*(row for row in new if not completed_row(row)),
])
dump(out / f"board-after-phase-{phase}.json", {
"posts": [post for value in boards.values() for post in export_board(value["path"], value["run_id"])["posts"]],
"audit": [event for value in boards.values() for event in export_board(value["path"], value["run_id"])["audit"]],
})
if scratch_volume:
if not snapshot_scratch(scratch_volume, out / f"scratch-after-phase-{phase}.tar"):
raise RuntimeError(f"shared scratch snapshot failed: phase {phase}")
if len(new) != len(tasks):
raise RuntimeError("phase did not produce one terminal record per assignment")
failed = [row["sample_id"] for row in new if not completed_row(row)]
if failed:
raise RuntimeError(
"phase produced failed assignments eligible for retry: "
+ ", ".join(failed)
)
# The first adjacent control/board pair is an engineering sentinel.
# Later sample errors are terminal outcomes and do not trigger reruns.
if uses_engineering_sentinel(plan) and phase <= 2 and sentinel_failed(
results, team=team, condition=condition, instance_ids=selected
):
@@ -452,6 +734,20 @@ def main(argv: list[str] | None = None) -> int:
raise
finally:
dump(out / "status.json", status)
if scratch_volume and not scratch_identity["removed"]:
saved = snapshot_scratch(scratch_volume, out / "scratch-final.tar")
if not saved and status["status"] == "completed":
status.update(status="interrupted", error="shared scratch final snapshot failed")
dump(out / "status.json", status)
if saved and status["status"] == "completed":
removed = subprocess.run(
["docker", "volume", "rm", scratch_volume],
capture_output=True, text=True,
)
scratch_identity["removed"] = removed.returncode == 0
if removed.returncode:
scratch_identity["remove_error"] = removed.stderr
dump(identity_path, scratch_identity)
snapshots = [export_board(value["path"], value["run_id"]) for value in boards.values()]
dump(out / "board-final.json", {
"run_ids": [value["run_id"] for value in snapshots],
@@ -463,11 +759,16 @@ def main(argv: list[str] | None = None) -> int:
feedback["path"], feedback["run_id"]
))
try:
after = account_budget()
after["usage_delta"] = after["usage"] - before["usage"]
after = account_budget(provider)
if provider == "openrouter":
after["usage_delta"] = after["usage"] - before["usage"]
else:
after["usage_delta"] = None
except Exception as error:
after = {"accounting_error": repr(error), "usage_delta": None}
dump(out / "budget-after.json", after)
if status["status"] != "completed":
raise RuntimeError(status["error"])
return 0
@@ -14,6 +14,7 @@ from messageboardbench.swe_prerequisites import (
validate_task_manifest,
)
from messageboardbench.swe_validation import (
GRADER_ENVIRONMENT,
ValidationError,
docker_preflight,
manifest as trial_manifest,
@@ -96,6 +97,8 @@ def main() -> int:
plan = json.loads(args.plan.read_text())
if plan.get("plan_sha256") != plan_hash(plan):
raise SystemExit("frozen plan self-hash mismatch")
if plan.get("parameters", {}).get("grader_environment") != GRADER_ENVIRONMENT:
raise SystemExit("frozen plan grader environment mismatch")
declared = (ROOT / plan["environment_validation"]["index_path"]).resolve()
out = args.out.resolve()
if declared != out / "index.json":