mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 11:07:07 +00:00
archive old experiments and publish swe series
This commit is contained in:
1 parent
480100587e
commit
638e978227
1522 files changed
+220002
-4900
No files matched your search
@@ -37,6 +37,11 @@ def main():
|
||||
if feedback_operations_path.is_file() else [])
|
||||
before = json.loads((args.run / 'budget-before.json').read_text())
|
||||
after = json.loads((args.run / 'budget-after.json').read_text())
|
||||
if before.get('provider') == 'clinepass':
|
||||
account_limitation = ('ClinePass account quota is unavailable from the '
|
||||
'documented API; use episode usage and the Cline dashboard.')
|
||||
else:
|
||||
account_limitation = 'Account usage changes may include billing delay or other account activity.'
|
||||
result = {
|
||||
'all': summarize(rows),
|
||||
'by_condition': {c: summarize([r for r in rows if r['condition'] == c])
|
||||
@@ -56,7 +61,7 @@ def main():
|
||||
'Reasoning tokens are a subset of output, not an additional cost.',
|
||||
'Input is uncached; cached input is reported separately and contributes to total.',
|
||||
'Working seconds are summed episode time, not experiment wall time.',
|
||||
'Account usage changes may include billing delay or other account activity.',
|
||||
account_limitation,
|
||||
'Small dependent development-task samples; descriptive comparisons only.',
|
||||
],
|
||||
}
|
||||
|
||||
@@ -13,7 +13,7 @@ from typing import Any
|
||||
from inspect_ai.log import read_eval_log
|
||||
|
||||
|
||||
BOARD_POST_OPERATIONS = {'board_post', 'post_team_message', 'send_message'}
|
||||
BOARD_POST_OPERATIONS = {'board_post', 'post_team_message', 'send_message', 'post_message'}
|
||||
BOARD_READ_OPERATIONS = {'board_read', 'read_team_messages', 'read_messages'}
|
||||
|
||||
|
||||
|
||||
+10
-4
@@ -13,12 +13,18 @@ from inspect_ai.model import ChatMessageUser, GenerateConfig, get_model
|
||||
from inspect_ai.tool import ToolInfo, ToolParams
|
||||
from inspect_ai.util import JSONSchema
|
||||
|
||||
MODEL = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
|
||||
|
||||
|
||||
async def main() -> None:
|
||||
load_dotenv()
|
||||
model = get_model(MODEL)
|
||||
model_name = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
|
||||
model_args = {}
|
||||
if model_name.startswith("openai-api/cline/cline-pass/"):
|
||||
model_args = {
|
||||
"base_url": "https://api.cline.bot/api/v1",
|
||||
"responses_api": False,
|
||||
"stream": True,
|
||||
"strict_tools": False,
|
||||
}
|
||||
model = get_model(model_name, **model_args)
|
||||
|
||||
out = await model.generate("hi")
|
||||
print(f"[1/2] generate: {out.completion.strip()[:120]!r}")
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
"""Write an automatic, unreviewed report for separate SWE control runs."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from inspect_ai.log import read_eval_log
|
||||
|
||||
|
||||
def write_json(path: Path, value: object) -> None:
|
||||
path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n")
|
||||
|
||||
|
||||
def score_label(row: dict | None) -> str:
|
||||
if row is None:
|
||||
return "missing"
|
||||
value = row.get("score")
|
||||
if value is None:
|
||||
return "null"
|
||||
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
||||
if value == 1:
|
||||
return "1"
|
||||
if value == 0:
|
||||
return "0"
|
||||
return str(value)
|
||||
|
||||
|
||||
def run_report(label: str, run_dir: Path) -> tuple[dict, dict[str, dict], set[str]]:
|
||||
manifest = json.loads((run_dir / "manifest.json").read_text())
|
||||
if manifest.get("conditions") != ["control"]:
|
||||
raise ValueError(f"{label}: expected a control-only run")
|
||||
rows = json.loads((run_dir / "results.json").read_text())
|
||||
task_ids = list(manifest["records_sha256"])
|
||||
by_task = {row["sample_id"]: row for row in rows}
|
||||
if len(by_task) != len(rows):
|
||||
raise ValueError(f"{label}: duplicate terminal task rows")
|
||||
if set(by_task) - set(task_ids):
|
||||
raise ValueError(f"{label}: terminal task is absent from the plan")
|
||||
|
||||
tokens = {name: 0 for name in (
|
||||
"input", "output", "cache_read", "reasoning", "total", "limit_usage"
|
||||
)}
|
||||
limits = {name: 0 for name in ("token", "message", "time", "other")}
|
||||
for row in rows:
|
||||
log_path = Path(row["log"])
|
||||
if not log_path.exists():
|
||||
log_path = run_dir / "evals" / log_path.name
|
||||
log = read_eval_log(log_path)
|
||||
sample = next(sample for sample in log.samples or []
|
||||
if str(sample.id) == row["sample_id"])
|
||||
for usage in sample.model_usage.values():
|
||||
for source, target in (
|
||||
("input_tokens", "input"), ("output_tokens", "output"),
|
||||
("input_tokens_cache_read", "cache_read"),
|
||||
("reasoning_tokens", "reasoning"), ("total_tokens", "total"),
|
||||
):
|
||||
tokens[target] += getattr(usage, source, None) or 0
|
||||
tokens["limit_usage"] += sample.token_limit_usage or 0
|
||||
if sample.limit:
|
||||
kind = sample.limit.type
|
||||
limits[kind if kind in limits else "other"] += 1
|
||||
|
||||
scores = [score_label(row) for row in rows]
|
||||
report = {
|
||||
"model": manifest.get("model") or next(iter(manifest["models_by_team"].values())),
|
||||
"run_dir": str(run_dir.resolve()),
|
||||
"planned": manifest.get("planned_episodes", len(task_ids)),
|
||||
"terminal": len(rows),
|
||||
"score_1": scores.count("1"),
|
||||
"score_0": scores.count("0"),
|
||||
"score_null": scores.count("null"),
|
||||
"errors": sum(bool(row.get("error")) or row.get("log_status") == "error"
|
||||
for row in rows),
|
||||
"test_modified_ever": sum(row.get("test_modified_ever") is True for row in rows),
|
||||
"test_modified_unknown": sum(row.get("test_modified_ever") is None for row in rows),
|
||||
"tokens": tokens,
|
||||
"limits": limits,
|
||||
}
|
||||
return report, by_task, set(task_ids)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--run", action="append", required=True, metavar="LABEL=RUN_DIR")
|
||||
parser.add_argument("--out", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
runs = {}
|
||||
for item in args.run:
|
||||
if "=" not in item:
|
||||
parser.error("--run must have the form LABEL=RUN_DIR")
|
||||
label, path = item.split("=", 1)
|
||||
if not label or label in runs:
|
||||
parser.error("--run labels must be nonempty and unique")
|
||||
runs[label] = Path(path)
|
||||
|
||||
summaries = {}
|
||||
rows_by_model = {}
|
||||
planned_tasks = set()
|
||||
for label, path in runs.items():
|
||||
summaries[label], rows_by_model[label], task_ids = run_report(label, path)
|
||||
planned_tasks.update(task_ids)
|
||||
task_ids = sorted(planned_tasks)
|
||||
paired = [
|
||||
{"task_id": task_id,
|
||||
"scores": {label: score_label(rows.get(task_id))
|
||||
for label, rows in rows_by_model.items()}}
|
||||
for task_id in task_ids
|
||||
]
|
||||
report = {
|
||||
"automatic_unreviewed": True,
|
||||
"scope": "control-only ImpossibleBench SWE baseline; no communication tools",
|
||||
"models": summaries,
|
||||
"per_task": paired,
|
||||
"limitations": [
|
||||
"A numeric scorer pass is not an automatic cheating or behavioral label.",
|
||||
"Null and missing outcomes are shown separately and excluded from numeric score counts.",
|
||||
"This baseline does not estimate any communication effect.",
|
||||
],
|
||||
}
|
||||
args.out.mkdir(parents=True, exist_ok=True)
|
||||
write_json(args.out / "report.json", report)
|
||||
lines = [
|
||||
"# Automatic SWE baseline report", "",
|
||||
"Deterministic and unreviewed. Scores are evaluator outputs, not behavioral labels.", "",
|
||||
"| Model | Terminal / planned | Score 1 | Score 0 | Null | Errors | Test modified | Token limit | Message limit | Total metered tokens |",
|
||||
"|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|",
|
||||
]
|
||||
for label, row in summaries.items():
|
||||
lines.append(
|
||||
f"| {label} | {row['terminal']} / {row['planned']} | {row['score_1']} | "
|
||||
f"{row['score_0']} | {row['score_null']} | {row['errors']} | "
|
||||
f"{row['test_modified_ever']} | {row['limits']['token']} | "
|
||||
f"{row['limits']['message']} | {row['tokens']['total']} |"
|
||||
)
|
||||
lines += ["", "## Paired task scores", "",
|
||||
"| Task | " + " | ".join(runs) + " |",
|
||||
"|---|" + "---:|" * len(runs)]
|
||||
for item in paired:
|
||||
lines.append("| " + item["task_id"] + " | " +
|
||||
" | ".join(item["scores"][label] for label in runs) + " |")
|
||||
lines += ["", "Missing means no terminal row; null means a terminal row without a numeric score.",
|
||||
"Raw logs remain in each run directory. This control-only baseline makes no communication claim.", ""]
|
||||
(args.out / "REPORT.md").write_text("\n".join(lines))
|
||||
print(f"Wrote {args.out / 'report.json'} and {args.out / 'REPORT.md'}", flush=True)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+339
-38
@@ -19,15 +19,57 @@ from messageboardbench.swe_validation import REMOTE_DOCKER_HOST
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_CONDITIONS = ("control", "board")
|
||||
CLINEPASS_MODELS = {
|
||||
"openai-api/cline/cline-pass/kimi-k3",
|
||||
"openai-api/cline/cline-pass/deepseek-v4-pro",
|
||||
"openai-api/cline/cline-pass/deepseek-v4-flash",
|
||||
}
|
||||
CLINE_API_BASE_URL = "https://api.cline.bot/api/v1"
|
||||
|
||||
|
||||
def model_provider(model: str) -> str:
|
||||
if model.startswith("openrouter/"):
|
||||
return "openrouter"
|
||||
if model in CLINEPASS_MODELS:
|
||||
return "clinepass"
|
||||
raise ValueError(f"unsupported SWE experiment model: {model}")
|
||||
|
||||
|
||||
def plan_models(plan: dict) -> list[str]:
|
||||
models = plan.get("models_by_team")
|
||||
if models is not None:
|
||||
return [models.get(str(team["team"]), plan.get("model"))
|
||||
for team in plan["team_plans"]]
|
||||
return [plan["model"]]
|
||||
|
||||
|
||||
def uses_engineering_sentinel(plan: dict) -> bool:
|
||||
"""Keep the legacy paired-pilot stop rule out of completed-validation runs."""
|
||||
return not str(plan.get("purpose", "")).startswith("swe-board-activation-")
|
||||
purpose = str(plan.get("purpose", ""))
|
||||
return (purpose != "baseline-swebench"
|
||||
and not purpose.startswith("swe-board-activation-")
|
||||
and not purpose.startswith("swe-scenario-"))
|
||||
|
||||
|
||||
def treatment_metadata(plan: dict) -> dict:
|
||||
"""Describe the actual model-visible intervention without legacy-arm claims."""
|
||||
if str(plan.get("purpose", "")).startswith("swe-scenario-"):
|
||||
return {
|
||||
"conditions": plan["conditions"],
|
||||
"agent_network_mode": plan["agent_network_mode"],
|
||||
"shared_scratch": bool(plan.get("shared_scratch")),
|
||||
"board_interface": plan.get("tool_interface"),
|
||||
"organizer_feedback": plan.get("organizer_feedback_interface"),
|
||||
"token_status_tool": bool(plan.get("token_status_tool")),
|
||||
"system_prompt_change": plan["custom_prompt"],
|
||||
}
|
||||
if plan.get("purpose") == "baseline-swebench":
|
||||
return {
|
||||
"conditions": ["control"],
|
||||
"control": "upstream ImpossibleBench SWE tools with no communication tools",
|
||||
"system_prompt_change": plan["custom_prompt"],
|
||||
"organizer_feedback": None,
|
||||
}
|
||||
if str(plan.get("purpose", "")).startswith("swe-board-activation-"):
|
||||
return {
|
||||
"conditions": ["board"],
|
||||
@@ -56,6 +98,11 @@ def parser() -> argparse.ArgumentParser:
|
||||
p.add_argument("--out", type=Path, required=True)
|
||||
p.add_argument("--plan", type=Path, required=True)
|
||||
p.add_argument("--execute", action="store_true")
|
||||
p.add_argument(
|
||||
"--resume-runner-upgrade",
|
||||
choices=("terminal-grader-null-v1",),
|
||||
help="explicit runner-only upgrade for an existing interrupted run",
|
||||
)
|
||||
return p
|
||||
|
||||
|
||||
@@ -63,22 +110,101 @@ def dump(path: Path, value) -> None:
|
||||
path.write_text(json.dumps(value, indent=2, ensure_ascii=False, default=str) + "\n")
|
||||
|
||||
|
||||
def validate_resume_sources(archive: Path, sources: list[Path]) -> None:
|
||||
def snapshot_scratch(volume: str, destination: Path) -> bool:
|
||||
"""Save a shared Docker volume to the local run directory after a phase."""
|
||||
with destination.open("wb") as output:
|
||||
result = subprocess.run(
|
||||
["docker", "run", "--rm", "--network", "none",
|
||||
"--volume", f"{volume}:/scratch:ro",
|
||||
"aisiuk/inspect-tool-support", "tar", "-C", "/scratch", "-cf", "-", "."],
|
||||
stdout=output, stderr=subprocess.PIPE,
|
||||
)
|
||||
if result.returncode:
|
||||
destination.with_suffix(".error.txt").write_text(
|
||||
f"docker snapshot exited {result.returncode}\n"
|
||||
+ result.stderr.decode(errors="replace")
|
||||
)
|
||||
elif destination.with_suffix(".error.txt").exists():
|
||||
destination.with_suffix(".error.txt").unlink()
|
||||
return result.returncode == 0
|
||||
|
||||
|
||||
def validate_resume_sources(
|
||||
archive: Path,
|
||||
sources: list[Path],
|
||||
*,
|
||||
runner_upgrade: str | None = None,
|
||||
recovery_archive: Path | None = None,
|
||||
) -> None:
|
||||
"""Reject resume when archived or current behavioral source bytes changed."""
|
||||
index = json.loads((archive / "index.json").read_text())
|
||||
current = {str(path.resolve()): path for path in sources}
|
||||
if set(current) != {item["source"] for item in index}:
|
||||
raise RuntimeError("resume source set differs from frozen source snapshot")
|
||||
mismatches = []
|
||||
for item in index:
|
||||
archived = archive / item["archived"]
|
||||
if hashlib.sha256(archived.read_bytes()).hexdigest() != item["sha256"]:
|
||||
raise RuntimeError("resume source snapshot hash mismatch")
|
||||
if hashlib.sha256(current[item["source"]].read_bytes()).hexdigest() != item["sha256"]:
|
||||
raise RuntimeError("current behavioral source differs from frozen resume snapshot")
|
||||
mismatches.append(item)
|
||||
if not mismatches:
|
||||
return
|
||||
runner = str(Path(__file__).resolve())
|
||||
if (runner_upgrade != "terminal-grader-null-v1"
|
||||
or len(mismatches) != 1
|
||||
or mismatches[0]["source"] != runner
|
||||
or recovery_archive is None):
|
||||
raise RuntimeError("current behavioral source differs from frozen resume snapshot")
|
||||
raw = current[runner].read_bytes()
|
||||
static = {
|
||||
"upgrade": runner_upgrade,
|
||||
"source": runner,
|
||||
"base_sha256": mismatches[0]["sha256"],
|
||||
"recovery_sha256": hashlib.sha256(raw).hexdigest(),
|
||||
"archived": "swe_board_experiment.py",
|
||||
"base_index_sha256": hashlib.sha256(
|
||||
(archive / "index.json").read_bytes()
|
||||
).hexdigest(),
|
||||
"reason": "retain completed model trajectories with grader-only null outcomes",
|
||||
}
|
||||
if recovery_archive.exists():
|
||||
recovery = json.loads((recovery_archive / "index.json").read_text())
|
||||
if any(recovery.get(key) != value for key, value in static.items()):
|
||||
raise RuntimeError("recovery source snapshot differs from current runner")
|
||||
if hashlib.sha256(
|
||||
(recovery_archive / static["archived"]).read_bytes()
|
||||
).hexdigest() != static["recovery_sha256"]:
|
||||
raise RuntimeError("recovery source snapshot hash mismatch")
|
||||
for item in recovery["preexisting_evals"]:
|
||||
path = archive.parent / item["path"]
|
||||
if hashlib.sha256(path.read_bytes()).hexdigest() != item["sha256"]:
|
||||
raise RuntimeError("pre-recovery eval changed after runner upgrade")
|
||||
else:
|
||||
recovery = {
|
||||
**static,
|
||||
"created_at": datetime.now(timezone.utc).isoformat(),
|
||||
"preexisting_evals": [
|
||||
{
|
||||
"path": str(path.relative_to(archive.parent)),
|
||||
"sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
|
||||
}
|
||||
for path in sorted((archive.parent / "evals").glob("*.eval"))
|
||||
],
|
||||
}
|
||||
recovery_archive.mkdir()
|
||||
(recovery_archive / static["archived"]).write_bytes(raw)
|
||||
dump(recovery_archive / "index.json", recovery)
|
||||
|
||||
|
||||
def terminal_row(log, evaluated) -> dict:
|
||||
final_score = next(iter(evaluated.scores.values())) if evaluated.scores else None
|
||||
score_metadata = (final_score.metadata or {}) if final_score else {}
|
||||
test_modified_ever = score_metadata.get("test_modified_ever")
|
||||
if test_modified_ever is None:
|
||||
test_modified_ever = evaluated.metadata.get(
|
||||
"_messageboardbench_test_modified_ever"
|
||||
)
|
||||
return {
|
||||
"log": log.location, "log_status": log.status,
|
||||
"sample_id": str(evaluated.id),
|
||||
@@ -89,14 +215,46 @@ def terminal_row(log, evaluated) -> dict:
|
||||
"messages": len(evaluated.messages),
|
||||
"error": evaluated.error.message if evaluated.error else None,
|
||||
"model_patch_captured": (
|
||||
isinstance((final_score.metadata or {}).get("model_patch"), str)
|
||||
isinstance(score_metadata.get("model_patch"), str)
|
||||
if final_score else False
|
||||
),
|
||||
"test_modified_ever": (final_score.metadata or {}).get("test_modified_ever") if final_score else None,
|
||||
"test_modified_ever": test_modified_ever,
|
||||
"grader_diagnostic": evaluated.metadata.get(
|
||||
"_messageboardbench_grader_diagnostic"
|
||||
),
|
||||
"manual_behavior_review": "pending",
|
||||
}
|
||||
|
||||
|
||||
def grader_null_row(row: dict) -> bool:
|
||||
diagnostic = row.get("grader_diagnostic")
|
||||
if not (
|
||||
row.get("log_status") == "success"
|
||||
and row.get("error") is not None
|
||||
and row.get("score") is None
|
||||
and isinstance(diagnostic, dict)
|
||||
and isinstance(diagnostic.get("exit_code"), int)
|
||||
and isinstance(diagnostic.get("target_statuses"), dict)
|
||||
and isinstance(diagnostic.get("eval_script_sha256"), str)
|
||||
and len(diagnostic["eval_script_sha256"]) == 64
|
||||
and isinstance(diagnostic.get("output_tail"), str)
|
||||
):
|
||||
return False
|
||||
statuses = set(diagnostic["target_statuses"].values())
|
||||
return not statuses or bool(statuses & {"MISSING", "ERROR"})
|
||||
|
||||
|
||||
def completed_row(row: dict) -> bool:
|
||||
"""Return whether model work ended in a score or a grader-only null outcome."""
|
||||
return (
|
||||
row.get("log_status") == "success"
|
||||
and (
|
||||
(row.get("error") is None and row.get("score") is not None)
|
||||
or grader_null_row(row)
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def sentinel_failed(
|
||||
rows: list[dict], *, team: int, condition: str, instance_ids: list[str]
|
||||
) -> bool:
|
||||
@@ -117,20 +275,42 @@ def sentinel_failed(
|
||||
|
||||
|
||||
def recover_terminal_rows(out: Path) -> list[dict]:
|
||||
"""Recover completed/error samples written before a process interruption."""
|
||||
"""Recover scored and grader-null outcomes, leaving other failures retryable."""
|
||||
from inspect_ai.log import read_eval_log
|
||||
|
||||
rows = []
|
||||
by_assignment = {}
|
||||
for path in sorted((out / "evals").rglob("*.eval")) if (out / "evals").exists() else []:
|
||||
log = read_eval_log(path, resolve_attachments=True)
|
||||
if log.status not in {"success", "error", "cancelled"}:
|
||||
continue
|
||||
for sample in log.samples or []:
|
||||
if sample.metadata and sample.metadata.get("episode_id"):
|
||||
rows.append(terminal_row(log, sample))
|
||||
keys = [(row["team"], row["condition"], row["sample_id"]) for row in rows]
|
||||
if len(keys) != len(set(keys)):
|
||||
raise RuntimeError("duplicate terminal SWE assignments found during resume")
|
||||
row = terminal_row(log, sample)
|
||||
if completed_row(row):
|
||||
key = (row["team"], row["condition"], row["sample_id"])
|
||||
by_assignment.setdefault(key, []).append(row)
|
||||
rows = []
|
||||
for key, attempts in by_assignment.items():
|
||||
if len(attempts) == 1:
|
||||
rows.append(attempts[0])
|
||||
continue
|
||||
if not all(grader_null_row(row) for row in attempts):
|
||||
raise RuntimeError(f"ambiguous duplicate terminal SWE assignment: {key}")
|
||||
signatures = {
|
||||
(
|
||||
row["grader_diagnostic"]["eval_script_sha256"],
|
||||
json.dumps(
|
||||
row["grader_diagnostic"]["target_statuses"], sort_keys=True
|
||||
),
|
||||
)
|
||||
for row in attempts
|
||||
}
|
||||
if len(signatures) != 1:
|
||||
raise RuntimeError(f"grader-null retries disagree for assignment: {key}")
|
||||
selected = dict(attempts[0])
|
||||
selected["attempt_count"] = len(attempts)
|
||||
selected["redundant_attempt_logs"] = [row["log"] for row in attempts[1:]]
|
||||
rows.append(selected)
|
||||
return rows
|
||||
|
||||
|
||||
@@ -174,7 +354,7 @@ def cleanup_matched_images(
|
||||
dump(path, lifecycle)
|
||||
|
||||
|
||||
def account_budget() -> dict:
|
||||
def openrouter_budget() -> dict:
|
||||
import httpx
|
||||
|
||||
response = httpx.get(
|
||||
@@ -190,6 +370,18 @@ def account_budget() -> dict:
|
||||
}
|
||||
|
||||
|
||||
def account_budget(provider: str) -> dict:
|
||||
if provider == "openrouter":
|
||||
return openrouter_budget()
|
||||
return {
|
||||
"checked_at": datetime.now(timezone.utc).isoformat(),
|
||||
"provider": "clinepass",
|
||||
"account_usage": None,
|
||||
"account_quota": None,
|
||||
"status": "unavailable_from_documented_api",
|
||||
}
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = parser().parse_args(argv)
|
||||
if args.execute and os.environ.get("DOCKER_HOST") != REMOTE_DOCKER_HOST:
|
||||
@@ -199,6 +391,10 @@ def main(argv: list[str] | None = None) -> int:
|
||||
from messageboardbench.swe_board import load_records
|
||||
plan_bytes = args.plan.read_bytes()
|
||||
plan = json.loads(plan_bytes)
|
||||
providers = {model_provider(model) for model in plan_models(plan)}
|
||||
if len(providers) != 1:
|
||||
raise ValueError("use separate experiment plans for OpenRouter and ClinePass populations")
|
||||
provider = next(iter(providers))
|
||||
conditions = tuple(plan.get("conditions", DEFAULT_CONDITIONS))
|
||||
split = plan["dataset"]["split"]
|
||||
records = load_records(plan["dataset"]["revision"], split)
|
||||
@@ -211,6 +407,8 @@ def main(argv: list[str] | None = None) -> int:
|
||||
"remote_docker_host": REMOTE_DOCKER_HOST,
|
||||
"host_mounts": [],
|
||||
}
|
||||
if provider == "clinepass":
|
||||
config["clinepass_base_url"] = CLINE_API_BASE_URL
|
||||
print(json.dumps(config, indent=2), flush=True)
|
||||
if not args.execute:
|
||||
print("Preview only; no Docker container or model request was started.", flush=True)
|
||||
@@ -230,22 +428,33 @@ def main(argv: list[str] | None = None) -> int:
|
||||
load_dotenv(ROOT / ".env")
|
||||
out = args.out.resolve()
|
||||
fresh = not out.exists()
|
||||
if fresh and args.resume_runner_upgrade is not None:
|
||||
raise RuntimeError("runner upgrades apply only to an existing interrupted run")
|
||||
current_budget = account_budget(provider)
|
||||
if (provider == "openrouter"
|
||||
and (current_budget["limit_remaining"] is None
|
||||
or current_budget["limit_remaining"] <= 0)):
|
||||
raise RuntimeError("OpenRouter key has no remaining budget")
|
||||
if fresh:
|
||||
before = current_budget
|
||||
out.mkdir(parents=True)
|
||||
elif not (out / "manifest.json").is_file():
|
||||
raise RuntimeError("resume output lacks a manifest")
|
||||
parameters = plan["parameters"]
|
||||
schedule = plan["schedule"]
|
||||
team_plans = plan["team_plans"]
|
||||
configs = out / "compose"
|
||||
compose_by_assignment = {
|
||||
instance_id: write_compose(
|
||||
records[instance_id], configs, parameters["memory"],
|
||||
)
|
||||
for instance_id in records
|
||||
}
|
||||
scratch_volume = None
|
||||
scratch_identity = None
|
||||
if plan.get("shared_scratch"):
|
||||
identity_path = out / "scratch-volume.json"
|
||||
if fresh:
|
||||
scratch_volume = "messageboardbench-scratch-" + uuid.uuid4().hex
|
||||
scratch_identity = {"name": scratch_volume, "created": False, "removed": False}
|
||||
dump(identity_path, scratch_identity)
|
||||
else:
|
||||
scratch_identity = json.loads(identity_path.read_text())
|
||||
scratch_volume = scratch_identity["name"]
|
||||
if fresh:
|
||||
before = account_budget()
|
||||
dump(out / "manifest.json", config)
|
||||
dump(out / "budget-before.json", before)
|
||||
else:
|
||||
@@ -272,6 +481,8 @@ def main(argv: list[str] | None = None) -> int:
|
||||
Path(upstream_scorer.__file__),
|
||||
Path(upstream_tasks.__file__),
|
||||
]
|
||||
if str(plan.get("purpose", "")).startswith("swe-scenario-"):
|
||||
sources.append(ROOT / "experiments/baseline-swebench/assets/flit_core-3.7.1-py3-none-any.whl")
|
||||
if str(plan.get("purpose", "")).startswith("swe-board-activation-"):
|
||||
sources.extend([
|
||||
ROOT / "scripts/swe_activation_report.py",
|
||||
@@ -291,17 +502,43 @@ def main(argv: list[str] | None = None) -> int:
|
||||
)
|
||||
dump(archive / "index.json", index)
|
||||
else:
|
||||
validate_resume_sources(archive, [*sources, args.plan])
|
||||
validate_resume_sources(
|
||||
archive,
|
||||
[*sources, args.plan],
|
||||
runner_upgrade=args.resume_runner_upgrade,
|
||||
recovery_archive=out / "resume-source-snapshot",
|
||||
)
|
||||
|
||||
configs = out / "compose"
|
||||
compose_by_assignment = {
|
||||
instance_id: write_compose(
|
||||
records[instance_id], configs, parameters["memory"],
|
||||
agent_network_mode=plan.get("agent_network_mode"),
|
||||
scratch_volume=scratch_volume,
|
||||
)
|
||||
for instance_id in records
|
||||
}
|
||||
|
||||
if scratch_identity and not scratch_identity["removed"]:
|
||||
if scratch_identity["created"]:
|
||||
subprocess.run(["docker", "volume", "inspect", scratch_volume], check=True,
|
||||
capture_output=True, text=True)
|
||||
else:
|
||||
subprocess.run(["docker", "volume", "create", scratch_volume], check=True,
|
||||
capture_output=True, text=True)
|
||||
scratch_identity["created"] = True
|
||||
dump(identity_path, scratch_identity)
|
||||
|
||||
boards = {}
|
||||
has_board = "board" in conditions
|
||||
feedback = None
|
||||
if fresh:
|
||||
identities = []
|
||||
for team_plan in team_plans:
|
||||
team = team_plan["team"]
|
||||
run_id = uuid.uuid4().hex
|
||||
path = out / f"board-team-{team}.sqlite"
|
||||
initialize_board(path, run_id)
|
||||
run_id = uuid.uuid4().hex if has_board else None
|
||||
if run_id is not None:
|
||||
initialize_board(out / f"board-team-{team}.sqlite", run_id)
|
||||
episodes = {condition: {instance_id: "worker-" + uuid.uuid4().hex[:12]
|
||||
for instance_id in team_plan["instance_ids"]}
|
||||
for condition in conditions}
|
||||
@@ -317,10 +554,11 @@ def main(argv: list[str] | None = None) -> int:
|
||||
if json.loads((out / "schedule.json").read_text()) != schedule:
|
||||
raise RuntimeError("resume schedule differs")
|
||||
for identity in identities:
|
||||
boards[identity["team"]] = {
|
||||
"path": out / f"board-team-{identity['team']}.sqlite",
|
||||
"run_id": identity["board_run_id"],
|
||||
}
|
||||
if identity["board_run_id"] is not None:
|
||||
boards[identity["team"]] = {
|
||||
"path": out / f"board-team-{identity['team']}.sqlite",
|
||||
"run_id": identity["board_run_id"],
|
||||
}
|
||||
if plan.get("organizer_feedback_interface"):
|
||||
feedback_identity = json.loads((out / "feedback-identity.json").read_text())
|
||||
feedback = {
|
||||
@@ -341,6 +579,12 @@ def main(argv: list[str] | None = None) -> int:
|
||||
pending = [instance_id for instance_id in selected
|
||||
if (team, condition, instance_id) not in terminal]
|
||||
if not pending:
|
||||
if scratch_volume and not scratch_identity["removed"]:
|
||||
snapshot = out / f"scratch-after-phase-{phase}.tar"
|
||||
if (not snapshot.exists()
|
||||
or snapshot.with_suffix(".error.txt").exists()):
|
||||
if not snapshot_scratch(scratch_volume, snapshot):
|
||||
raise RuntimeError(f"shared scratch snapshot failed: phase {phase}")
|
||||
if uses_engineering_sentinel(plan) and phase <= 2 and sentinel_failed(
|
||||
results, team=team, condition=condition, instance_ids=selected
|
||||
):
|
||||
@@ -359,7 +603,7 @@ def main(argv: list[str] | None = None) -> int:
|
||||
offset = selected.index(instance_id) + 1
|
||||
slot = team_plans[team - 1]["instance_ids"].index(instance_id) + 1
|
||||
episode_id = identity["episodes"][condition][instance_id]
|
||||
board = boards[team]
|
||||
board = boards.get(team)
|
||||
sample = sample_from_record(
|
||||
records[instance_id], compose_by_assignment[instance_id],
|
||||
)
|
||||
@@ -383,10 +627,18 @@ def main(argv: list[str] | None = None) -> int:
|
||||
tool_interface=plan.get("tool_interface"),
|
||||
feedback_path=feedback["path"] if feedback else None,
|
||||
feedback_run_id=feedback["run_id"] if feedback else None,
|
||||
enable_feedback=bool(plan.get("organizer_feedback_interface")),
|
||||
enable_token_checker=bool(plan.get("token_status_tool")),
|
||||
),
|
||||
scorer=swe_board_scorer(
|
||||
memory=parameters["memory"],
|
||||
timeout_seconds=parameters["scorer_timeout_seconds"],
|
||||
pin_grader_image=str(plan.get("purpose", "")).startswith(
|
||||
"swe-scenario-"
|
||||
),
|
||||
strict_grader_statuses=str(plan.get("purpose", "")).startswith(
|
||||
"swe-scenario-"
|
||||
),
|
||||
),
|
||||
message_limit=parameters["message_limit"],
|
||||
metadata={**config, "condition": condition, "team": team,
|
||||
@@ -402,13 +654,28 @@ def main(argv: list[str] | None = None) -> int:
|
||||
else:
|
||||
dump(input_path, inputs)
|
||||
print(f"Starting phase {phase}: team {team} {condition} cohort {cohort}", flush=True)
|
||||
model = plan.get("models_by_team", {}).get(str(team), plan.get("model"))
|
||||
model_args = {"strict_tools": False}
|
||||
model_base_url = None
|
||||
if model_provider(model) == "clinepass":
|
||||
if (parameters["reasoning_effort"] is not None
|
||||
or parameters["reasoning_tokens"] is not None):
|
||||
raise ValueError(
|
||||
"ClinePass plans must set reasoning_effort and reasoning_tokens "
|
||||
"to null; their existing OpenRouter settings are not portable"
|
||||
)
|
||||
model_base_url = CLINE_API_BASE_URL
|
||||
model_args.update(responses_api=False, stream=True)
|
||||
logs = inspect_eval(
|
||||
tasks,
|
||||
model=plan.get("models_by_team", {}).get(str(team), plan.get("model")),
|
||||
model_args={"strict_tools": False},
|
||||
model=model,
|
||||
model_base_url=model_base_url,
|
||||
model_args=model_args,
|
||||
log_dir=str(out / "evals"),
|
||||
max_tasks=len(tasks), max_samples=len(tasks), max_sandboxes=len(tasks),
|
||||
max_connections=len(tasks), max_retries=1, retry_on_error=0,
|
||||
max_connections=parameters.get("max_connections", len(tasks)),
|
||||
max_retries=parameters.get("request_retries", 1),
|
||||
retry_on_error=parameters.get("sample_retries", 0),
|
||||
fail_on_error=False, time_limit=parameters["time_limit_seconds"],
|
||||
token_limit=parameters["token_limit"],
|
||||
reasoning_effort=parameters["reasoning_effort"],
|
||||
@@ -419,17 +686,32 @@ def main(argv: list[str] | None = None) -> int:
|
||||
for log in logs:
|
||||
for evaluated in log.samples or []:
|
||||
new.append(terminal_row(log, evaluated))
|
||||
results.extend(new)
|
||||
terminal.update((row["team"], row["condition"], row["sample_id"]) for row in new)
|
||||
dump(out / "results.json", results)
|
||||
completed = [row for row in new if completed_row(row)]
|
||||
results.extend(completed)
|
||||
terminal.update(
|
||||
(row["team"], row["condition"], row["sample_id"])
|
||||
for row in completed
|
||||
)
|
||||
dump(out / "results.json", [
|
||||
*results,
|
||||
*(row for row in new if not completed_row(row)),
|
||||
])
|
||||
dump(out / f"board-after-phase-{phase}.json", {
|
||||
"posts": [post for value in boards.values() for post in export_board(value["path"], value["run_id"])["posts"]],
|
||||
"audit": [event for value in boards.values() for event in export_board(value["path"], value["run_id"])["audit"]],
|
||||
})
|
||||
if scratch_volume:
|
||||
if not snapshot_scratch(scratch_volume, out / f"scratch-after-phase-{phase}.tar"):
|
||||
raise RuntimeError(f"shared scratch snapshot failed: phase {phase}")
|
||||
if len(new) != len(tasks):
|
||||
raise RuntimeError("phase did not produce one terminal record per assignment")
|
||||
failed = [row["sample_id"] for row in new if not completed_row(row)]
|
||||
if failed:
|
||||
raise RuntimeError(
|
||||
"phase produced failed assignments eligible for retry: "
|
||||
+ ", ".join(failed)
|
||||
)
|
||||
# The first adjacent control/board pair is an engineering sentinel.
|
||||
# Later sample errors are terminal outcomes and do not trigger reruns.
|
||||
if uses_engineering_sentinel(plan) and phase <= 2 and sentinel_failed(
|
||||
results, team=team, condition=condition, instance_ids=selected
|
||||
):
|
||||
@@ -452,6 +734,20 @@ def main(argv: list[str] | None = None) -> int:
|
||||
raise
|
||||
finally:
|
||||
dump(out / "status.json", status)
|
||||
if scratch_volume and not scratch_identity["removed"]:
|
||||
saved = snapshot_scratch(scratch_volume, out / "scratch-final.tar")
|
||||
if not saved and status["status"] == "completed":
|
||||
status.update(status="interrupted", error="shared scratch final snapshot failed")
|
||||
dump(out / "status.json", status)
|
||||
if saved and status["status"] == "completed":
|
||||
removed = subprocess.run(
|
||||
["docker", "volume", "rm", scratch_volume],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
scratch_identity["removed"] = removed.returncode == 0
|
||||
if removed.returncode:
|
||||
scratch_identity["remove_error"] = removed.stderr
|
||||
dump(identity_path, scratch_identity)
|
||||
snapshots = [export_board(value["path"], value["run_id"]) for value in boards.values()]
|
||||
dump(out / "board-final.json", {
|
||||
"run_ids": [value["run_id"] for value in snapshots],
|
||||
@@ -463,11 +759,16 @@ def main(argv: list[str] | None = None) -> int:
|
||||
feedback["path"], feedback["run_id"]
|
||||
))
|
||||
try:
|
||||
after = account_budget()
|
||||
after["usage_delta"] = after["usage"] - before["usage"]
|
||||
after = account_budget(provider)
|
||||
if provider == "openrouter":
|
||||
after["usage_delta"] = after["usage"] - before["usage"]
|
||||
else:
|
||||
after["usage_delta"] = None
|
||||
except Exception as error:
|
||||
after = {"accounting_error": repr(error), "usage_delta": None}
|
||||
dump(out / "budget-after.json", after)
|
||||
if status["status"] != "completed":
|
||||
raise RuntimeError(status["error"])
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ from messageboardbench.swe_prerequisites import (
|
||||
validate_task_manifest,
|
||||
)
|
||||
from messageboardbench.swe_validation import (
|
||||
GRADER_ENVIRONMENT,
|
||||
ValidationError,
|
||||
docker_preflight,
|
||||
manifest as trial_manifest,
|
||||
@@ -96,6 +97,8 @@ def main() -> int:
|
||||
plan = json.loads(args.plan.read_text())
|
||||
if plan.get("plan_sha256") != plan_hash(plan):
|
||||
raise SystemExit("frozen plan self-hash mismatch")
|
||||
if plan.get("parameters", {}).get("grader_environment") != GRADER_ENVIRONMENT:
|
||||
raise SystemExit("frozen plan grader environment mismatch")
|
||||
declared = (ROOT / plan["environment_validation"]["index_path"]).resolve()
|
||||
out = args.out.resolve()
|
||||
if declared != out / "index.json":
|
||||
|
||||
Reference in new issue
Block a user