Add independent-agent SWE prompt-ablation pilot

This commit is contained in:
pj committed 2026-09-15 16:25:27 +05:30
1 parent b896955d3e
commit b3c935fbfa
13 files changed
+673 -12

No files matched your search

+31 -1
View File
@@ -9,6 +9,7 @@ from pathlib import Path
from inspect_ai.log import read_eval_log
from messageboardbench.swe_board import plan_hash
from messageboardbench.swe_validation import swebench_spec
from messageboardbench.swe_reporting import paired_analysis
@@ -52,6 +53,7 @@ def main() -> int:
system_prompts = {}
scorer_checks = []
tool_checks = []
prompt_checks = []
log_cache = {}
for row in rows:
directory = args.export / row["report_directory"]
@@ -79,6 +81,25 @@ def main() -> int:
samples = [sample for sample in log.samples or [] if sample.uuid == row["sample_uuid"]]
model_events = ([event for event in samples[0].events if event.event == "model"]
if len(samples) == 1 else [])
if len(samples) == 1:
from impossiblebench.swebench_agent_full import _get_default_template
_, directives, test_command = swebench_spec(samples[0].metadata)
test_info = (
"TEST COMMAND THAT WILL BE RUN FOR EVALUATION:\n"
+ test_command + " " + " ".join(directives) + "\n"
)
expected_system = _get_default_template(
test_info, manifest["parameters"]["max_attempts"],
manifest["upstream_system_prompt"]["prompt_suffix"],
)
prompt_checks.append({
"episode_id": row["episode_id"],
"system_prompt_exact": system == [expected_system],
})
else:
prompt_checks.append({
"episode_id": row["episode_id"], "sample_present": False,
})
for event in model_events:
tools = {tool.name: tool.model_dump(exclude_none=True) for tool in event.tools or []}
public = {"send_message", "read_messages"}
@@ -124,6 +145,12 @@ def main() -> int:
system_prompts.get((team, task, "control")) == system_prompts.get((team, task, "board"))
for team, _, task in expected
)
checks["rendered_system_prompts_exact"] = (
bool(prompt_checks) and all(
value for row in prompt_checks for name, value in row.items()
if name != "episode_id"
) if manifest.get("organizer_feedback_interface") else True
)
sources = json.loads((args.run / "source-snapshot/index.json").read_text())
checks["source_snapshot_hashes"] = all(
sha(args.run / "source-snapshot" / item["archived"]) == item["sha256"]
@@ -165,8 +192,11 @@ def main() -> int:
for name, value in row.items() if name != "episode_id" and not value)
failures.extend(f"{row['episode_id']}:{name}" for row in tool_checks
for name, value in row.items() if name != "episode_id" and not value)
failures.extend(f"{row['episode_id']}:{name}" for row in prompt_checks
for name, value in row.items() if name != "episode_id" and not value)
result = {"checks": checks, "scorer_checks": scorer_checks,
"tool_checks": tool_checks, "failures": failures}
"tool_checks": tool_checks, "prompt_checks": prompt_checks,
"failures": failures}
with args.out.open("x") as handle:
json.dump(result, handle, indent=2)
handle.write("\n")
+45 -1
View File
@@ -7,7 +7,12 @@ import json
from pathlib import Path
import subprocess
from messageboardbench.swe_board import build_population_plan, load_records, plan_hash
from messageboardbench.swe_board import (
NO_STOP_PROMPT_POLICY,
build_population_plan,
load_records,
plan_hash,
)
from messageboardbench.board import MESSAGEBOARD_V2_INTERFACE_VERSION
from messageboardbench.swe_validation import DATASET
@@ -35,11 +40,25 @@ def main() -> int:
"--messageboard-v2", action="store_true",
help="freeze the send_message/read_messages plus organizer-feedback interface",
)
parser.add_argument(
"--reuse-selection-plan", type=Path,
help="reuse the exact selected task IDs from an earlier frozen plan",
)
parser.add_argument(
"--no-stop-prompt", action="store_true",
help="retain DO NOT MODIFY THE TESTS but omit the extra stop/carve-out text",
)
args = parser.parse_args()
if args.messageboard_v2 and args.sample_size is None:
parser.error("--messageboard-v2 requires --sample-size")
if args.exclude_plan is not None and args.sample_size is None:
parser.error("--exclude-plan requires --sample-size")
if args.reuse_selection_plan is not None and args.sample_size is not None:
parser.error("--reuse-selection-plan cannot be combined with --sample-size")
if args.reuse_selection_plan is not None and args.exclude_plan is not None:
parser.error("--reuse-selection-plan cannot be combined with --exclude-plan")
if args.no_stop_prompt and not args.messageboard_v2:
parser.error("--no-stop-prompt requires --messageboard-v2")
upstream = (ROOT.parent / "impossiblebench").resolve()
commit = subprocess.run(
["git", "rev-parse", "HEAD"], cwd=upstream, check=True,
@@ -47,6 +66,20 @@ def main() -> int:
).stdout.strip()
records = load_records(args.revision, "conflicting")
selected = None
reused_plan_bytes = None
reused_plan = None
if args.reuse_selection_plan is not None:
reused_plan_bytes = args.reuse_selection_plan.read_bytes()
reused_plan = json.loads(reused_plan_bytes)
selected = reused_plan.get("selection", {}).get("instance_ids")
if (reused_plan.get("status") != "frozen"
or reused_plan.get("plan_sha256") != plan_hash(reused_plan)
or reused_plan.get("dataset") != {
"path": DATASET, "revision": args.revision, "split": "conflicting",
}
or not isinstance(selected, list) or not selected
or len(selected) != len(set(selected)) or not set(selected) <= set(records)):
parser.error("--reuse-selection-plan is not a valid matching frozen plan")
if args.sample_size is not None:
if not 1 <= args.sample_size <= len(records):
parser.error("--sample-size must be between 1 and the split size")
@@ -82,6 +115,7 @@ def main() -> int:
upstream_git_commit=commit, teams=args.teams, cohorts=args.cohorts,
seed=args.seed, selected_instance_ids=selected,
tool_interface=(MESSAGEBOARD_V2_INTERFACE_VERSION if args.messageboard_v2 else None),
prompt_policy=(NO_STOP_PROMPT_POLICY if args.no_stop_prompt else None),
)
if selected is not None:
plan["selection"].update({
@@ -95,6 +129,16 @@ def main() -> int:
} if args.exclude_plan is not None else None),
})
plan["plan_sha256"] = plan_hash(plan)
if reused_plan is not None:
plan["selection"].update({
"kind": "reused_frozen_subset",
"source_plan": {
"path": str(args.reuse_selection_plan),
"file_sha256": hashlib.sha256(reused_plan_bytes).hexdigest(),
"plan_sha256": reused_plan["plan_sha256"],
},
})
plan["plan_sha256"] = plan_hash(plan)
args.out.parent.mkdir(parents=True, exist_ok=True)
with args.out.open("x") as handle:
json.dump(plan, handle, indent=2)
+19
View File
@@ -11,6 +11,7 @@ import hashlib
import json
import os
from pathlib import Path
import shutil
import subprocess
import uuid
@@ -178,6 +179,16 @@ def main(argv: list[str] | None = None) -> int:
raise SystemExit("installed ImpossibleBench checkout differs from frozen plan")
if not str(plan["model"]).startswith("openrouter/"):
raise SystemExit("frozen plan model is not an explicit OpenRouter identifier")
environment_validation = None
if plan.get("purpose") == "population-propensity-control-vs-board-swe-pilot-v3":
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
if args.execute:
environment_validation = validate_environment_index_for_records(
plan, ROOT, records
)
environment_validation["snapshot_path"] = str(
(args.out.resolve() / "environment-validation").resolve()
)
config = {
**plan,
"frozen_plan": {"path": str(args.plan.resolve()),
@@ -196,6 +207,7 @@ def main(argv: list[str] | None = None) -> int:
"remote_docker_host": REMOTE_DOCKER_HOST,
"container_network": "none",
"host_mounts": [],
"environment_validation": environment_validation,
}
print(json.dumps(config, indent=2), flush=True)
if not args.execute:
@@ -228,6 +240,11 @@ def main(argv: list[str] | None = None) -> int:
instance_id: write_compose(records[instance_id], configs, parameters["memory"])
for instance_id in records
}
if fresh and environment_validation is not None:
shutil.copytree(
Path(environment_validation["index_path"]).parent,
out / "environment-validation",
)
if fresh:
before = account_budget()
dump(out / "manifest.json", config)
@@ -246,6 +263,8 @@ def main(argv: list[str] | None = None) -> int:
ROOT / "src/messageboardbench/swe_board.py",
ROOT / "src/messageboardbench/board.py",
ROOT / "src/messageboardbench/feedback.py",
ROOT / "src/messageboardbench/swe_prerequisites.py",
ROOT / "scripts/validate_swe_population_prerequisites.py",
ROOT / "src/messageboardbench/swe_reporting.py",
ROOT / "scripts/swe_population_report.py",
ROOT / "scripts/board_report.py",
@@ -0,0 +1,102 @@
"""Build or validate the no-model SWE readiness index for a frozen pilot."""
from __future__ import annotations
import argparse
import hashlib
import json
import os
from pathlib import Path
import subprocess
from messageboardbench.swe_board import load_records, plan_hash
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
from messageboardbench.swe_validation import (
ValidationError,
docker_preflight,
load_pair,
manifest as trial_manifest,
run_trial,
swebench_spec,
validate_expected_matrix,
)
ROOT = Path(__file__).resolve().parents[1]
def sha(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) -> None:
"""Pull an exact image tag once before run_trial tries to inspect it."""
if image in pulled:
return
result = run(["docker", "pull", image], env=dict(environ), text=True,
capture_output=True)
if result.returncode:
detail = (result.stderr or result.stdout or "").strip()
raise ValidationError(f"image pull failed for {image}: {detail}")
pulled.add(image)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--plan", type=Path, required=True)
parser.add_argument("--out", type=Path, required=True)
args = parser.parse_args()
plan = json.loads(args.plan.read_text())
if plan.get("plan_sha256") != plan_hash(plan):
raise SystemExit("frozen plan self-hash mismatch")
declared = (ROOT / plan["environment_validation"]["index_path"]).resolve()
out = args.out.resolve()
if declared != out / "index.json":
raise SystemExit("--out does not match the frozen validation index location")
if declared.is_file():
records = load_records(plan["dataset"]["revision"], "conflicting")
result = validate_environment_index_for_records(plan, ROOT, records)
print(json.dumps({"status": "already validated", **result}, indent=2))
return 0
docker_preflight(os.environ)
out.mkdir(parents=True, exist_ok=True)
entries = {}
pulled_images: set[str] = set()
for instance_id in plan["selection"]["instance_ids"]:
task_dir = out / instance_id.replace("/", "_")
manifest_path = task_dir / "manifest.json"
if not manifest_path.exists():
original, conflicting = load_pair(plan["dataset"]["revision"], instance_id)
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
results = [
run_trial(record, split=split, mode=mode, out_dir=task_dir,
environ=os.environ,
memory=plan["parameters"]["memory"],
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"])
for split, record in (("original", original), ("conflicting", conflicting))
for mode in ("nochange", "oracle")
]
validate_expected_matrix(results)
value = trial_manifest(
plan["dataset"]["revision"], instance_id, original, conflicting, results
)
with manifest_path.open("x") as handle:
json.dump(value, handle, indent=2, sort_keys=True)
handle.write("\n")
entries[instance_id] = {
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
}
index = {"schema_version": 1, "status": "validated",
"plan_sha256": plan["plan_sha256"], "dataset": plan["dataset"],
"manifests": entries}
with declared.open("x") as handle:
json.dump(index, handle, indent=2, sort_keys=True)
handle.write("\n")
records = load_records(plan["dataset"]["revision"], "conflicting")
result = validate_environment_index_for_records(plan, ROOT, records)
print(json.dumps({"status": "validated", **result}, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())