mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 11:07:07 +00:00
Add independent-agent SWE prompt-ablation pilot
This commit is contained in:
1 parent
b896955d3e
commit
b3c935fbfa
13 files changed
+673
-12
No files matched your search
@@ -9,6 +9,7 @@ from pathlib import Path
|
||||
from inspect_ai.log import read_eval_log
|
||||
|
||||
from messageboardbench.swe_board import plan_hash
|
||||
from messageboardbench.swe_validation import swebench_spec
|
||||
from messageboardbench.swe_reporting import paired_analysis
|
||||
|
||||
|
||||
@@ -52,6 +53,7 @@ def main() -> int:
|
||||
system_prompts = {}
|
||||
scorer_checks = []
|
||||
tool_checks = []
|
||||
prompt_checks = []
|
||||
log_cache = {}
|
||||
for row in rows:
|
||||
directory = args.export / row["report_directory"]
|
||||
@@ -79,6 +81,25 @@ def main() -> int:
|
||||
samples = [sample for sample in log.samples or [] if sample.uuid == row["sample_uuid"]]
|
||||
model_events = ([event for event in samples[0].events if event.event == "model"]
|
||||
if len(samples) == 1 else [])
|
||||
if len(samples) == 1:
|
||||
from impossiblebench.swebench_agent_full import _get_default_template
|
||||
_, directives, test_command = swebench_spec(samples[0].metadata)
|
||||
test_info = (
|
||||
"TEST COMMAND THAT WILL BE RUN FOR EVALUATION:\n"
|
||||
+ test_command + " " + " ".join(directives) + "\n"
|
||||
)
|
||||
expected_system = _get_default_template(
|
||||
test_info, manifest["parameters"]["max_attempts"],
|
||||
manifest["upstream_system_prompt"]["prompt_suffix"],
|
||||
)
|
||||
prompt_checks.append({
|
||||
"episode_id": row["episode_id"],
|
||||
"system_prompt_exact": system == [expected_system],
|
||||
})
|
||||
else:
|
||||
prompt_checks.append({
|
||||
"episode_id": row["episode_id"], "sample_present": False,
|
||||
})
|
||||
for event in model_events:
|
||||
tools = {tool.name: tool.model_dump(exclude_none=True) for tool in event.tools or []}
|
||||
public = {"send_message", "read_messages"}
|
||||
@@ -124,6 +145,12 @@ def main() -> int:
|
||||
system_prompts.get((team, task, "control")) == system_prompts.get((team, task, "board"))
|
||||
for team, _, task in expected
|
||||
)
|
||||
checks["rendered_system_prompts_exact"] = (
|
||||
bool(prompt_checks) and all(
|
||||
value for row in prompt_checks for name, value in row.items()
|
||||
if name != "episode_id"
|
||||
) if manifest.get("organizer_feedback_interface") else True
|
||||
)
|
||||
sources = json.loads((args.run / "source-snapshot/index.json").read_text())
|
||||
checks["source_snapshot_hashes"] = all(
|
||||
sha(args.run / "source-snapshot" / item["archived"]) == item["sha256"]
|
||||
@@ -165,8 +192,11 @@ def main() -> int:
|
||||
for name, value in row.items() if name != "episode_id" and not value)
|
||||
failures.extend(f"{row['episode_id']}:{name}" for row in tool_checks
|
||||
for name, value in row.items() if name != "episode_id" and not value)
|
||||
failures.extend(f"{row['episode_id']}:{name}" for row in prompt_checks
|
||||
for name, value in row.items() if name != "episode_id" and not value)
|
||||
result = {"checks": checks, "scorer_checks": scorer_checks,
|
||||
"tool_checks": tool_checks, "failures": failures}
|
||||
"tool_checks": tool_checks, "prompt_checks": prompt_checks,
|
||||
"failures": failures}
|
||||
with args.out.open("x") as handle:
|
||||
json.dump(result, handle, indent=2)
|
||||
handle.write("\n")
|
||||
|
||||
@@ -7,7 +7,12 @@ import json
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
|
||||
from messageboardbench.swe_board import build_population_plan, load_records, plan_hash
|
||||
from messageboardbench.swe_board import (
|
||||
NO_STOP_PROMPT_POLICY,
|
||||
build_population_plan,
|
||||
load_records,
|
||||
plan_hash,
|
||||
)
|
||||
from messageboardbench.board import MESSAGEBOARD_V2_INTERFACE_VERSION
|
||||
from messageboardbench.swe_validation import DATASET
|
||||
|
||||
@@ -35,11 +40,25 @@ def main() -> int:
|
||||
"--messageboard-v2", action="store_true",
|
||||
help="freeze the send_message/read_messages plus organizer-feedback interface",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--reuse-selection-plan", type=Path,
|
||||
help="reuse the exact selected task IDs from an earlier frozen plan",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--no-stop-prompt", action="store_true",
|
||||
help="retain DO NOT MODIFY THE TESTS but omit the extra stop/carve-out text",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
if args.messageboard_v2 and args.sample_size is None:
|
||||
parser.error("--messageboard-v2 requires --sample-size")
|
||||
if args.exclude_plan is not None and args.sample_size is None:
|
||||
parser.error("--exclude-plan requires --sample-size")
|
||||
if args.reuse_selection_plan is not None and args.sample_size is not None:
|
||||
parser.error("--reuse-selection-plan cannot be combined with --sample-size")
|
||||
if args.reuse_selection_plan is not None and args.exclude_plan is not None:
|
||||
parser.error("--reuse-selection-plan cannot be combined with --exclude-plan")
|
||||
if args.no_stop_prompt and not args.messageboard_v2:
|
||||
parser.error("--no-stop-prompt requires --messageboard-v2")
|
||||
upstream = (ROOT.parent / "impossiblebench").resolve()
|
||||
commit = subprocess.run(
|
||||
["git", "rev-parse", "HEAD"], cwd=upstream, check=True,
|
||||
@@ -47,6 +66,20 @@ def main() -> int:
|
||||
).stdout.strip()
|
||||
records = load_records(args.revision, "conflicting")
|
||||
selected = None
|
||||
reused_plan_bytes = None
|
||||
reused_plan = None
|
||||
if args.reuse_selection_plan is not None:
|
||||
reused_plan_bytes = args.reuse_selection_plan.read_bytes()
|
||||
reused_plan = json.loads(reused_plan_bytes)
|
||||
selected = reused_plan.get("selection", {}).get("instance_ids")
|
||||
if (reused_plan.get("status") != "frozen"
|
||||
or reused_plan.get("plan_sha256") != plan_hash(reused_plan)
|
||||
or reused_plan.get("dataset") != {
|
||||
"path": DATASET, "revision": args.revision, "split": "conflicting",
|
||||
}
|
||||
or not isinstance(selected, list) or not selected
|
||||
or len(selected) != len(set(selected)) or not set(selected) <= set(records)):
|
||||
parser.error("--reuse-selection-plan is not a valid matching frozen plan")
|
||||
if args.sample_size is not None:
|
||||
if not 1 <= args.sample_size <= len(records):
|
||||
parser.error("--sample-size must be between 1 and the split size")
|
||||
@@ -82,6 +115,7 @@ def main() -> int:
|
||||
upstream_git_commit=commit, teams=args.teams, cohorts=args.cohorts,
|
||||
seed=args.seed, selected_instance_ids=selected,
|
||||
tool_interface=(MESSAGEBOARD_V2_INTERFACE_VERSION if args.messageboard_v2 else None),
|
||||
prompt_policy=(NO_STOP_PROMPT_POLICY if args.no_stop_prompt else None),
|
||||
)
|
||||
if selected is not None:
|
||||
plan["selection"].update({
|
||||
@@ -95,6 +129,16 @@ def main() -> int:
|
||||
} if args.exclude_plan is not None else None),
|
||||
})
|
||||
plan["plan_sha256"] = plan_hash(plan)
|
||||
if reused_plan is not None:
|
||||
plan["selection"].update({
|
||||
"kind": "reused_frozen_subset",
|
||||
"source_plan": {
|
||||
"path": str(args.reuse_selection_plan),
|
||||
"file_sha256": hashlib.sha256(reused_plan_bytes).hexdigest(),
|
||||
"plan_sha256": reused_plan["plan_sha256"],
|
||||
},
|
||||
})
|
||||
plan["plan_sha256"] = plan_hash(plan)
|
||||
args.out.parent.mkdir(parents=True, exist_ok=True)
|
||||
with args.out.open("x") as handle:
|
||||
json.dump(plan, handle, indent=2)
|
||||
|
||||
@@ -11,6 +11,7 @@ import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
import subprocess
|
||||
import uuid
|
||||
|
||||
@@ -178,6 +179,16 @@ def main(argv: list[str] | None = None) -> int:
|
||||
raise SystemExit("installed ImpossibleBench checkout differs from frozen plan")
|
||||
if not str(plan["model"]).startswith("openrouter/"):
|
||||
raise SystemExit("frozen plan model is not an explicit OpenRouter identifier")
|
||||
environment_validation = None
|
||||
if plan.get("purpose") == "population-propensity-control-vs-board-swe-pilot-v3":
|
||||
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
|
||||
if args.execute:
|
||||
environment_validation = validate_environment_index_for_records(
|
||||
plan, ROOT, records
|
||||
)
|
||||
environment_validation["snapshot_path"] = str(
|
||||
(args.out.resolve() / "environment-validation").resolve()
|
||||
)
|
||||
config = {
|
||||
**plan,
|
||||
"frozen_plan": {"path": str(args.plan.resolve()),
|
||||
@@ -196,6 +207,7 @@ def main(argv: list[str] | None = None) -> int:
|
||||
"remote_docker_host": REMOTE_DOCKER_HOST,
|
||||
"container_network": "none",
|
||||
"host_mounts": [],
|
||||
"environment_validation": environment_validation,
|
||||
}
|
||||
print(json.dumps(config, indent=2), flush=True)
|
||||
if not args.execute:
|
||||
@@ -228,6 +240,11 @@ def main(argv: list[str] | None = None) -> int:
|
||||
instance_id: write_compose(records[instance_id], configs, parameters["memory"])
|
||||
for instance_id in records
|
||||
}
|
||||
if fresh and environment_validation is not None:
|
||||
shutil.copytree(
|
||||
Path(environment_validation["index_path"]).parent,
|
||||
out / "environment-validation",
|
||||
)
|
||||
if fresh:
|
||||
before = account_budget()
|
||||
dump(out / "manifest.json", config)
|
||||
@@ -246,6 +263,8 @@ def main(argv: list[str] | None = None) -> int:
|
||||
ROOT / "src/messageboardbench/swe_board.py",
|
||||
ROOT / "src/messageboardbench/board.py",
|
||||
ROOT / "src/messageboardbench/feedback.py",
|
||||
ROOT / "src/messageboardbench/swe_prerequisites.py",
|
||||
ROOT / "scripts/validate_swe_population_prerequisites.py",
|
||||
ROOT / "src/messageboardbench/swe_reporting.py",
|
||||
ROOT / "scripts/swe_population_report.py",
|
||||
ROOT / "scripts/board_report.py",
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Build or validate the no-model SWE readiness index for a frozen pilot."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
|
||||
from messageboardbench.swe_board import load_records, plan_hash
|
||||
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
|
||||
from messageboardbench.swe_validation import (
|
||||
ValidationError,
|
||||
docker_preflight,
|
||||
load_pair,
|
||||
manifest as trial_manifest,
|
||||
run_trial,
|
||||
swebench_spec,
|
||||
validate_expected_matrix,
|
||||
)
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
def sha(path: Path) -> str:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
|
||||
def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) -> None:
|
||||
"""Pull an exact image tag once before run_trial tries to inspect it."""
|
||||
if image in pulled:
|
||||
return
|
||||
result = run(["docker", "pull", image], env=dict(environ), text=True,
|
||||
capture_output=True)
|
||||
if result.returncode:
|
||||
detail = (result.stderr or result.stdout or "").strip()
|
||||
raise ValidationError(f"image pull failed for {image}: {detail}")
|
||||
pulled.add(image)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--plan", type=Path, required=True)
|
||||
parser.add_argument("--out", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
plan = json.loads(args.plan.read_text())
|
||||
if plan.get("plan_sha256") != plan_hash(plan):
|
||||
raise SystemExit("frozen plan self-hash mismatch")
|
||||
declared = (ROOT / plan["environment_validation"]["index_path"]).resolve()
|
||||
out = args.out.resolve()
|
||||
if declared != out / "index.json":
|
||||
raise SystemExit("--out does not match the frozen validation index location")
|
||||
if declared.is_file():
|
||||
records = load_records(plan["dataset"]["revision"], "conflicting")
|
||||
result = validate_environment_index_for_records(plan, ROOT, records)
|
||||
print(json.dumps({"status": "already validated", **result}, indent=2))
|
||||
return 0
|
||||
|
||||
docker_preflight(os.environ)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
entries = {}
|
||||
pulled_images: set[str] = set()
|
||||
for instance_id in plan["selection"]["instance_ids"]:
|
||||
task_dir = out / instance_id.replace("/", "_")
|
||||
manifest_path = task_dir / "manifest.json"
|
||||
if not manifest_path.exists():
|
||||
original, conflicting = load_pair(plan["dataset"]["revision"], instance_id)
|
||||
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
|
||||
results = [
|
||||
run_trial(record, split=split, mode=mode, out_dir=task_dir,
|
||||
environ=os.environ,
|
||||
memory=plan["parameters"]["memory"],
|
||||
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"])
|
||||
for split, record in (("original", original), ("conflicting", conflicting))
|
||||
for mode in ("nochange", "oracle")
|
||||
]
|
||||
validate_expected_matrix(results)
|
||||
value = trial_manifest(
|
||||
plan["dataset"]["revision"], instance_id, original, conflicting, results
|
||||
)
|
||||
with manifest_path.open("x") as handle:
|
||||
json.dump(value, handle, indent=2, sort_keys=True)
|
||||
handle.write("\n")
|
||||
entries[instance_id] = {
|
||||
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
|
||||
}
|
||||
index = {"schema_version": 1, "status": "validated",
|
||||
"plan_sha256": plan["plan_sha256"], "dataset": plan["dataset"],
|
||||
"manifests": entries}
|
||||
with declared.open("x") as handle:
|
||||
json.dump(index, handle, indent=2, sort_keys=True)
|
||||
handle.write("\n")
|
||||
records = load_records(plan["dataset"]["revision"], "conflicting")
|
||||
result = validate_environment_index_for_records(plan, ROOT, records)
|
||||
print(json.dumps({"status": "validated", **result}, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in new issue
Block a user