Fail fast on invalid SWE prerequisite tasks

This commit is contained in:
pj committed 2026-09-15 17:16:54 +05:30
1 parent b3c935fbfa
commit aeaae4291f
6 files changed
+173 -45

No files matched your search

@@ -4,6 +4,12 @@ This frozen developmental bundle reuses v2's ten tasks and changes only the poli
suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out
instruction. It creates fresh identities and stores when executed. instruction. It creates fresh identities and stores when executed.
The bundle is currently blocked. The first real prerequisite run established that
`django__django-15315` has an unusable conflicting evaluator: its patch raises a
`NameError` during import and every target is `MISSING`. No behavioral model call
started. The task set must be replaced through a frozen deterministic candidate-pool
screen, rather than by an ad hoc substitution.
Validate the bundle offline: Validate the bundle offline:
```sh ```sh
@@ -1,10 +1,12 @@
{ {
"schema_version": 1, "schema_version": 1,
"status": "ready", "status": "blocked",
"experiment_id": "swe-population-pilot-10-v3", "experiment_id": "swe-population-pilot-10-v3",
"purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.", "purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.",
"remote_docker_host": "ssh://[email protected]", "remote_docker_host": "ssh://[email protected]",
"blockers": [], "blockers": [
"The reused django__django-15315 conflicting evaluator crashes during import and records every target as MISSING. Replace the exact-ten selection through a frozen deterministic candidate-pool screening design before execution."
],
"outputs": { "outputs": {
"run_dir": "logs/swe-population-pilot-10-v3/run", "run_dir": "logs/swe-population-pilot-10-v3/run",
"report_dir": "logs/swe-population-pilot-10-v3/report", "report_dir": "logs/swe-population-pilot-10-v3/report",
@@ -33,5 +35,5 @@
"argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"] "argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"]
} }
], ],
"manifest_sha256": "5e7201286c193c87d01f308a7d65615c6f99c2b10412091cd983ebae3dd33aef" "manifest_sha256": "e648799bad82182e46c15dccc6f14280d565e17fe7d19d2ce1bf4d7bacd7e28e"
} }
@@ -1,6 +1,7 @@
root := "../.." root := "../.."
start: start:
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 --validate-only
cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3
@@ -9,15 +9,18 @@ from pathlib import Path
import subprocess import subprocess
from messageboardbench.swe_board import load_records, plan_hash from messageboardbench.swe_board import load_records, plan_hash
from messageboardbench.swe_prerequisites import validate_environment_index_for_records from messageboardbench.swe_prerequisites import (
validate_environment_index_for_records,
validate_task_manifest,
)
from messageboardbench.swe_validation import ( from messageboardbench.swe_validation import (
ValidationError, ValidationError,
docker_preflight, docker_preflight,
load_pair,
manifest as trial_manifest, manifest as trial_manifest,
run_trial, run_trial,
swebench_spec, swebench_spec,
validate_expected_matrix, validate_expected_matrix,
validate_pair,
) )
@@ -40,6 +43,51 @@ def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) -
pulled.add(image) pulled.add(image)
def load_selected_pairs(plan, loader=None):
"""Load each pinned dataset split once, then validate all selected pairs."""
loader = load_records if loader is None else loader
revision = plan["dataset"]["revision"]
original_records = loader(revision, "original")
conflicting_records = loader(revision, "conflicting")
pairs = {}
for instance_id in plan["selection"]["instance_ids"]:
try:
original = original_records[instance_id]
conflicting = conflicting_records[instance_id]
except KeyError as exc:
raise ValidationError(
f"selected task is absent from a pinned dataset split: {instance_id}"
) from exc
validate_pair(original, conflicting)
pairs[instance_id] = (original, conflicting)
return pairs, conflicting_records
def require_resolved_targets(instance_id: str, result) -> None:
"""Reject an unusable validation cell before another cell is run."""
statuses = result.target_statuses
if not statuses or any(status in {"MISSING", "ERROR"} for status in statuses.values()):
raise ValidationError(
f"validation contains missing/error targets: {instance_id} "
f"{result.split}/{result.mode}"
)
def validate_existing_manifests(plan, out: Path, conflicting_records) -> None:
"""Fail on the first invalid resume artifact before Docker work continues."""
selected = plan["selection"]["instance_ids"]
for position, instance_id in enumerate(selected, 1):
manifest_path = out / instance_id.replace("/", "_") / "manifest.json"
if manifest_path.exists():
print(
f"[{position}/{len(selected)}] {instance_id}: validating existing manifest",
flush=True,
)
validate_task_manifest(
plan, instance_id, manifest_path, conflicting_records[instance_id]
)
def main() -> int: def main() -> int:
parser = argparse.ArgumentParser(description=__doc__) parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--plan", type=Path, required=True) parser.add_argument("--plan", type=Path, required=True)
@@ -52,30 +100,39 @@ def main() -> int:
out = args.out.resolve() out = args.out.resolve()
if declared != out / "index.json": if declared != out / "index.json":
raise SystemExit("--out does not match the frozen validation index location") raise SystemExit("--out does not match the frozen validation index location")
pairs, conflicting_records = load_selected_pairs(plan)
if declared.is_file(): if declared.is_file():
records = load_records(plan["dataset"]["revision"], "conflicting") result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
result = validate_environment_index_for_records(plan, ROOT, records)
print(json.dumps({"status": "already validated", **result}, indent=2)) print(json.dumps({"status": "already validated", **result}, indent=2))
return 0 return 0
validate_existing_manifests(plan, out, conflicting_records)
docker_preflight(os.environ) docker_preflight(os.environ)
out.mkdir(parents=True, exist_ok=True) out.mkdir(parents=True, exist_ok=True)
entries = {} entries = {}
pulled_images: set[str] = set() pulled_images: set[str] = set()
for instance_id in plan["selection"]["instance_ids"]: selected = plan["selection"]["instance_ids"]
for position, instance_id in enumerate(selected, 1):
prefix = f"[{position}/{len(selected)}] {instance_id}"
task_dir = out / instance_id.replace("/", "_") task_dir = out / instance_id.replace("/", "_")
manifest_path = task_dir / "manifest.json" manifest_path = task_dir / "manifest.json"
if not manifest_path.exists(): original, conflicting = pairs[instance_id]
original, conflicting = load_pair(plan["dataset"]["revision"], instance_id) if manifest_path.exists():
print(f"{prefix}: reusing validated manifest", flush=True)
else:
print(f"{prefix}: pulling exact image", flush=True)
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ) pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
results = [ results = []
run_trial(record, split=split, mode=mode, out_dir=task_dir, for split, record in (("original", original), ("conflicting", conflicting)):
environ=os.environ, for mode in ("nochange", "oracle"):
memory=plan["parameters"]["memory"], print(f"{prefix}: running {split}/{mode}", flush=True)
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"]) result = run_trial(
for split, record in (("original", original), ("conflicting", conflicting)) record, split=split, mode=mode, out_dir=task_dir,
for mode in ("nochange", "oracle") environ=os.environ, memory=plan["parameters"]["memory"],
] timeout_seconds=plan["parameters"]["scorer_timeout_seconds"],
)
require_resolved_targets(instance_id, result)
results.append(result)
validate_expected_matrix(results) validate_expected_matrix(results)
value = trial_manifest( value = trial_manifest(
plan["dataset"]["revision"], instance_id, original, conflicting, results plan["dataset"]["revision"], instance_id, original, conflicting, results
@@ -83,6 +140,7 @@ def main() -> int:
with manifest_path.open("x") as handle: with manifest_path.open("x") as handle:
json.dump(value, handle, indent=2, sort_keys=True) json.dump(value, handle, indent=2, sort_keys=True)
handle.write("\n") handle.write("\n")
print(f"{prefix}: validation passed", flush=True)
entries[instance_id] = { entries[instance_id] = {
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path) "path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
} }
@@ -92,8 +150,7 @@ def main() -> int:
with declared.open("x") as handle: with declared.open("x") as handle:
json.dump(index, handle, indent=2, sort_keys=True) json.dump(index, handle, indent=2, sort_keys=True)
handle.write("\n") handle.write("\n")
records = load_records(plan["dataset"]["revision"], "conflicting") result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
result = validate_environment_index_for_records(plan, ROOT, records)
print(json.dumps({"status": "validated", **result}, indent=2)) print(json.dumps({"status": "validated", **result}, indent=2))
return 0 return 0
+35 -24
View File
@@ -44,17 +44,31 @@ def validate_environment_index_for_records(
or set(index.get("manifests", {})) != set(selected)): or set(index.get("manifests", {})) != set(selected)):
raise ValueError("environment validation index does not match the frozen plan") raise ValueError("environment validation index does not match the frozen plan")
evidence = [] evidence = []
expected_cells = {
("original", "nochange"): False,
("original", "oracle"): True,
("conflicting", "nochange"): False,
("conflicting", "oracle"): False,
}
for instance_id in selected: for instance_id in selected:
entry = index["manifests"][instance_id] entry = index["manifests"][instance_id]
manifest_path = _local(root, entry.get("path")) manifest_path = _local(root, entry.get("path"))
if _sha(manifest_path) != entry.get("sha256"): if _sha(manifest_path) != entry.get("sha256"):
raise ValueError(f"validation manifest hash mismatch: {instance_id}") raise ValueError(f"validation manifest hash mismatch: {instance_id}")
record = records.get(instance_id) if records is not None else None
if records is not None and record is None:
raise ValueError(f"frozen validation record missing: {instance_id}")
validate_task_manifest(plan, instance_id, manifest_path, record)
evidence.append({
"instance_id": instance_id,
"manifest_path": str(manifest_path),
"manifest_sha256": entry["sha256"],
})
return {"index_path": str(index_path), "index_sha256": _sha(index_path),
"validated_instances": evidence}
def validate_task_manifest(
plan: Mapping[str, Any],
instance_id: str,
manifest_path: Path,
record: Mapping[str, Any] | None = None,
) -> dict[str, Any]:
"""Validate one task manifest, including partial evidence during resume."""
manifest = json.loads(manifest_path.read_text()) manifest = json.loads(manifest_path.read_text())
if (manifest.get("schema_version") != 1 if (manifest.get("schema_version") != 1
or manifest.get("dataset") != DATASET or manifest.get("dataset") != DATASET
@@ -62,10 +76,7 @@ def validate_environment_index_for_records(
or manifest.get("instance_id") != instance_id or manifest.get("instance_id") != instance_id
or manifest.get("network") != "none"): or manifest.get("network") != "none"):
raise ValueError(f"validation manifest identity mismatch: {instance_id}") raise ValueError(f"validation manifest identity mismatch: {instance_id}")
if records is not None: if record is not None:
record = records.get(instance_id)
if record is None:
raise ValueError(f"frozen validation record missing: {instance_id}")
canonical = hashlib.sha256(json.dumps( canonical = hashlib.sha256(json.dumps(
dict(record), sort_keys=True, separators=(",", ":") dict(record), sort_keys=True, separators=(",", ":")
).encode()).hexdigest() ).encode()).hexdigest()
@@ -90,16 +101,15 @@ def validate_environment_index_for_records(
results = manifest.get("results") results = manifest.get("results")
if not isinstance(results, list): if not isinstance(results, list):
raise ValueError(f"validation results missing: {instance_id}") raise ValueError(f"validation results missing: {instance_id}")
expected_cells = {
("original", "nochange"): False,
("original", "oracle"): True,
("conflicting", "nochange"): False,
("conflicting", "oracle"): False,
}
cells = {(row.get("split"), row.get("mode")): row for row in results} cells = {(row.get("split"), row.get("mode")): row for row in results}
if set(cells) != set(expected_cells) or len(results) != 4: if set(cells) != set(expected_cells) or len(results) != 4:
raise ValueError(f"validation matrix incomplete: {instance_id}") raise ValueError(f"validation matrix incomplete: {instance_id}")
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
for row in results}
if len(identities) != 1 or any(
cells[cell].get("resolved") is not expected
for cell, expected in expected_cells.items()
):
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
if any( if any(
not row.get("target_statuses") not row.get("target_statuses")
or any(status in {"MISSING", "ERROR"} or any(status in {"MISSING", "ERROR"}
@@ -107,14 +117,15 @@ def validate_environment_index_for_records(
for row in results for row in results
): ):
raise ValueError(f"validation contains missing/error targets: {instance_id}") raise ValueError(f"validation contains missing/error targets: {instance_id}")
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
for row in results}
if len(identities) != 1 or any(
cells[cell].get("resolved") is not expected
for cell, expected in expected_cells.items()
):
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
for row in results: for row in results:
output = manifest_path.parent / str(row.get("output_file", "")) output = manifest_path.parent / str(row.get("output_file", ""))
if not output.is_file() or _sha(output) != row.get("output_sha256"): if not output.is_file() or _sha(output) != row.get("output_sha256"):
raise ValueError(f"validation output hash mismatch: {instance_id}") raise ValueError(f"validation output hash mismatch: {instance_id}")
evidence.append({ return manifest
"instance_id": instance_id,
"manifest_path": str(manifest_path),
"manifest_sha256": entry["sha256"],
})
return {"index_path": str(index_path), "index_sha256": _sha(index_path),
"validated_instances": evidence}
+52 -1
View File
@@ -5,11 +5,18 @@ import json
import pytest import pytest
from messageboardbench.swe_validation import ValidationError
from messageboardbench.swe_prerequisites import ( from messageboardbench.swe_prerequisites import (
validate_environment_index, validate_environment_index,
validate_environment_index_for_records, validate_environment_index_for_records,
validate_task_manifest,
)
from scripts.validate_swe_population_prerequisites import (
load_selected_pairs,
pull_image_once,
require_resolved_targets,
validate_existing_manifests,
) )
from scripts.validate_swe_population_prerequisites import pull_image_once
def write(path, value): def write(path, value):
@@ -105,3 +112,47 @@ def test_image_is_pulled_once_before_any_inspection_or_trial():
["docker", "pull", "image:tag"], ["docker", "pull", "image:tag"],
["docker", "image", "inspect", "image:tag"], ["docker", "image", "inspect", "image:tag"],
] ]
def test_selected_pairs_load_each_dataset_split_exactly_once():
calls = []
common = {
"instance_id": "task", "repo": "org/repo", "version": "1",
"base_commit": "base", "patch": "oracle", "original_test_patch": "original",
"FAIL_TO_PASS": ["target"], "PASS_TO_PASS": [],
}
rows = {
"original": {"task": {**common, "test_patch": "original"}},
"conflicting": {"task": {**common, "test_patch": "conflict"}},
}
def loader(revision, split):
calls.append((revision, split))
return rows[split]
plan = {"dataset": {"revision": "1" * 40},
"selection": {"instance_ids": ["task"]}}
pairs, conflicting = load_selected_pairs(plan, loader=loader)
assert calls == [("1" * 40, "original"), ("1" * 40, "conflicting")]
assert pairs["task"] == (rows["original"]["task"], rows["conflicting"]["task"])
assert conflicting is rows["conflicting"]
def test_new_cell_rejects_missing_targets_immediately():
result = __import__("types").SimpleNamespace(
split="conflicting", mode="oracle", target_statuses={"target": "MISSING"}
)
with pytest.raises(ValidationError, match="missing/error targets.*conflicting/oracle"):
require_resolved_targets("task", result)
def test_resume_rejects_existing_missing_manifest_before_reuse(tmp_path):
plan, manifest_path, record = fixture(tmp_path)
manifest = json.loads(manifest_path.read_text())
manifest["results"][0]["target_statuses"] = {"target": "ERROR"}
task_dir = tmp_path / "task"
for row in manifest["results"]:
write(task_dir / row["output_file"], "test output")
write(task_dir / "manifest.json", manifest)
with pytest.raises(ValueError, match="missing/error targets"):
validate_existing_manifests(plan, tmp_path, {"task": record})