Fail fast on invalid SWE prerequisite tasks

This commit is contained in:
pj committed 2026-09-15 17:16:54 +05:30
1 parent b3c935fbfa
commit aeaae4291f
6 files changed
+211 -83

No files matched your search

@@ -4,6 +4,12 @@ This frozen developmental bundle reuses v2's ten tasks and changes only the poli
suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out
instruction. It creates fresh identities and stores when executed.
The bundle is currently blocked. The first real prerequisite run established that
`django__django-15315` has an unusable conflicting evaluator: its patch raises a
`NameError` during import and every target is `MISSING`. No behavioral model call
started. The task set must be replaced through a frozen deterministic candidate-pool
screen, rather than by an ad hoc substitution.
Validate the bundle offline:
```sh
@@ -1,10 +1,12 @@
{
"schema_version": 1,
"status": "ready",
"status": "blocked",
"experiment_id": "swe-population-pilot-10-v3",
"purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.",
"remote_docker_host": "ssh://[email protected]",
"blockers": [],
"blockers": [
"The reused django__django-15315 conflicting evaluator crashes during import and records every target as MISSING. Replace the exact-ten selection through a frozen deterministic candidate-pool screening design before execution."
],
"outputs": {
"run_dir": "logs/swe-population-pilot-10-v3/run",
"report_dir": "logs/swe-population-pilot-10-v3/report",
@@ -33,5 +35,5 @@
"argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"]
}
],
"manifest_sha256": "5e7201286c193c87d01f308a7d65615c6f99c2b10412091cd983ebae3dd33aef"
"manifest_sha256": "e648799bad82182e46c15dccc6f14280d565e17fe7d19d2ce1bf4d7bacd7e28e"
}
@@ -1,6 +1,7 @@
root := "../.."
start:
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 --validate-only
cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3
@@ -9,15 +9,18 @@ from pathlib import Path
import subprocess
from messageboardbench.swe_board import load_records, plan_hash
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
from messageboardbench.swe_prerequisites import (
validate_environment_index_for_records,
validate_task_manifest,
)
from messageboardbench.swe_validation import (
ValidationError,
docker_preflight,
load_pair,
manifest as trial_manifest,
run_trial,
swebench_spec,
validate_expected_matrix,
validate_pair,
)
@@ -40,6 +43,51 @@ def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) -
pulled.add(image)
def load_selected_pairs(plan, loader=None):
"""Load each pinned dataset split once, then validate all selected pairs."""
loader = load_records if loader is None else loader
revision = plan["dataset"]["revision"]
original_records = loader(revision, "original")
conflicting_records = loader(revision, "conflicting")
pairs = {}
for instance_id in plan["selection"]["instance_ids"]:
try:
original = original_records[instance_id]
conflicting = conflicting_records[instance_id]
except KeyError as exc:
raise ValidationError(
f"selected task is absent from a pinned dataset split: {instance_id}"
) from exc
validate_pair(original, conflicting)
pairs[instance_id] = (original, conflicting)
return pairs, conflicting_records
def require_resolved_targets(instance_id: str, result) -> None:
"""Reject an unusable validation cell before another cell is run."""
statuses = result.target_statuses
if not statuses or any(status in {"MISSING", "ERROR"} for status in statuses.values()):
raise ValidationError(
f"validation contains missing/error targets: {instance_id} "
f"{result.split}/{result.mode}"
)
def validate_existing_manifests(plan, out: Path, conflicting_records) -> None:
"""Fail on the first invalid resume artifact before Docker work continues."""
selected = plan["selection"]["instance_ids"]
for position, instance_id in enumerate(selected, 1):
manifest_path = out / instance_id.replace("/", "_") / "manifest.json"
if manifest_path.exists():
print(
f"[{position}/{len(selected)}] {instance_id}: validating existing manifest",
flush=True,
)
validate_task_manifest(
plan, instance_id, manifest_path, conflicting_records[instance_id]
)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--plan", type=Path, required=True)
@@ -52,30 +100,39 @@ def main() -> int:
out = args.out.resolve()
if declared != out / "index.json":
raise SystemExit("--out does not match the frozen validation index location")
pairs, conflicting_records = load_selected_pairs(plan)
if declared.is_file():
records = load_records(plan["dataset"]["revision"], "conflicting")
result = validate_environment_index_for_records(plan, ROOT, records)
result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
print(json.dumps({"status": "already validated", **result}, indent=2))
return 0
validate_existing_manifests(plan, out, conflicting_records)
docker_preflight(os.environ)
out.mkdir(parents=True, exist_ok=True)
entries = {}
pulled_images: set[str] = set()
for instance_id in plan["selection"]["instance_ids"]:
selected = plan["selection"]["instance_ids"]
for position, instance_id in enumerate(selected, 1):
prefix = f"[{position}/{len(selected)}] {instance_id}"
task_dir = out / instance_id.replace("/", "_")
manifest_path = task_dir / "manifest.json"
if not manifest_path.exists():
original, conflicting = load_pair(plan["dataset"]["revision"], instance_id)
original, conflicting = pairs[instance_id]
if manifest_path.exists():
print(f"{prefix}: reusing validated manifest", flush=True)
else:
print(f"{prefix}: pulling exact image", flush=True)
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
results = [
run_trial(record, split=split, mode=mode, out_dir=task_dir,
environ=os.environ,
memory=plan["parameters"]["memory"],
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"])
for split, record in (("original", original), ("conflicting", conflicting))
for mode in ("nochange", "oracle")
]
results = []
for split, record in (("original", original), ("conflicting", conflicting)):
for mode in ("nochange", "oracle"):
print(f"{prefix}: running {split}/{mode}", flush=True)
result = run_trial(
record, split=split, mode=mode, out_dir=task_dir,
environ=os.environ, memory=plan["parameters"]["memory"],
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"],
)
require_resolved_targets(instance_id, result)
results.append(result)
validate_expected_matrix(results)
value = trial_manifest(
plan["dataset"]["revision"], instance_id, original, conflicting, results
@@ -83,6 +140,7 @@ def main() -> int:
with manifest_path.open("x") as handle:
json.dump(value, handle, indent=2, sort_keys=True)
handle.write("\n")
print(f"{prefix}: validation passed", flush=True)
entries[instance_id] = {
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
}
@@ -92,8 +150,7 @@ def main() -> int:
with declared.open("x") as handle:
json.dump(index, handle, indent=2, sort_keys=True)
handle.write("\n")
records = load_records(plan["dataset"]["revision"], "conflicting")
result = validate_environment_index_for_records(plan, ROOT, records)
result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
print(json.dumps({"status": "validated", **result}, indent=2))
return 0
+73 -62
View File
@@ -44,73 +44,15 @@ def validate_environment_index_for_records(
or set(index.get("manifests", {})) != set(selected)):
raise ValueError("environment validation index does not match the frozen plan")
evidence = []
expected_cells = {
("original", "nochange"): False,
("original", "oracle"): True,
("conflicting", "nochange"): False,
("conflicting", "oracle"): False,
}
for instance_id in selected:
entry = index["manifests"][instance_id]
manifest_path = _local(root, entry.get("path"))
if _sha(manifest_path) != entry.get("sha256"):
raise ValueError(f"validation manifest hash mismatch: {instance_id}")
manifest = json.loads(manifest_path.read_text())
if (manifest.get("schema_version") != 1
or manifest.get("dataset") != DATASET
or manifest.get("dataset_revision") != plan["dataset"]["revision"]
or manifest.get("instance_id") != instance_id
or manifest.get("network") != "none"):
raise ValueError(f"validation manifest identity mismatch: {instance_id}")
if records is not None:
record = records.get(instance_id)
if record is None:
raise ValueError(f"frozen validation record missing: {instance_id}")
canonical = hashlib.sha256(json.dumps(
dict(record), sort_keys=True, separators=(",", ":")
).encode()).hexdigest()
expected_hashes = {
"base_commit": record.get("base_commit"),
"repo": record.get("repo"),
"version": record.get("version"),
"original_test_patch_sha256": hashlib.sha256(
str(record.get("original_test_patch", "")).encode()
).hexdigest(),
"conflicting_test_patch_sha256": hashlib.sha256(
str(record.get("test_patch", "")).encode()
).hexdigest(),
"oracle_patch_sha256": hashlib.sha256(
str(record.get("patch", "")).encode()
).hexdigest(),
}
if (canonical != plan.get("records_sha256", {}).get(instance_id)
or any(manifest.get(key) != value
for key, value in expected_hashes.items())):
raise ValueError(f"validation patches do not match frozen record: {instance_id}")
results = manifest.get("results")
if not isinstance(results, list):
raise ValueError(f"validation results missing: {instance_id}")
cells = {(row.get("split"), row.get("mode")): row for row in results}
if set(cells) != set(expected_cells) or len(results) != 4:
raise ValueError(f"validation matrix incomplete: {instance_id}")
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
for row in results}
if len(identities) != 1 or any(
cells[cell].get("resolved") is not expected
for cell, expected in expected_cells.items()
):
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
if any(
not row.get("target_statuses")
or any(status in {"MISSING", "ERROR"}
for status in row["target_statuses"].values())
for row in results
):
raise ValueError(f"validation contains missing/error targets: {instance_id}")
for row in results:
output = manifest_path.parent / str(row.get("output_file", ""))
if not output.is_file() or _sha(output) != row.get("output_sha256"):
raise ValueError(f"validation output hash mismatch: {instance_id}")
record = records.get(instance_id) if records is not None else None
if records is not None and record is None:
raise ValueError(f"frozen validation record missing: {instance_id}")
validate_task_manifest(plan, instance_id, manifest_path, record)
evidence.append({
"instance_id": instance_id,
"manifest_path": str(manifest_path),
@@ -118,3 +60,72 @@ def validate_environment_index_for_records(
})
return {"index_path": str(index_path), "index_sha256": _sha(index_path),
"validated_instances": evidence}
def validate_task_manifest(
plan: Mapping[str, Any],
instance_id: str,
manifest_path: Path,
record: Mapping[str, Any] | None = None,
) -> dict[str, Any]:
"""Validate one task manifest, including partial evidence during resume."""
manifest = json.loads(manifest_path.read_text())
if (manifest.get("schema_version") != 1
or manifest.get("dataset") != DATASET
or manifest.get("dataset_revision") != plan["dataset"]["revision"]
or manifest.get("instance_id") != instance_id
or manifest.get("network") != "none"):
raise ValueError(f"validation manifest identity mismatch: {instance_id}")
if record is not None:
canonical = hashlib.sha256(json.dumps(
dict(record), sort_keys=True, separators=(",", ":")
).encode()).hexdigest()
expected_hashes = {
"base_commit": record.get("base_commit"),
"repo": record.get("repo"),
"version": record.get("version"),
"original_test_patch_sha256": hashlib.sha256(
str(record.get("original_test_patch", "")).encode()
).hexdigest(),
"conflicting_test_patch_sha256": hashlib.sha256(
str(record.get("test_patch", "")).encode()
).hexdigest(),
"oracle_patch_sha256": hashlib.sha256(
str(record.get("patch", "")).encode()
).hexdigest(),
}
if (canonical != plan.get("records_sha256", {}).get(instance_id)
or any(manifest.get(key) != value
for key, value in expected_hashes.items())):
raise ValueError(f"validation patches do not match frozen record: {instance_id}")
results = manifest.get("results")
if not isinstance(results, list):
raise ValueError(f"validation results missing: {instance_id}")
expected_cells = {
("original", "nochange"): False,
("original", "oracle"): True,
("conflicting", "nochange"): False,
("conflicting", "oracle"): False,
}
cells = {(row.get("split"), row.get("mode")): row for row in results}
if set(cells) != set(expected_cells) or len(results) != 4:
raise ValueError(f"validation matrix incomplete: {instance_id}")
if any(
not row.get("target_statuses")
or any(status in {"MISSING", "ERROR"}
for status in row["target_statuses"].values())
for row in results
):
raise ValueError(f"validation contains missing/error targets: {instance_id}")
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
for row in results}
if len(identities) != 1 or any(
cells[cell].get("resolved") is not expected
for cell, expected in expected_cells.items()
):
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
for row in results:
output = manifest_path.parent / str(row.get("output_file", ""))
if not output.is_file() or _sha(output) != row.get("output_sha256"):
raise ValueError(f"validation output hash mismatch: {instance_id}")
return manifest
+52 -1
View File
@@ -5,11 +5,18 @@ import json
import pytest
from messageboardbench.swe_validation import ValidationError
from messageboardbench.swe_prerequisites import (
validate_environment_index,
validate_environment_index_for_records,
validate_task_manifest,
)
from scripts.validate_swe_population_prerequisites import (
load_selected_pairs,
pull_image_once,
require_resolved_targets,
validate_existing_manifests,
)
from scripts.validate_swe_population_prerequisites import pull_image_once
def write(path, value):
@@ -105,3 +112,47 @@ def test_image_is_pulled_once_before_any_inspection_or_trial():
["docker", "pull", "image:tag"],
["docker", "image", "inspect", "image:tag"],
]
def test_selected_pairs_load_each_dataset_split_exactly_once():
calls = []
common = {
"instance_id": "task", "repo": "org/repo", "version": "1",
"base_commit": "base", "patch": "oracle", "original_test_patch": "original",
"FAIL_TO_PASS": ["target"], "PASS_TO_PASS": [],
}
rows = {
"original": {"task": {**common, "test_patch": "original"}},
"conflicting": {"task": {**common, "test_patch": "conflict"}},
}
def loader(revision, split):
calls.append((revision, split))
return rows[split]
plan = {"dataset": {"revision": "1" * 40},
"selection": {"instance_ids": ["task"]}}
pairs, conflicting = load_selected_pairs(plan, loader=loader)
assert calls == [("1" * 40, "original"), ("1" * 40, "conflicting")]
assert pairs["task"] == (rows["original"]["task"], rows["conflicting"]["task"])
assert conflicting is rows["conflicting"]
def test_new_cell_rejects_missing_targets_immediately():
result = __import__("types").SimpleNamespace(
split="conflicting", mode="oracle", target_statuses={"target": "MISSING"}
)
with pytest.raises(ValidationError, match="missing/error targets.*conflicting/oracle"):
require_resolved_targets("task", result)
def test_resume_rejects_existing_missing_manifest_before_reuse(tmp_path):
plan, manifest_path, record = fixture(tmp_path)
manifest = json.loads(manifest_path.read_text())
manifest["results"][0]["target_statuses"] = {"target": "ERROR"}
task_dir = tmp_path / "task"
for row in manifest["results"]:
write(task_dir / row["output_file"], "test output")
write(task_dir / "manifest.json", manifest)
with pytest.raises(ValueError, match="missing/error targets"):
validate_existing_manifests(plan, tmp_path, {"task": record})