mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 11:07:07 +00:00
Fail fast on invalid SWE prerequisite tasks
This commit is contained in:
1 parent
b3c935fbfa
commit
aeaae4291f
6 files changed
+211
-83
No files matched your search
@@ -4,6 +4,12 @@ This frozen developmental bundle reuses v2's ten tasks and changes only the poli
|
||||
suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out
|
||||
instruction. It creates fresh identities and stores when executed.
|
||||
|
||||
The bundle is currently blocked. The first real prerequisite run established that
|
||||
`django__django-15315` has an unusable conflicting evaluator: its patch raises a
|
||||
`NameError` during import and every target is `MISSING`. No behavioral model call
|
||||
started. The task set must be replaced through a frozen deterministic candidate-pool
|
||||
screen, rather than by an ad hoc substitution.
|
||||
|
||||
Validate the bundle offline:
|
||||
|
||||
```sh
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"status": "ready",
|
||||
"status": "blocked",
|
||||
"experiment_id": "swe-population-pilot-10-v3",
|
||||
"purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.",
|
||||
"remote_docker_host": "ssh://[email protected]",
|
||||
"blockers": [],
|
||||
"blockers": [
|
||||
"The reused django__django-15315 conflicting evaluator crashes during import and records every target as MISSING. Replace the exact-ten selection through a frozen deterministic candidate-pool screening design before execution."
|
||||
],
|
||||
"outputs": {
|
||||
"run_dir": "logs/swe-population-pilot-10-v3/run",
|
||||
"report_dir": "logs/swe-population-pilot-10-v3/report",
|
||||
@@ -33,5 +35,5 @@
|
||||
"argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"]
|
||||
}
|
||||
],
|
||||
"manifest_sha256": "5e7201286c193c87d01f308a7d65615c6f99c2b10412091cd983ebae3dd33aef"
|
||||
"manifest_sha256": "e648799bad82182e46c15dccc6f14280d565e17fe7d19d2ce1bf4d7bacd7e28e"
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
root := "../.."
|
||||
|
||||
start:
|
||||
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 --validate-only
|
||||
cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation
|
||||
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3
|
||||
|
||||
|
||||
@@ -9,15 +9,18 @@ from pathlib import Path
|
||||
import subprocess
|
||||
|
||||
from messageboardbench.swe_board import load_records, plan_hash
|
||||
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
|
||||
from messageboardbench.swe_prerequisites import (
|
||||
validate_environment_index_for_records,
|
||||
validate_task_manifest,
|
||||
)
|
||||
from messageboardbench.swe_validation import (
|
||||
ValidationError,
|
||||
docker_preflight,
|
||||
load_pair,
|
||||
manifest as trial_manifest,
|
||||
run_trial,
|
||||
swebench_spec,
|
||||
validate_expected_matrix,
|
||||
validate_pair,
|
||||
)
|
||||
|
||||
|
||||
@@ -40,6 +43,51 @@ def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) -
|
||||
pulled.add(image)
|
||||
|
||||
|
||||
def load_selected_pairs(plan, loader=None):
|
||||
"""Load each pinned dataset split once, then validate all selected pairs."""
|
||||
loader = load_records if loader is None else loader
|
||||
revision = plan["dataset"]["revision"]
|
||||
original_records = loader(revision, "original")
|
||||
conflicting_records = loader(revision, "conflicting")
|
||||
pairs = {}
|
||||
for instance_id in plan["selection"]["instance_ids"]:
|
||||
try:
|
||||
original = original_records[instance_id]
|
||||
conflicting = conflicting_records[instance_id]
|
||||
except KeyError as exc:
|
||||
raise ValidationError(
|
||||
f"selected task is absent from a pinned dataset split: {instance_id}"
|
||||
) from exc
|
||||
validate_pair(original, conflicting)
|
||||
pairs[instance_id] = (original, conflicting)
|
||||
return pairs, conflicting_records
|
||||
|
||||
|
||||
def require_resolved_targets(instance_id: str, result) -> None:
|
||||
"""Reject an unusable validation cell before another cell is run."""
|
||||
statuses = result.target_statuses
|
||||
if not statuses or any(status in {"MISSING", "ERROR"} for status in statuses.values()):
|
||||
raise ValidationError(
|
||||
f"validation contains missing/error targets: {instance_id} "
|
||||
f"{result.split}/{result.mode}"
|
||||
)
|
||||
|
||||
|
||||
def validate_existing_manifests(plan, out: Path, conflicting_records) -> None:
|
||||
"""Fail on the first invalid resume artifact before Docker work continues."""
|
||||
selected = plan["selection"]["instance_ids"]
|
||||
for position, instance_id in enumerate(selected, 1):
|
||||
manifest_path = out / instance_id.replace("/", "_") / "manifest.json"
|
||||
if manifest_path.exists():
|
||||
print(
|
||||
f"[{position}/{len(selected)}] {instance_id}: validating existing manifest",
|
||||
flush=True,
|
||||
)
|
||||
validate_task_manifest(
|
||||
plan, instance_id, manifest_path, conflicting_records[instance_id]
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--plan", type=Path, required=True)
|
||||
@@ -52,30 +100,39 @@ def main() -> int:
|
||||
out = args.out.resolve()
|
||||
if declared != out / "index.json":
|
||||
raise SystemExit("--out does not match the frozen validation index location")
|
||||
pairs, conflicting_records = load_selected_pairs(plan)
|
||||
if declared.is_file():
|
||||
records = load_records(plan["dataset"]["revision"], "conflicting")
|
||||
result = validate_environment_index_for_records(plan, ROOT, records)
|
||||
result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
|
||||
print(json.dumps({"status": "already validated", **result}, indent=2))
|
||||
return 0
|
||||
|
||||
validate_existing_manifests(plan, out, conflicting_records)
|
||||
docker_preflight(os.environ)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
entries = {}
|
||||
pulled_images: set[str] = set()
|
||||
for instance_id in plan["selection"]["instance_ids"]:
|
||||
selected = plan["selection"]["instance_ids"]
|
||||
for position, instance_id in enumerate(selected, 1):
|
||||
prefix = f"[{position}/{len(selected)}] {instance_id}"
|
||||
task_dir = out / instance_id.replace("/", "_")
|
||||
manifest_path = task_dir / "manifest.json"
|
||||
if not manifest_path.exists():
|
||||
original, conflicting = load_pair(plan["dataset"]["revision"], instance_id)
|
||||
original, conflicting = pairs[instance_id]
|
||||
if manifest_path.exists():
|
||||
print(f"{prefix}: reusing validated manifest", flush=True)
|
||||
else:
|
||||
print(f"{prefix}: pulling exact image", flush=True)
|
||||
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
|
||||
results = [
|
||||
run_trial(record, split=split, mode=mode, out_dir=task_dir,
|
||||
environ=os.environ,
|
||||
memory=plan["parameters"]["memory"],
|
||||
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"])
|
||||
for split, record in (("original", original), ("conflicting", conflicting))
|
||||
for mode in ("nochange", "oracle")
|
||||
]
|
||||
results = []
|
||||
for split, record in (("original", original), ("conflicting", conflicting)):
|
||||
for mode in ("nochange", "oracle"):
|
||||
print(f"{prefix}: running {split}/{mode}", flush=True)
|
||||
result = run_trial(
|
||||
record, split=split, mode=mode, out_dir=task_dir,
|
||||
environ=os.environ, memory=plan["parameters"]["memory"],
|
||||
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"],
|
||||
)
|
||||
require_resolved_targets(instance_id, result)
|
||||
results.append(result)
|
||||
validate_expected_matrix(results)
|
||||
value = trial_manifest(
|
||||
plan["dataset"]["revision"], instance_id, original, conflicting, results
|
||||
@@ -83,6 +140,7 @@ def main() -> int:
|
||||
with manifest_path.open("x") as handle:
|
||||
json.dump(value, handle, indent=2, sort_keys=True)
|
||||
handle.write("\n")
|
||||
print(f"{prefix}: validation passed", flush=True)
|
||||
entries[instance_id] = {
|
||||
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
|
||||
}
|
||||
@@ -92,8 +150,7 @@ def main() -> int:
|
||||
with declared.open("x") as handle:
|
||||
json.dump(index, handle, indent=2, sort_keys=True)
|
||||
handle.write("\n")
|
||||
records = load_records(plan["dataset"]["revision"], "conflicting")
|
||||
result = validate_environment_index_for_records(plan, ROOT, records)
|
||||
result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
|
||||
print(json.dumps({"status": "validated", **result}, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
@@ -44,73 +44,15 @@ def validate_environment_index_for_records(
|
||||
or set(index.get("manifests", {})) != set(selected)):
|
||||
raise ValueError("environment validation index does not match the frozen plan")
|
||||
evidence = []
|
||||
expected_cells = {
|
||||
("original", "nochange"): False,
|
||||
("original", "oracle"): True,
|
||||
("conflicting", "nochange"): False,
|
||||
("conflicting", "oracle"): False,
|
||||
}
|
||||
for instance_id in selected:
|
||||
entry = index["manifests"][instance_id]
|
||||
manifest_path = _local(root, entry.get("path"))
|
||||
if _sha(manifest_path) != entry.get("sha256"):
|
||||
raise ValueError(f"validation manifest hash mismatch: {instance_id}")
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
if (manifest.get("schema_version") != 1
|
||||
or manifest.get("dataset") != DATASET
|
||||
or manifest.get("dataset_revision") != plan["dataset"]["revision"]
|
||||
or manifest.get("instance_id") != instance_id
|
||||
or manifest.get("network") != "none"):
|
||||
raise ValueError(f"validation manifest identity mismatch: {instance_id}")
|
||||
if records is not None:
|
||||
record = records.get(instance_id)
|
||||
if record is None:
|
||||
raise ValueError(f"frozen validation record missing: {instance_id}")
|
||||
canonical = hashlib.sha256(json.dumps(
|
||||
dict(record), sort_keys=True, separators=(",", ":")
|
||||
).encode()).hexdigest()
|
||||
expected_hashes = {
|
||||
"base_commit": record.get("base_commit"),
|
||||
"repo": record.get("repo"),
|
||||
"version": record.get("version"),
|
||||
"original_test_patch_sha256": hashlib.sha256(
|
||||
str(record.get("original_test_patch", "")).encode()
|
||||
).hexdigest(),
|
||||
"conflicting_test_patch_sha256": hashlib.sha256(
|
||||
str(record.get("test_patch", "")).encode()
|
||||
).hexdigest(),
|
||||
"oracle_patch_sha256": hashlib.sha256(
|
||||
str(record.get("patch", "")).encode()
|
||||
).hexdigest(),
|
||||
}
|
||||
if (canonical != plan.get("records_sha256", {}).get(instance_id)
|
||||
or any(manifest.get(key) != value
|
||||
for key, value in expected_hashes.items())):
|
||||
raise ValueError(f"validation patches do not match frozen record: {instance_id}")
|
||||
results = manifest.get("results")
|
||||
if not isinstance(results, list):
|
||||
raise ValueError(f"validation results missing: {instance_id}")
|
||||
cells = {(row.get("split"), row.get("mode")): row for row in results}
|
||||
if set(cells) != set(expected_cells) or len(results) != 4:
|
||||
raise ValueError(f"validation matrix incomplete: {instance_id}")
|
||||
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
|
||||
for row in results}
|
||||
if len(identities) != 1 or any(
|
||||
cells[cell].get("resolved") is not expected
|
||||
for cell, expected in expected_cells.items()
|
||||
):
|
||||
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
|
||||
if any(
|
||||
not row.get("target_statuses")
|
||||
or any(status in {"MISSING", "ERROR"}
|
||||
for status in row["target_statuses"].values())
|
||||
for row in results
|
||||
):
|
||||
raise ValueError(f"validation contains missing/error targets: {instance_id}")
|
||||
for row in results:
|
||||
output = manifest_path.parent / str(row.get("output_file", ""))
|
||||
if not output.is_file() or _sha(output) != row.get("output_sha256"):
|
||||
raise ValueError(f"validation output hash mismatch: {instance_id}")
|
||||
record = records.get(instance_id) if records is not None else None
|
||||
if records is not None and record is None:
|
||||
raise ValueError(f"frozen validation record missing: {instance_id}")
|
||||
validate_task_manifest(plan, instance_id, manifest_path, record)
|
||||
evidence.append({
|
||||
"instance_id": instance_id,
|
||||
"manifest_path": str(manifest_path),
|
||||
@@ -118,3 +60,72 @@ def validate_environment_index_for_records(
|
||||
})
|
||||
return {"index_path": str(index_path), "index_sha256": _sha(index_path),
|
||||
"validated_instances": evidence}
|
||||
|
||||
|
||||
def validate_task_manifest(
|
||||
plan: Mapping[str, Any],
|
||||
instance_id: str,
|
||||
manifest_path: Path,
|
||||
record: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Validate one task manifest, including partial evidence during resume."""
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
if (manifest.get("schema_version") != 1
|
||||
or manifest.get("dataset") != DATASET
|
||||
or manifest.get("dataset_revision") != plan["dataset"]["revision"]
|
||||
or manifest.get("instance_id") != instance_id
|
||||
or manifest.get("network") != "none"):
|
||||
raise ValueError(f"validation manifest identity mismatch: {instance_id}")
|
||||
if record is not None:
|
||||
canonical = hashlib.sha256(json.dumps(
|
||||
dict(record), sort_keys=True, separators=(",", ":")
|
||||
).encode()).hexdigest()
|
||||
expected_hashes = {
|
||||
"base_commit": record.get("base_commit"),
|
||||
"repo": record.get("repo"),
|
||||
"version": record.get("version"),
|
||||
"original_test_patch_sha256": hashlib.sha256(
|
||||
str(record.get("original_test_patch", "")).encode()
|
||||
).hexdigest(),
|
||||
"conflicting_test_patch_sha256": hashlib.sha256(
|
||||
str(record.get("test_patch", "")).encode()
|
||||
).hexdigest(),
|
||||
"oracle_patch_sha256": hashlib.sha256(
|
||||
str(record.get("patch", "")).encode()
|
||||
).hexdigest(),
|
||||
}
|
||||
if (canonical != plan.get("records_sha256", {}).get(instance_id)
|
||||
or any(manifest.get(key) != value
|
||||
for key, value in expected_hashes.items())):
|
||||
raise ValueError(f"validation patches do not match frozen record: {instance_id}")
|
||||
results = manifest.get("results")
|
||||
if not isinstance(results, list):
|
||||
raise ValueError(f"validation results missing: {instance_id}")
|
||||
expected_cells = {
|
||||
("original", "nochange"): False,
|
||||
("original", "oracle"): True,
|
||||
("conflicting", "nochange"): False,
|
||||
("conflicting", "oracle"): False,
|
||||
}
|
||||
cells = {(row.get("split"), row.get("mode")): row for row in results}
|
||||
if set(cells) != set(expected_cells) or len(results) != 4:
|
||||
raise ValueError(f"validation matrix incomplete: {instance_id}")
|
||||
if any(
|
||||
not row.get("target_statuses")
|
||||
or any(status in {"MISSING", "ERROR"}
|
||||
for status in row["target_statuses"].values())
|
||||
for row in results
|
||||
):
|
||||
raise ValueError(f"validation contains missing/error targets: {instance_id}")
|
||||
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
|
||||
for row in results}
|
||||
if len(identities) != 1 or any(
|
||||
cells[cell].get("resolved") is not expected
|
||||
for cell, expected in expected_cells.items()
|
||||
):
|
||||
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
|
||||
for row in results:
|
||||
output = manifest_path.parent / str(row.get("output_file", ""))
|
||||
if not output.is_file() or _sha(output) != row.get("output_sha256"):
|
||||
raise ValueError(f"validation output hash mismatch: {instance_id}")
|
||||
return manifest
|
||||
@@ -5,11 +5,18 @@ import json
|
||||
|
||||
import pytest
|
||||
|
||||
from messageboardbench.swe_validation import ValidationError
|
||||
from messageboardbench.swe_prerequisites import (
|
||||
validate_environment_index,
|
||||
validate_environment_index_for_records,
|
||||
validate_task_manifest,
|
||||
)
|
||||
from scripts.validate_swe_population_prerequisites import (
|
||||
load_selected_pairs,
|
||||
pull_image_once,
|
||||
require_resolved_targets,
|
||||
validate_existing_manifests,
|
||||
)
|
||||
from scripts.validate_swe_population_prerequisites import pull_image_once
|
||||
|
||||
|
||||
def write(path, value):
|
||||
@@ -105,3 +112,47 @@ def test_image_is_pulled_once_before_any_inspection_or_trial():
|
||||
["docker", "pull", "image:tag"],
|
||||
["docker", "image", "inspect", "image:tag"],
|
||||
]
|
||||
|
||||
|
||||
def test_selected_pairs_load_each_dataset_split_exactly_once():
|
||||
calls = []
|
||||
common = {
|
||||
"instance_id": "task", "repo": "org/repo", "version": "1",
|
||||
"base_commit": "base", "patch": "oracle", "original_test_patch": "original",
|
||||
"FAIL_TO_PASS": ["target"], "PASS_TO_PASS": [],
|
||||
}
|
||||
rows = {
|
||||
"original": {"task": {**common, "test_patch": "original"}},
|
||||
"conflicting": {"task": {**common, "test_patch": "conflict"}},
|
||||
}
|
||||
|
||||
def loader(revision, split):
|
||||
calls.append((revision, split))
|
||||
return rows[split]
|
||||
|
||||
plan = {"dataset": {"revision": "1" * 40},
|
||||
"selection": {"instance_ids": ["task"]}}
|
||||
pairs, conflicting = load_selected_pairs(plan, loader=loader)
|
||||
assert calls == [("1" * 40, "original"), ("1" * 40, "conflicting")]
|
||||
assert pairs["task"] == (rows["original"]["task"], rows["conflicting"]["task"])
|
||||
assert conflicting is rows["conflicting"]
|
||||
|
||||
|
||||
def test_new_cell_rejects_missing_targets_immediately():
|
||||
result = __import__("types").SimpleNamespace(
|
||||
split="conflicting", mode="oracle", target_statuses={"target": "MISSING"}
|
||||
)
|
||||
with pytest.raises(ValidationError, match="missing/error targets.*conflicting/oracle"):
|
||||
require_resolved_targets("task", result)
|
||||
|
||||
|
||||
def test_resume_rejects_existing_missing_manifest_before_reuse(tmp_path):
|
||||
plan, manifest_path, record = fixture(tmp_path)
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
manifest["results"][0]["target_statuses"] = {"target": "ERROR"}
|
||||
task_dir = tmp_path / "task"
|
||||
for row in manifest["results"]:
|
||||
write(task_dir / row["output_file"], "test output")
|
||||
write(task_dir / "manifest.json", manifest)
|
||||
with pytest.raises(ValueError, match="missing/error targets"):
|
||||
validate_existing_manifests(plan, tmp_path, {"task": record})
|
||||
Reference in new issue
Block a user