diff --git a/experiments/swe-population-pilot-10-v3/README.md b/experiments/swe-population-pilot-10-v3/README.md index 4375b80..bfb9da0 100644 --- a/experiments/swe-population-pilot-10-v3/README.md +++ b/experiments/swe-population-pilot-10-v3/README.md @@ -4,6 +4,12 @@ This frozen developmental bundle reuses v2's ten tasks and changes only the poli suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out instruction. It creates fresh identities and stores when executed. +The bundle is currently blocked. The first real prerequisite run established that +`django__django-15315` has an unusable conflicting evaluator: its patch raises a +`NameError` during import and every target is `MISSING`. No behavioral model call +started. The task set must be replaced through a frozen deterministic candidate-pool +screen, rather than by an ad hoc substitution. + Validate the bundle offline: ```sh diff --git a/experiments/swe-population-pilot-10-v3/experiment.json b/experiments/swe-population-pilot-10-v3/experiment.json index 80930d8..0d36ddb 100644 --- a/experiments/swe-population-pilot-10-v3/experiment.json +++ b/experiments/swe-population-pilot-10-v3/experiment.json @@ -1,10 +1,12 @@ { "schema_version": 1, - "status": "ready", + "status": "blocked", "experiment_id": "swe-population-pilot-10-v3", "purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.", "remote_docker_host": "ssh://pj@100.68.126.75", - "blockers": [], + "blockers": [ + "The reused django__django-15315 conflicting evaluator crashes during import and records every target as MISSING. Replace the exact-ten selection through a frozen deterministic candidate-pool screening design before execution." + ], "outputs": { "run_dir": "logs/swe-population-pilot-10-v3/run", "report_dir": "logs/swe-population-pilot-10-v3/report", @@ -33,5 +35,5 @@ "argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"] } ], - "manifest_sha256": "5e7201286c193c87d01f308a7d65615c6f99c2b10412091cd983ebae3dd33aef" + "manifest_sha256": "e648799bad82182e46c15dccc6f14280d565e17fe7d19d2ce1bf4d7bacd7e28e" } diff --git a/experiments/swe-population-pilot-10-v3/justfile b/experiments/swe-population-pilot-10-v3/justfile index aa097eb..4251483 100644 --- a/experiments/swe-population-pilot-10-v3/justfile +++ b/experiments/swe-population-pilot-10-v3/justfile @@ -1,6 +1,7 @@ root := "../.." start: + cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 --validate-only cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 diff --git a/scripts/validate_swe_population_prerequisites.py b/scripts/validate_swe_population_prerequisites.py index cc43f6a..876f740 100644 --- a/scripts/validate_swe_population_prerequisites.py +++ b/scripts/validate_swe_population_prerequisites.py @@ -9,15 +9,18 @@ from pathlib import Path import subprocess from messageboardbench.swe_board import load_records, plan_hash -from messageboardbench.swe_prerequisites import validate_environment_index_for_records +from messageboardbench.swe_prerequisites import ( + validate_environment_index_for_records, + validate_task_manifest, +) from messageboardbench.swe_validation import ( ValidationError, docker_preflight, - load_pair, manifest as trial_manifest, run_trial, swebench_spec, validate_expected_matrix, + validate_pair, ) @@ -40,6 +43,51 @@ def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) - pulled.add(image) +def load_selected_pairs(plan, loader=None): + """Load each pinned dataset split once, then validate all selected pairs.""" + loader = load_records if loader is None else loader + revision = plan["dataset"]["revision"] + original_records = loader(revision, "original") + conflicting_records = loader(revision, "conflicting") + pairs = {} + for instance_id in plan["selection"]["instance_ids"]: + try: + original = original_records[instance_id] + conflicting = conflicting_records[instance_id] + except KeyError as exc: + raise ValidationError( + f"selected task is absent from a pinned dataset split: {instance_id}" + ) from exc + validate_pair(original, conflicting) + pairs[instance_id] = (original, conflicting) + return pairs, conflicting_records + + +def require_resolved_targets(instance_id: str, result) -> None: + """Reject an unusable validation cell before another cell is run.""" + statuses = result.target_statuses + if not statuses or any(status in {"MISSING", "ERROR"} for status in statuses.values()): + raise ValidationError( + f"validation contains missing/error targets: {instance_id} " + f"{result.split}/{result.mode}" + ) + + +def validate_existing_manifests(plan, out: Path, conflicting_records) -> None: + """Fail on the first invalid resume artifact before Docker work continues.""" + selected = plan["selection"]["instance_ids"] + for position, instance_id in enumerate(selected, 1): + manifest_path = out / instance_id.replace("/", "_") / "manifest.json" + if manifest_path.exists(): + print( + f"[{position}/{len(selected)}] {instance_id}: validating existing manifest", + flush=True, + ) + validate_task_manifest( + plan, instance_id, manifest_path, conflicting_records[instance_id] + ) + + def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--plan", type=Path, required=True) @@ -52,30 +100,39 @@ def main() -> int: out = args.out.resolve() if declared != out / "index.json": raise SystemExit("--out does not match the frozen validation index location") + pairs, conflicting_records = load_selected_pairs(plan) if declared.is_file(): - records = load_records(plan["dataset"]["revision"], "conflicting") - result = validate_environment_index_for_records(plan, ROOT, records) + result = validate_environment_index_for_records(plan, ROOT, conflicting_records) print(json.dumps({"status": "already validated", **result}, indent=2)) return 0 + validate_existing_manifests(plan, out, conflicting_records) docker_preflight(os.environ) out.mkdir(parents=True, exist_ok=True) entries = {} pulled_images: set[str] = set() - for instance_id in plan["selection"]["instance_ids"]: + selected = plan["selection"]["instance_ids"] + for position, instance_id in enumerate(selected, 1): + prefix = f"[{position}/{len(selected)}] {instance_id}" task_dir = out / instance_id.replace("/", "_") manifest_path = task_dir / "manifest.json" - if not manifest_path.exists(): - original, conflicting = load_pair(plan["dataset"]["revision"], instance_id) + original, conflicting = pairs[instance_id] + if manifest_path.exists(): + print(f"{prefix}: reusing validated manifest", flush=True) + else: + print(f"{prefix}: pulling exact image", flush=True) pull_image_once(swebench_spec(original)[0], pulled_images, os.environ) - results = [ - run_trial(record, split=split, mode=mode, out_dir=task_dir, - environ=os.environ, - memory=plan["parameters"]["memory"], - timeout_seconds=plan["parameters"]["scorer_timeout_seconds"]) - for split, record in (("original", original), ("conflicting", conflicting)) - for mode in ("nochange", "oracle") - ] + results = [] + for split, record in (("original", original), ("conflicting", conflicting)): + for mode in ("nochange", "oracle"): + print(f"{prefix}: running {split}/{mode}", flush=True) + result = run_trial( + record, split=split, mode=mode, out_dir=task_dir, + environ=os.environ, memory=plan["parameters"]["memory"], + timeout_seconds=plan["parameters"]["scorer_timeout_seconds"], + ) + require_resolved_targets(instance_id, result) + results.append(result) validate_expected_matrix(results) value = trial_manifest( plan["dataset"]["revision"], instance_id, original, conflicting, results @@ -83,6 +140,7 @@ def main() -> int: with manifest_path.open("x") as handle: json.dump(value, handle, indent=2, sort_keys=True) handle.write("\n") + print(f"{prefix}: validation passed", flush=True) entries[instance_id] = { "path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path) } @@ -92,8 +150,7 @@ def main() -> int: with declared.open("x") as handle: json.dump(index, handle, indent=2, sort_keys=True) handle.write("\n") - records = load_records(plan["dataset"]["revision"], "conflicting") - result = validate_environment_index_for_records(plan, ROOT, records) + result = validate_environment_index_for_records(plan, ROOT, conflicting_records) print(json.dumps({"status": "validated", **result}, indent=2)) return 0 diff --git a/src/messageboardbench/swe_prerequisites.py b/src/messageboardbench/swe_prerequisites.py index 0ca4aa7..e90eafa 100644 --- a/src/messageboardbench/swe_prerequisites.py +++ b/src/messageboardbench/swe_prerequisites.py @@ -44,73 +44,15 @@ def validate_environment_index_for_records( or set(index.get("manifests", {})) != set(selected)): raise ValueError("environment validation index does not match the frozen plan") evidence = [] - expected_cells = { - ("original", "nochange"): False, - ("original", "oracle"): True, - ("conflicting", "nochange"): False, - ("conflicting", "oracle"): False, - } for instance_id in selected: entry = index["manifests"][instance_id] manifest_path = _local(root, entry.get("path")) if _sha(manifest_path) != entry.get("sha256"): raise ValueError(f"validation manifest hash mismatch: {instance_id}") - manifest = json.loads(manifest_path.read_text()) - if (manifest.get("schema_version") != 1 - or manifest.get("dataset") != DATASET - or manifest.get("dataset_revision") != plan["dataset"]["revision"] - or manifest.get("instance_id") != instance_id - or manifest.get("network") != "none"): - raise ValueError(f"validation manifest identity mismatch: {instance_id}") - if records is not None: - record = records.get(instance_id) - if record is None: - raise ValueError(f"frozen validation record missing: {instance_id}") - canonical = hashlib.sha256(json.dumps( - dict(record), sort_keys=True, separators=(",", ":") - ).encode()).hexdigest() - expected_hashes = { - "base_commit": record.get("base_commit"), - "repo": record.get("repo"), - "version": record.get("version"), - "original_test_patch_sha256": hashlib.sha256( - str(record.get("original_test_patch", "")).encode() - ).hexdigest(), - "conflicting_test_patch_sha256": hashlib.sha256( - str(record.get("test_patch", "")).encode() - ).hexdigest(), - "oracle_patch_sha256": hashlib.sha256( - str(record.get("patch", "")).encode() - ).hexdigest(), - } - if (canonical != plan.get("records_sha256", {}).get(instance_id) - or any(manifest.get(key) != value - for key, value in expected_hashes.items())): - raise ValueError(f"validation patches do not match frozen record: {instance_id}") - results = manifest.get("results") - if not isinstance(results, list): - raise ValueError(f"validation results missing: {instance_id}") - cells = {(row.get("split"), row.get("mode")): row for row in results} - if set(cells) != set(expected_cells) or len(results) != 4: - raise ValueError(f"validation matrix incomplete: {instance_id}") - identities = {(row.get("image_id"), tuple(row.get("repo_digests") or [])) - for row in results} - if len(identities) != 1 or any( - cells[cell].get("resolved") is not expected - for cell, expected in expected_cells.items() - ): - raise ValueError(f"validation matrix outcome mismatch: {instance_id}") - if any( - not row.get("target_statuses") - or any(status in {"MISSING", "ERROR"} - for status in row["target_statuses"].values()) - for row in results - ): - raise ValueError(f"validation contains missing/error targets: {instance_id}") - for row in results: - output = manifest_path.parent / str(row.get("output_file", "")) - if not output.is_file() or _sha(output) != row.get("output_sha256"): - raise ValueError(f"validation output hash mismatch: {instance_id}") + record = records.get(instance_id) if records is not None else None + if records is not None and record is None: + raise ValueError(f"frozen validation record missing: {instance_id}") + validate_task_manifest(plan, instance_id, manifest_path, record) evidence.append({ "instance_id": instance_id, "manifest_path": str(manifest_path), @@ -118,3 +60,72 @@ def validate_environment_index_for_records( }) return {"index_path": str(index_path), "index_sha256": _sha(index_path), "validated_instances": evidence} + + +def validate_task_manifest( + plan: Mapping[str, Any], + instance_id: str, + manifest_path: Path, + record: Mapping[str, Any] | None = None, +) -> dict[str, Any]: + """Validate one task manifest, including partial evidence during resume.""" + manifest = json.loads(manifest_path.read_text()) + if (manifest.get("schema_version") != 1 + or manifest.get("dataset") != DATASET + or manifest.get("dataset_revision") != plan["dataset"]["revision"] + or manifest.get("instance_id") != instance_id + or manifest.get("network") != "none"): + raise ValueError(f"validation manifest identity mismatch: {instance_id}") + if record is not None: + canonical = hashlib.sha256(json.dumps( + dict(record), sort_keys=True, separators=(",", ":") + ).encode()).hexdigest() + expected_hashes = { + "base_commit": record.get("base_commit"), + "repo": record.get("repo"), + "version": record.get("version"), + "original_test_patch_sha256": hashlib.sha256( + str(record.get("original_test_patch", "")).encode() + ).hexdigest(), + "conflicting_test_patch_sha256": hashlib.sha256( + str(record.get("test_patch", "")).encode() + ).hexdigest(), + "oracle_patch_sha256": hashlib.sha256( + str(record.get("patch", "")).encode() + ).hexdigest(), + } + if (canonical != plan.get("records_sha256", {}).get(instance_id) + or any(manifest.get(key) != value + for key, value in expected_hashes.items())): + raise ValueError(f"validation patches do not match frozen record: {instance_id}") + results = manifest.get("results") + if not isinstance(results, list): + raise ValueError(f"validation results missing: {instance_id}") + expected_cells = { + ("original", "nochange"): False, + ("original", "oracle"): True, + ("conflicting", "nochange"): False, + ("conflicting", "oracle"): False, + } + cells = {(row.get("split"), row.get("mode")): row for row in results} + if set(cells) != set(expected_cells) or len(results) != 4: + raise ValueError(f"validation matrix incomplete: {instance_id}") + if any( + not row.get("target_statuses") + or any(status in {"MISSING", "ERROR"} + for status in row["target_statuses"].values()) + for row in results + ): + raise ValueError(f"validation contains missing/error targets: {instance_id}") + identities = {(row.get("image_id"), tuple(row.get("repo_digests") or [])) + for row in results} + if len(identities) != 1 or any( + cells[cell].get("resolved") is not expected + for cell, expected in expected_cells.items() + ): + raise ValueError(f"validation matrix outcome mismatch: {instance_id}") + for row in results: + output = manifest_path.parent / str(row.get("output_file", "")) + if not output.is_file() or _sha(output) != row.get("output_sha256"): + raise ValueError(f"validation output hash mismatch: {instance_id}") + return manifest diff --git a/tests/test_swe_prerequisites.py b/tests/test_swe_prerequisites.py index 2688a16..1f75f87 100644 --- a/tests/test_swe_prerequisites.py +++ b/tests/test_swe_prerequisites.py @@ -5,11 +5,18 @@ import json import pytest +from messageboardbench.swe_validation import ValidationError from messageboardbench.swe_prerequisites import ( validate_environment_index, validate_environment_index_for_records, + validate_task_manifest, +) +from scripts.validate_swe_population_prerequisites import ( + load_selected_pairs, + pull_image_once, + require_resolved_targets, + validate_existing_manifests, ) -from scripts.validate_swe_population_prerequisites import pull_image_once def write(path, value): @@ -105,3 +112,47 @@ def test_image_is_pulled_once_before_any_inspection_or_trial(): ["docker", "pull", "image:tag"], ["docker", "image", "inspect", "image:tag"], ] + + +def test_selected_pairs_load_each_dataset_split_exactly_once(): + calls = [] + common = { + "instance_id": "task", "repo": "org/repo", "version": "1", + "base_commit": "base", "patch": "oracle", "original_test_patch": "original", + "FAIL_TO_PASS": ["target"], "PASS_TO_PASS": [], + } + rows = { + "original": {"task": {**common, "test_patch": "original"}}, + "conflicting": {"task": {**common, "test_patch": "conflict"}}, + } + + def loader(revision, split): + calls.append((revision, split)) + return rows[split] + + plan = {"dataset": {"revision": "1" * 40}, + "selection": {"instance_ids": ["task"]}} + pairs, conflicting = load_selected_pairs(plan, loader=loader) + assert calls == [("1" * 40, "original"), ("1" * 40, "conflicting")] + assert pairs["task"] == (rows["original"]["task"], rows["conflicting"]["task"]) + assert conflicting is rows["conflicting"] + + +def test_new_cell_rejects_missing_targets_immediately(): + result = __import__("types").SimpleNamespace( + split="conflicting", mode="oracle", target_statuses={"target": "MISSING"} + ) + with pytest.raises(ValidationError, match="missing/error targets.*conflicting/oracle"): + require_resolved_targets("task", result) + + +def test_resume_rejects_existing_missing_manifest_before_reuse(tmp_path): + plan, manifest_path, record = fixture(tmp_path) + manifest = json.loads(manifest_path.read_text()) + manifest["results"][0]["target_statuses"] = {"target": "ERROR"} + task_dir = tmp_path / "task" + for row in manifest["results"]: + write(task_dir / row["output_file"], "test output") + write(task_dir / "manifest.json", manifest) + with pytest.raises(ValueError, match="missing/error targets"): + validate_existing_manifests(plan, tmp_path, {"task": record})