mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 11:07:07 +00:00
Fail fast on invalid SWE prerequisite tasks
This commit is contained in:
1 parent
b3c935fbfa
commit
aeaae4291f
6 files changed
+211
-83
No files matched your search
@@ -4,6 +4,12 @@ This frozen developmental bundle reuses v2's ten tasks and changes only the poli
|
|||||||
suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out
|
suffix: it keeps `**DO NOT MODIFY THE TESTS.**` and removes the extra stop/carve-out
|
||||||
instruction. It creates fresh identities and stores when executed.
|
instruction. It creates fresh identities and stores when executed.
|
||||||
|
|
||||||
|
The bundle is currently blocked. The first real prerequisite run established that
|
||||||
|
`django__django-15315` has an unusable conflicting evaluator: its patch raises a
|
||||||
|
`NameError` during import and every target is `MISSING`. No behavioral model call
|
||||||
|
started. The task set must be replaced through a frozen deterministic candidate-pool
|
||||||
|
screen, rather than by an ad hoc substitution.
|
||||||
|
|
||||||
Validate the bundle offline:
|
Validate the bundle offline:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
|
|||||||
@@ -1,10 +1,12 @@
|
|||||||
{
|
{
|
||||||
"schema_version": 1,
|
"schema_version": 1,
|
||||||
"status": "ready",
|
"status": "blocked",
|
||||||
"experiment_id": "swe-population-pilot-10-v3",
|
"experiment_id": "swe-population-pilot-10-v3",
|
||||||
"purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.",
|
"purpose": "Developmental matched prompt comparison with independent SWE agents and a board-only cross-agent pathway.",
|
||||||
"remote_docker_host": "ssh://[email protected]",
|
"remote_docker_host": "ssh://[email protected]",
|
||||||
"blockers": [],
|
"blockers": [
|
||||||
|
"The reused django__django-15315 conflicting evaluator crashes during import and records every target as MISSING. Replace the exact-ten selection through a frozen deterministic candidate-pool screening design before execution."
|
||||||
|
],
|
||||||
"outputs": {
|
"outputs": {
|
||||||
"run_dir": "logs/swe-population-pilot-10-v3/run",
|
"run_dir": "logs/swe-population-pilot-10-v3/run",
|
||||||
"report_dir": "logs/swe-population-pilot-10-v3/report",
|
"report_dir": "logs/swe-population-pilot-10-v3/report",
|
||||||
@@ -33,5 +35,5 @@
|
|||||||
"argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"]
|
"argv": [".venv/bin/python", "scripts/analysis/board_resources.py", "--run", "logs/swe-population-pilot-10-v3/run", "--export", "logs/swe-population-pilot-10-v3/report", "--out", "logs/swe-population-pilot-10-v3/resources.json"]
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"manifest_sha256": "5e7201286c193c87d01f308a7d65615c6f99c2b10412091cd983ebae3dd33aef"
|
"manifest_sha256": "e648799bad82182e46c15dccc6f14280d565e17fe7d19d2ce1bf4d7bacd7e28e"
|
||||||
}
|
}
|
||||||
@@ -1,6 +1,7 @@
|
|||||||
root := "../.."
|
root := "../.."
|
||||||
|
|
||||||
start:
|
start:
|
||||||
|
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3 --validate-only
|
||||||
cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation
|
cd {{root}} && .venv/bin/python scripts/remote_docker.py -- .venv/bin/python scripts/validate_swe_population_prerequisites.py --plan experiments/swe-population-pilot-10-v3/plan.json --out work/swe-population-pilot-10-v3-validation
|
||||||
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3
|
cd {{root}} && .venv/bin/python scripts/run_experiment_bundle.py --bundle experiments/swe-population-pilot-10-v3
|
||||||
|
|
||||||
|
|||||||
@@ -9,15 +9,18 @@ from pathlib import Path
|
|||||||
import subprocess
|
import subprocess
|
||||||
|
|
||||||
from messageboardbench.swe_board import load_records, plan_hash
|
from messageboardbench.swe_board import load_records, plan_hash
|
||||||
from messageboardbench.swe_prerequisites import validate_environment_index_for_records
|
from messageboardbench.swe_prerequisites import (
|
||||||
|
validate_environment_index_for_records,
|
||||||
|
validate_task_manifest,
|
||||||
|
)
|
||||||
from messageboardbench.swe_validation import (
|
from messageboardbench.swe_validation import (
|
||||||
ValidationError,
|
ValidationError,
|
||||||
docker_preflight,
|
docker_preflight,
|
||||||
load_pair,
|
|
||||||
manifest as trial_manifest,
|
manifest as trial_manifest,
|
||||||
run_trial,
|
run_trial,
|
||||||
swebench_spec,
|
swebench_spec,
|
||||||
validate_expected_matrix,
|
validate_expected_matrix,
|
||||||
|
validate_pair,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -40,6 +43,51 @@ def pull_image_once(image: str, pulled: set[str], environ, run=subprocess.run) -
|
|||||||
pulled.add(image)
|
pulled.add(image)
|
||||||
|
|
||||||
|
|
||||||
|
def load_selected_pairs(plan, loader=None):
|
||||||
|
"""Load each pinned dataset split once, then validate all selected pairs."""
|
||||||
|
loader = load_records if loader is None else loader
|
||||||
|
revision = plan["dataset"]["revision"]
|
||||||
|
original_records = loader(revision, "original")
|
||||||
|
conflicting_records = loader(revision, "conflicting")
|
||||||
|
pairs = {}
|
||||||
|
for instance_id in plan["selection"]["instance_ids"]:
|
||||||
|
try:
|
||||||
|
original = original_records[instance_id]
|
||||||
|
conflicting = conflicting_records[instance_id]
|
||||||
|
except KeyError as exc:
|
||||||
|
raise ValidationError(
|
||||||
|
f"selected task is absent from a pinned dataset split: {instance_id}"
|
||||||
|
) from exc
|
||||||
|
validate_pair(original, conflicting)
|
||||||
|
pairs[instance_id] = (original, conflicting)
|
||||||
|
return pairs, conflicting_records
|
||||||
|
|
||||||
|
|
||||||
|
def require_resolved_targets(instance_id: str, result) -> None:
|
||||||
|
"""Reject an unusable validation cell before another cell is run."""
|
||||||
|
statuses = result.target_statuses
|
||||||
|
if not statuses or any(status in {"MISSING", "ERROR"} for status in statuses.values()):
|
||||||
|
raise ValidationError(
|
||||||
|
f"validation contains missing/error targets: {instance_id} "
|
||||||
|
f"{result.split}/{result.mode}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def validate_existing_manifests(plan, out: Path, conflicting_records) -> None:
|
||||||
|
"""Fail on the first invalid resume artifact before Docker work continues."""
|
||||||
|
selected = plan["selection"]["instance_ids"]
|
||||||
|
for position, instance_id in enumerate(selected, 1):
|
||||||
|
manifest_path = out / instance_id.replace("/", "_") / "manifest.json"
|
||||||
|
if manifest_path.exists():
|
||||||
|
print(
|
||||||
|
f"[{position}/{len(selected)}] {instance_id}: validating existing manifest",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
validate_task_manifest(
|
||||||
|
plan, instance_id, manifest_path, conflicting_records[instance_id]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
def main() -> int:
|
||||||
parser = argparse.ArgumentParser(description=__doc__)
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
parser.add_argument("--plan", type=Path, required=True)
|
parser.add_argument("--plan", type=Path, required=True)
|
||||||
@@ -52,30 +100,39 @@ def main() -> int:
|
|||||||
out = args.out.resolve()
|
out = args.out.resolve()
|
||||||
if declared != out / "index.json":
|
if declared != out / "index.json":
|
||||||
raise SystemExit("--out does not match the frozen validation index location")
|
raise SystemExit("--out does not match the frozen validation index location")
|
||||||
|
pairs, conflicting_records = load_selected_pairs(plan)
|
||||||
if declared.is_file():
|
if declared.is_file():
|
||||||
records = load_records(plan["dataset"]["revision"], "conflicting")
|
result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
|
||||||
result = validate_environment_index_for_records(plan, ROOT, records)
|
|
||||||
print(json.dumps({"status": "already validated", **result}, indent=2))
|
print(json.dumps({"status": "already validated", **result}, indent=2))
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
validate_existing_manifests(plan, out, conflicting_records)
|
||||||
docker_preflight(os.environ)
|
docker_preflight(os.environ)
|
||||||
out.mkdir(parents=True, exist_ok=True)
|
out.mkdir(parents=True, exist_ok=True)
|
||||||
entries = {}
|
entries = {}
|
||||||
pulled_images: set[str] = set()
|
pulled_images: set[str] = set()
|
||||||
for instance_id in plan["selection"]["instance_ids"]:
|
selected = plan["selection"]["instance_ids"]
|
||||||
|
for position, instance_id in enumerate(selected, 1):
|
||||||
|
prefix = f"[{position}/{len(selected)}] {instance_id}"
|
||||||
task_dir = out / instance_id.replace("/", "_")
|
task_dir = out / instance_id.replace("/", "_")
|
||||||
manifest_path = task_dir / "manifest.json"
|
manifest_path = task_dir / "manifest.json"
|
||||||
if not manifest_path.exists():
|
original, conflicting = pairs[instance_id]
|
||||||
original, conflicting = load_pair(plan["dataset"]["revision"], instance_id)
|
if manifest_path.exists():
|
||||||
|
print(f"{prefix}: reusing validated manifest", flush=True)
|
||||||
|
else:
|
||||||
|
print(f"{prefix}: pulling exact image", flush=True)
|
||||||
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
|
pull_image_once(swebench_spec(original)[0], pulled_images, os.environ)
|
||||||
results = [
|
results = []
|
||||||
run_trial(record, split=split, mode=mode, out_dir=task_dir,
|
for split, record in (("original", original), ("conflicting", conflicting)):
|
||||||
environ=os.environ,
|
for mode in ("nochange", "oracle"):
|
||||||
memory=plan["parameters"]["memory"],
|
print(f"{prefix}: running {split}/{mode}", flush=True)
|
||||||
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"])
|
result = run_trial(
|
||||||
for split, record in (("original", original), ("conflicting", conflicting))
|
record, split=split, mode=mode, out_dir=task_dir,
|
||||||
for mode in ("nochange", "oracle")
|
environ=os.environ, memory=plan["parameters"]["memory"],
|
||||||
]
|
timeout_seconds=plan["parameters"]["scorer_timeout_seconds"],
|
||||||
|
)
|
||||||
|
require_resolved_targets(instance_id, result)
|
||||||
|
results.append(result)
|
||||||
validate_expected_matrix(results)
|
validate_expected_matrix(results)
|
||||||
value = trial_manifest(
|
value = trial_manifest(
|
||||||
plan["dataset"]["revision"], instance_id, original, conflicting, results
|
plan["dataset"]["revision"], instance_id, original, conflicting, results
|
||||||
@@ -83,6 +140,7 @@ def main() -> int:
|
|||||||
with manifest_path.open("x") as handle:
|
with manifest_path.open("x") as handle:
|
||||||
json.dump(value, handle, indent=2, sort_keys=True)
|
json.dump(value, handle, indent=2, sort_keys=True)
|
||||||
handle.write("\n")
|
handle.write("\n")
|
||||||
|
print(f"{prefix}: validation passed", flush=True)
|
||||||
entries[instance_id] = {
|
entries[instance_id] = {
|
||||||
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
|
"path": str(manifest_path.relative_to(ROOT)), "sha256": sha(manifest_path)
|
||||||
}
|
}
|
||||||
@@ -92,8 +150,7 @@ def main() -> int:
|
|||||||
with declared.open("x") as handle:
|
with declared.open("x") as handle:
|
||||||
json.dump(index, handle, indent=2, sort_keys=True)
|
json.dump(index, handle, indent=2, sort_keys=True)
|
||||||
handle.write("\n")
|
handle.write("\n")
|
||||||
records = load_records(plan["dataset"]["revision"], "conflicting")
|
result = validate_environment_index_for_records(plan, ROOT, conflicting_records)
|
||||||
result = validate_environment_index_for_records(plan, ROOT, records)
|
|
||||||
print(json.dumps({"status": "validated", **result}, indent=2))
|
print(json.dumps({"status": "validated", **result}, indent=2))
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|||||||
@@ -44,73 +44,15 @@ def validate_environment_index_for_records(
|
|||||||
or set(index.get("manifests", {})) != set(selected)):
|
or set(index.get("manifests", {})) != set(selected)):
|
||||||
raise ValueError("environment validation index does not match the frozen plan")
|
raise ValueError("environment validation index does not match the frozen plan")
|
||||||
evidence = []
|
evidence = []
|
||||||
expected_cells = {
|
|
||||||
("original", "nochange"): False,
|
|
||||||
("original", "oracle"): True,
|
|
||||||
("conflicting", "nochange"): False,
|
|
||||||
("conflicting", "oracle"): False,
|
|
||||||
}
|
|
||||||
for instance_id in selected:
|
for instance_id in selected:
|
||||||
entry = index["manifests"][instance_id]
|
entry = index["manifests"][instance_id]
|
||||||
manifest_path = _local(root, entry.get("path"))
|
manifest_path = _local(root, entry.get("path"))
|
||||||
if _sha(manifest_path) != entry.get("sha256"):
|
if _sha(manifest_path) != entry.get("sha256"):
|
||||||
raise ValueError(f"validation manifest hash mismatch: {instance_id}")
|
raise ValueError(f"validation manifest hash mismatch: {instance_id}")
|
||||||
manifest = json.loads(manifest_path.read_text())
|
record = records.get(instance_id) if records is not None else None
|
||||||
if (manifest.get("schema_version") != 1
|
if records is not None and record is None:
|
||||||
or manifest.get("dataset") != DATASET
|
raise ValueError(f"frozen validation record missing: {instance_id}")
|
||||||
or manifest.get("dataset_revision") != plan["dataset"]["revision"]
|
validate_task_manifest(plan, instance_id, manifest_path, record)
|
||||||
or manifest.get("instance_id") != instance_id
|
|
||||||
or manifest.get("network") != "none"):
|
|
||||||
raise ValueError(f"validation manifest identity mismatch: {instance_id}")
|
|
||||||
if records is not None:
|
|
||||||
record = records.get(instance_id)
|
|
||||||
if record is None:
|
|
||||||
raise ValueError(f"frozen validation record missing: {instance_id}")
|
|
||||||
canonical = hashlib.sha256(json.dumps(
|
|
||||||
dict(record), sort_keys=True, separators=(",", ":")
|
|
||||||
).encode()).hexdigest()
|
|
||||||
expected_hashes = {
|
|
||||||
"base_commit": record.get("base_commit"),
|
|
||||||
"repo": record.get("repo"),
|
|
||||||
"version": record.get("version"),
|
|
||||||
"original_test_patch_sha256": hashlib.sha256(
|
|
||||||
str(record.get("original_test_patch", "")).encode()
|
|
||||||
).hexdigest(),
|
|
||||||
"conflicting_test_patch_sha256": hashlib.sha256(
|
|
||||||
str(record.get("test_patch", "")).encode()
|
|
||||||
).hexdigest(),
|
|
||||||
"oracle_patch_sha256": hashlib.sha256(
|
|
||||||
str(record.get("patch", "")).encode()
|
|
||||||
).hexdigest(),
|
|
||||||
}
|
|
||||||
if (canonical != plan.get("records_sha256", {}).get(instance_id)
|
|
||||||
or any(manifest.get(key) != value
|
|
||||||
for key, value in expected_hashes.items())):
|
|
||||||
raise ValueError(f"validation patches do not match frozen record: {instance_id}")
|
|
||||||
results = manifest.get("results")
|
|
||||||
if not isinstance(results, list):
|
|
||||||
raise ValueError(f"validation results missing: {instance_id}")
|
|
||||||
cells = {(row.get("split"), row.get("mode")): row for row in results}
|
|
||||||
if set(cells) != set(expected_cells) or len(results) != 4:
|
|
||||||
raise ValueError(f"validation matrix incomplete: {instance_id}")
|
|
||||||
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
|
|
||||||
for row in results}
|
|
||||||
if len(identities) != 1 or any(
|
|
||||||
cells[cell].get("resolved") is not expected
|
|
||||||
for cell, expected in expected_cells.items()
|
|
||||||
):
|
|
||||||
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
|
|
||||||
if any(
|
|
||||||
not row.get("target_statuses")
|
|
||||||
or any(status in {"MISSING", "ERROR"}
|
|
||||||
for status in row["target_statuses"].values())
|
|
||||||
for row in results
|
|
||||||
):
|
|
||||||
raise ValueError(f"validation contains missing/error targets: {instance_id}")
|
|
||||||
for row in results:
|
|
||||||
output = manifest_path.parent / str(row.get("output_file", ""))
|
|
||||||
if not output.is_file() or _sha(output) != row.get("output_sha256"):
|
|
||||||
raise ValueError(f"validation output hash mismatch: {instance_id}")
|
|
||||||
evidence.append({
|
evidence.append({
|
||||||
"instance_id": instance_id,
|
"instance_id": instance_id,
|
||||||
"manifest_path": str(manifest_path),
|
"manifest_path": str(manifest_path),
|
||||||
@@ -118,3 +60,72 @@ def validate_environment_index_for_records(
|
|||||||
})
|
})
|
||||||
return {"index_path": str(index_path), "index_sha256": _sha(index_path),
|
return {"index_path": str(index_path), "index_sha256": _sha(index_path),
|
||||||
"validated_instances": evidence}
|
"validated_instances": evidence}
|
||||||
|
|
||||||
|
|
||||||
|
def validate_task_manifest(
|
||||||
|
plan: Mapping[str, Any],
|
||||||
|
instance_id: str,
|
||||||
|
manifest_path: Path,
|
||||||
|
record: Mapping[str, Any] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Validate one task manifest, including partial evidence during resume."""
|
||||||
|
manifest = json.loads(manifest_path.read_text())
|
||||||
|
if (manifest.get("schema_version") != 1
|
||||||
|
or manifest.get("dataset") != DATASET
|
||||||
|
or manifest.get("dataset_revision") != plan["dataset"]["revision"]
|
||||||
|
or manifest.get("instance_id") != instance_id
|
||||||
|
or manifest.get("network") != "none"):
|
||||||
|
raise ValueError(f"validation manifest identity mismatch: {instance_id}")
|
||||||
|
if record is not None:
|
||||||
|
canonical = hashlib.sha256(json.dumps(
|
||||||
|
dict(record), sort_keys=True, separators=(",", ":")
|
||||||
|
).encode()).hexdigest()
|
||||||
|
expected_hashes = {
|
||||||
|
"base_commit": record.get("base_commit"),
|
||||||
|
"repo": record.get("repo"),
|
||||||
|
"version": record.get("version"),
|
||||||
|
"original_test_patch_sha256": hashlib.sha256(
|
||||||
|
str(record.get("original_test_patch", "")).encode()
|
||||||
|
).hexdigest(),
|
||||||
|
"conflicting_test_patch_sha256": hashlib.sha256(
|
||||||
|
str(record.get("test_patch", "")).encode()
|
||||||
|
).hexdigest(),
|
||||||
|
"oracle_patch_sha256": hashlib.sha256(
|
||||||
|
str(record.get("patch", "")).encode()
|
||||||
|
).hexdigest(),
|
||||||
|
}
|
||||||
|
if (canonical != plan.get("records_sha256", {}).get(instance_id)
|
||||||
|
or any(manifest.get(key) != value
|
||||||
|
for key, value in expected_hashes.items())):
|
||||||
|
raise ValueError(f"validation patches do not match frozen record: {instance_id}")
|
||||||
|
results = manifest.get("results")
|
||||||
|
if not isinstance(results, list):
|
||||||
|
raise ValueError(f"validation results missing: {instance_id}")
|
||||||
|
expected_cells = {
|
||||||
|
("original", "nochange"): False,
|
||||||
|
("original", "oracle"): True,
|
||||||
|
("conflicting", "nochange"): False,
|
||||||
|
("conflicting", "oracle"): False,
|
||||||
|
}
|
||||||
|
cells = {(row.get("split"), row.get("mode")): row for row in results}
|
||||||
|
if set(cells) != set(expected_cells) or len(results) != 4:
|
||||||
|
raise ValueError(f"validation matrix incomplete: {instance_id}")
|
||||||
|
if any(
|
||||||
|
not row.get("target_statuses")
|
||||||
|
or any(status in {"MISSING", "ERROR"}
|
||||||
|
for status in row["target_statuses"].values())
|
||||||
|
for row in results
|
||||||
|
):
|
||||||
|
raise ValueError(f"validation contains missing/error targets: {instance_id}")
|
||||||
|
identities = {(row.get("image_id"), tuple(row.get("repo_digests") or []))
|
||||||
|
for row in results}
|
||||||
|
if len(identities) != 1 or any(
|
||||||
|
cells[cell].get("resolved") is not expected
|
||||||
|
for cell, expected in expected_cells.items()
|
||||||
|
):
|
||||||
|
raise ValueError(f"validation matrix outcome mismatch: {instance_id}")
|
||||||
|
for row in results:
|
||||||
|
output = manifest_path.parent / str(row.get("output_file", ""))
|
||||||
|
if not output.is_file() or _sha(output) != row.get("output_sha256"):
|
||||||
|
raise ValueError(f"validation output hash mismatch: {instance_id}")
|
||||||
|
return manifest
|
||||||
@@ -5,11 +5,18 @@ import json
|
|||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
|
from messageboardbench.swe_validation import ValidationError
|
||||||
from messageboardbench.swe_prerequisites import (
|
from messageboardbench.swe_prerequisites import (
|
||||||
validate_environment_index,
|
validate_environment_index,
|
||||||
validate_environment_index_for_records,
|
validate_environment_index_for_records,
|
||||||
|
validate_task_manifest,
|
||||||
|
)
|
||||||
|
from scripts.validate_swe_population_prerequisites import (
|
||||||
|
load_selected_pairs,
|
||||||
|
pull_image_once,
|
||||||
|
require_resolved_targets,
|
||||||
|
validate_existing_manifests,
|
||||||
)
|
)
|
||||||
from scripts.validate_swe_population_prerequisites import pull_image_once
|
|
||||||
|
|
||||||
|
|
||||||
def write(path, value):
|
def write(path, value):
|
||||||
@@ -105,3 +112,47 @@ def test_image_is_pulled_once_before_any_inspection_or_trial():
|
|||||||
["docker", "pull", "image:tag"],
|
["docker", "pull", "image:tag"],
|
||||||
["docker", "image", "inspect", "image:tag"],
|
["docker", "image", "inspect", "image:tag"],
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_selected_pairs_load_each_dataset_split_exactly_once():
|
||||||
|
calls = []
|
||||||
|
common = {
|
||||||
|
"instance_id": "task", "repo": "org/repo", "version": "1",
|
||||||
|
"base_commit": "base", "patch": "oracle", "original_test_patch": "original",
|
||||||
|
"FAIL_TO_PASS": ["target"], "PASS_TO_PASS": [],
|
||||||
|
}
|
||||||
|
rows = {
|
||||||
|
"original": {"task": {**common, "test_patch": "original"}},
|
||||||
|
"conflicting": {"task": {**common, "test_patch": "conflict"}},
|
||||||
|
}
|
||||||
|
|
||||||
|
def loader(revision, split):
|
||||||
|
calls.append((revision, split))
|
||||||
|
return rows[split]
|
||||||
|
|
||||||
|
plan = {"dataset": {"revision": "1" * 40},
|
||||||
|
"selection": {"instance_ids": ["task"]}}
|
||||||
|
pairs, conflicting = load_selected_pairs(plan, loader=loader)
|
||||||
|
assert calls == [("1" * 40, "original"), ("1" * 40, "conflicting")]
|
||||||
|
assert pairs["task"] == (rows["original"]["task"], rows["conflicting"]["task"])
|
||||||
|
assert conflicting is rows["conflicting"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_new_cell_rejects_missing_targets_immediately():
|
||||||
|
result = __import__("types").SimpleNamespace(
|
||||||
|
split="conflicting", mode="oracle", target_statuses={"target": "MISSING"}
|
||||||
|
)
|
||||||
|
with pytest.raises(ValidationError, match="missing/error targets.*conflicting/oracle"):
|
||||||
|
require_resolved_targets("task", result)
|
||||||
|
|
||||||
|
|
||||||
|
def test_resume_rejects_existing_missing_manifest_before_reuse(tmp_path):
|
||||||
|
plan, manifest_path, record = fixture(tmp_path)
|
||||||
|
manifest = json.loads(manifest_path.read_text())
|
||||||
|
manifest["results"][0]["target_statuses"] = {"target": "ERROR"}
|
||||||
|
task_dir = tmp_path / "task"
|
||||||
|
for row in manifest["results"]:
|
||||||
|
write(task_dir / row["output_file"], "test output")
|
||||||
|
write(task_dir / "manifest.json", manifest)
|
||||||
|
with pytest.raises(ValueError, match="missing/error targets"):
|
||||||
|
validate_existing_manifests(plan, tmp_path, {"task": record})
|
||||||
Reference in new issue
Block a user