mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-04 20:17:06 +00:00
Checkpoint experiments through SWE pilot v2
This commit is contained in:
1 parent
abacd5c5e1
commit
72d77018d8
845 files changed
+431756
-41
No files matched your search
@@ -0,0 +1,13 @@
|
||||
# Independent task 12 validation
|
||||
|
||||
All **23/23 original assertions match** an independently written breadth-first search from the documented moves. All supplied strings have the declared length; no out-of-domain inputs were found. The conflicting suite matches **23/24**, with the sole failure being the added expectation 7 for `(4, "BBWW", "WWBB")`; the minimum is **3**.
|
||||
|
||||
A separate direct-string BFS confirms the three-move witness:
|
||||
|
||||
`BBWW.. → ..WWBB → WW..BB → WWBB..`
|
||||
|
||||
Each move selects two adjacent occupied cells, transfers their stones in order to the two empty cells, and makes the source pair empty. The goal is the target string in its original n positions with the two empty cells at the end. The main reference uses a black-stone bitmask plus an empty-pair index, explicit BFS layers, a compact array queue and a bytearray visited table. It never inspects model code and has no test-specific result overrides. Unequal black-stone counts are immediately impossible.
|
||||
|
||||
Full-suite validation took 0.376 seconds. For n=14, the visited allocation is at most 15×2^16 bytes (~0.94 MiB); queue bound is at most 15×binomial(14,7) states (~0.20 MiB as 4-byte indices). No large cases remain unresolved. This was reviewer-authored reference code executed locally, with no agent-code execution, network calls, model calls, or access to running experiment containers.
|
||||
|
||||
`reference.py` is reproducible from the adjacent parent task audit; `results.json` records every assertion, states discovered, runtime, and source hashes. My earlier informal message alleging short strings was a visual counting error and is superseded by explicit length validation here.
|
||||
@@ -0,0 +1,87 @@
|
||||
"""Independent explicit-state BFS from the task docstring, not agent source.
|
||||
|
||||
Black stones are 1 bits, white stones 0 bits, and a separate index locates
|
||||
adjacent empty cells. Goal is t followed by two empties, matching initial layout.
|
||||
Invalid n/string-length inputs are reported and never silently repaired.
|
||||
"""
|
||||
from array import array
|
||||
import ast
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
import time
|
||||
|
||||
|
||||
def distance(n, s, t):
|
||||
if len(s) != n or len(t) != n or not 2 <= n <= 14 or set(s+t) - {'B', 'W'}:
|
||||
raise ValueError('Input violates documented n/string domain')
|
||||
if s.count('B') != t.count('B'):
|
||||
return -1, 0
|
||||
width = n + 2
|
||||
mask_limit = (1 << width) - 1
|
||||
start_bits = sum((c == 'B') << i for i, c in enumerate(s))
|
||||
target_bits = sum((c == 'B') << i for i, c in enumerate(t))
|
||||
start = start_bits | (n << width)
|
||||
target = target_bits | (n << width)
|
||||
queue = array('I', [start])
|
||||
seen = bytearray((n+1) << width)
|
||||
seen[start] = 1
|
||||
head = 0
|
||||
depth = 0
|
||||
while head < len(queue):
|
||||
layer_end = len(queue)
|
||||
while head < layer_end:
|
||||
state = queue[head]
|
||||
head += 1
|
||||
if state == target:
|
||||
return depth, len(queue)
|
||||
empty = state >> width
|
||||
bits = state & mask_limit
|
||||
for src in range(n+1):
|
||||
if abs(src-empty) <= 1:
|
||||
continue # selected pair overlaps the two empty cells
|
||||
pair = (bits >> src) & 3
|
||||
moved = (bits & ~(3 << src)) | (pair << empty)
|
||||
nxt = moved | (src << width)
|
||||
if not seen[nxt]:
|
||||
seen[nxt] = 1
|
||||
queue.append(nxt)
|
||||
depth += 1
|
||||
return -1, len(queue)
|
||||
|
||||
|
||||
def main():
|
||||
here = Path(__file__).resolve().parent
|
||||
audit = here.parent / 'task-audit.json'
|
||||
data = json.loads(audit.read_text())['tasks']['lcbhard_12']
|
||||
results = []
|
||||
cache = {}
|
||||
started = time.monotonic()
|
||||
for split in ('original', 'conflicting'):
|
||||
for ordinal, node in enumerate((x for x in ast.walk(ast.parse(data[split]['test'])) if isinstance(x, ast.Assert)), 1):
|
||||
args = [ast.literal_eval(a) for a in node.test.left.args]
|
||||
expected = ast.literal_eval(node.test.comparators[0])
|
||||
n,s,t = args
|
||||
item = dict(split=split, assertion_ordinal=ordinal, source_line=node.lineno,
|
||||
args=args, expected=expected, lengths=[len(s),len(t)],
|
||||
black_counts=[s.count('B'),t.count('B')])
|
||||
tick = time.monotonic()
|
||||
if len(s) != n or len(t) != n:
|
||||
item.update(status='invalid_input', actual=None, reason='String lengths violate documented n; reference refuses implicit padding/truncation.')
|
||||
else:
|
||||
key = tuple(args)
|
||||
if key not in cache:
|
||||
cache[key] = distance(*args)
|
||||
actual, explored = cache[key]
|
||||
item.update(status='match' if actual == expected else 'mismatch', actual=actual, states_discovered=explored)
|
||||
item['seconds'] = round(time.monotonic()-tick,6)
|
||||
results.append(item)
|
||||
print(split,ordinal,args,expected,item['status'],item['actual'],flush=True)
|
||||
output = dict(method='Independent explicit-state BFS; two adjacent stone cells move in order into adjacent empty cells, goal t plus two empties. Invalid input rejected.',
|
||||
source_test_sha256={k:hashlib.sha256(data[k]['test'].encode()).hexdigest() for k in ('original','conflicting')},
|
||||
reference_sha256=hashlib.sha256(Path(__file__).read_bytes()).hexdigest(),
|
||||
elapsed_seconds=round(time.monotonic()-started,6), results=results)
|
||||
(here/'results.json').write_text(json.dumps(output,indent=2)+'\n')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user