archive historical results and add token budget logs

This commit is contained in:
pj committed 2026-09-25 22:07:26 +05:30
1 parent 638e978227
commit e94b75c6f5
813 files changed
+12631 -410710

No files matched your search

+4 -4
View File
@@ -2,7 +2,7 @@
These entrypoints recompute existing evidence without model requests. Run from
`messageboardbench` with the installed `.venv`; output must be a fresh directory
outside the input evidence. The originals inside `results/` are frozen historical
outside the input evidence. The originals inside `archive/results/` are frozen historical
scripts, including their original paths. Use these portable copies for reanalysis.
```sh
@@ -10,13 +10,13 @@ scripts, including their original paths. Use these portable copies for reanalysi
.venv/bin/python scripts/analysis/token_audit.py --out work/historical-token-reanalysis
```
`board_synthesis.py` defaults to `results/board-pilot-sept8` and
`results/board-interface-v2-sept8`; override with `--results-v1` / `--results-v2`.
`board_synthesis.py` defaults to `archive/results/board-pilot-sept8` and
`archive/results/board-interface-v2-sept8`; override with `--results-v1` / `--results-v2`.
It joins the twelve v2 episodes to existing reviewed labels and compares descriptive
metrics with v1. It does not classify new trajectories or support arbitrary runs.
`review_file` paths in its outputs are relative to the input v2 evidence directory.
`token_audit.py` defaults to `logs/`; override with `--logs`. It requires all five
`token_audit.py` defaults to `archive/logs/`; override with `--logs`. It requires all five
historical run directories in its frozen inclusion list. It reproduces that audit's
metrics; newly created logs are excluded. Neither script infers honesty from a
failed attempt, estimates causal effects, or modifies the source reports/logs.
+2 -2
View File
@@ -76,8 +76,8 @@ def interface_stats(root, rows):
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--results-v1', type=Path, default=BENCH / 'results/board-pilot-sept8')
parser.add_argument('--results-v2', type=Path, default=BENCH / 'results/board-interface-v2-sept8')
parser.add_argument('--results-v1', type=Path, default=BENCH / 'archive/results/board-pilot-sept8')
parser.add_argument('--results-v2', type=Path, default=BENCH / 'archive/results/board-interface-v2-sept8')
parser.add_argument('--out', type=Path, required=True, help='Fresh output directory; frozen evidence is never overwritten.')
args = parser.parse_args(argv)
v1, root, out = args.results_v1.resolve(), args.results_v2.resolve(), args.out.resolve()
+1 -1
View File
@@ -19,7 +19,7 @@ def summarize(rows):
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--logs', type=Path, default=BENCH / 'logs')
parser.add_argument('--logs', type=Path, default=BENCH / 'archive/logs')
parser.add_argument('--out', type=Path, required=True, help='Fresh output directory for derived analysis.')
args = parser.parse_args(argv)
logs, out = args.logs.resolve(), args.out.resolve()
+2
View File
@@ -11,6 +11,8 @@ def main():
failures = []
for entry in manifest['files']:
path = ROOT / entry['path']
if entry['path'].startswith(('results/', 'docs/research/')):
path = ROOT / 'archive' / entry['path']
if not path.is_file():
failures.append({'path': entry['path'], 'reason': 'missing'})
continue