mirror of
https://github.com/priyanshujain/messageboardbench.git
synced 2026-10-02 19:17:07 +00:00
archive historical results and add token budget logs
This commit is contained in:
1 parent
638e978227
commit
e94b75c6f5
813 files changed
+12631
-410710
No files matched your search
@@ -2,7 +2,7 @@
|
||||
|
||||
These entrypoints recompute existing evidence without model requests. Run from
|
||||
`messageboardbench` with the installed `.venv`; output must be a fresh directory
|
||||
outside the input evidence. The originals inside `results/` are frozen historical
|
||||
outside the input evidence. The originals inside `archive/results/` are frozen historical
|
||||
scripts, including their original paths. Use these portable copies for reanalysis.
|
||||
|
||||
```sh
|
||||
@@ -10,13 +10,13 @@ scripts, including their original paths. Use these portable copies for reanalysi
|
||||
.venv/bin/python scripts/analysis/token_audit.py --out work/historical-token-reanalysis
|
||||
```
|
||||
|
||||
`board_synthesis.py` defaults to `results/board-pilot-sept8` and
|
||||
`results/board-interface-v2-sept8`; override with `--results-v1` / `--results-v2`.
|
||||
`board_synthesis.py` defaults to `archive/results/board-pilot-sept8` and
|
||||
`archive/results/board-interface-v2-sept8`; override with `--results-v1` / `--results-v2`.
|
||||
It joins the twelve v2 episodes to existing reviewed labels and compares descriptive
|
||||
metrics with v1. It does not classify new trajectories or support arbitrary runs.
|
||||
`review_file` paths in its outputs are relative to the input v2 evidence directory.
|
||||
|
||||
`token_audit.py` defaults to `logs/`; override with `--logs`. It requires all five
|
||||
`token_audit.py` defaults to `archive/logs/`; override with `--logs`. It requires all five
|
||||
historical run directories in its frozen inclusion list. It reproduces that audit's
|
||||
metrics; newly created logs are excluded. Neither script infers honesty from a
|
||||
failed attempt, estimates causal effects, or modifies the source reports/logs.
|
||||
|
||||
@@ -76,8 +76,8 @@ def interface_stats(root, rows):
|
||||
|
||||
def main(argv=None):
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument('--results-v1', type=Path, default=BENCH / 'results/board-pilot-sept8')
|
||||
parser.add_argument('--results-v2', type=Path, default=BENCH / 'results/board-interface-v2-sept8')
|
||||
parser.add_argument('--results-v1', type=Path, default=BENCH / 'archive/results/board-pilot-sept8')
|
||||
parser.add_argument('--results-v2', type=Path, default=BENCH / 'archive/results/board-interface-v2-sept8')
|
||||
parser.add_argument('--out', type=Path, required=True, help='Fresh output directory; frozen evidence is never overwritten.')
|
||||
args = parser.parse_args(argv)
|
||||
v1, root, out = args.results_v1.resolve(), args.results_v2.resolve(), args.out.resolve()
|
||||
|
||||
@@ -19,7 +19,7 @@ def summarize(rows):
|
||||
|
||||
def main(argv=None):
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument('--logs', type=Path, default=BENCH / 'logs')
|
||||
parser.add_argument('--logs', type=Path, default=BENCH / 'archive/logs')
|
||||
parser.add_argument('--out', type=Path, required=True, help='Fresh output directory for derived analysis.')
|
||||
args = parser.parse_args(argv)
|
||||
logs, out = args.logs.resolve(), args.out.resolve()
|
||||
|
||||
Reference in new issue
Block a user