fix(analyze): compare arms on censored runs, not on flattened step counts

stepTimes threw the censoring flag away and handed the rank-sum a plain
number per run, so a run the wall clock stopped at step 12 was ranked as one
that violated at step 12. That was defensible while every clean run sat at
the budget, the largest value any run could take, and it stopped being
defensible when a clean run started being censored where it stopped.

Twenty runs clean at step 12 against twenty violations at step 100 read a12
0.000 and p 4.683e-10 from the rank-sum, in the same report as a log-rank
reading p 1.0000. The pairwise comparison is now the Gehan test over the
observations themselves, and the report says how many run pairs censoring
left with no order between them, which is how much of the effect size is the
null value rather than an observation.
This commit is contained in:
pj committed 2026-08-18 20:13:12 +05:30
1 parent b7ee23942e
commit f0726a1e61
6 files changed
+144 -52

No files matched your search

@@ -92,7 +92,7 @@ func TestRun_EndToEndOverFixtureCampaignDirectories(t *testing.T) {
for _, fragment := range []string{
"steps to first violation, right-censored at the last step a clean run reached",
"log-rank across 2 arms",
"pairwise wilcoxon rank-sum",
"pairwise gehan generalized wilcoxon",
"llm vs seeded",
"excluded 1 run(s) as missing data",
} {
@@ -185,8 +185,10 @@ func TestRun_EndToEndOverFixtureCampaignDirectories(t *testing.T) {
if pair.HolmPValue != pair.PValue {
t.Errorf("holm p %v differs from raw p %v in a family of one", pair.HolmPValue, pair.PValue)
}
if pair.Exact {
t.Error("used the exact null distribution despite the tie mass at the budget")
// Six seeded runs and one llm run ran the budget clean, and two censored
// runs have no order between them whatever step either stopped on.
if pair.Unordered < 6 {
t.Errorf("%d unordered pair(s), want at least the six pairs of censored runs", pair.Unordered)
}
}