Files
sanderling/cmd/internal-tools/analyze/holm_test.go
T
pj 019d608f65 feat(analyze): survival analysis over campaign directories
Steps to first violation with clean runs right-censored at the budget, since
per-run yield is a binary at 11 to 45 percent and separating two arms on it
would need roughly 80 runs per arm. Kaplan-Meier, log-rank, Wilcoxon rank-sum
with Vargha-Delaney A12, Holm within each family.

A hand-rolled log-rank that is subtly wrong is a silent-wrong-number generator
and would be believed, so every statistic is validated against a published
worked example with the source named in the test: R survdiff on aml, Freireich
6-MP, Hollander and Wolfe 1973 for the rank sum, printed p.adjust output for
Holm. Two could not be: the k>2 log-rank, guarded by calibration instead, and
the tie-corrected variance, checked against an exact permutation variance.

Failed and timed-out runs are excluded as missing data and counted by reason,
never treated as censored observations, which would bias the result.

Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
2026-08-12 23:03:03 +05:30

55 lines
1.6 KiB
Go

package main
import (
"math"
"testing"
)
// Both cases are printed R output in Eve Slavich, "Four strategies for dealing
// with multiple comparisons", UNSW Stats Central, slides 9 and 10:
//
// pValues = c(0.01, 0.2, 0.08, 0.03)
// p.adjust(pValues, method = "holm")
// ## [1] 0.04 0.20 0.16 0.09
//
// pValues = c(0.01, 0.2, 0.08, 0.03, 0.02, 0.01)
// p.adjust(pValues, method = "holm")
// ## [1] 0.06 0.20 0.16 0.09 0.08 0.06
//
// The second case exercises the monotonicity step: sorted p are
// .01 .01 .02 .03 .08 .20, scaled by 6 5 4 3 2 1 to .06 .05 .08 .09 .16 .20,
// and the running maximum lifts the second back to .06.
func TestHolm_MatchesPublishedAdjustment(t *testing.T) {
cases := []struct {
raw []float64
expected []float64
}{
{[]float64{0.01, 0.2, 0.08, 0.03}, []float64{0.04, 0.20, 0.16, 0.09}},
{[]float64{0.01, 0.2, 0.08, 0.03, 0.02, 0.01}, []float64{0.06, 0.20, 0.16, 0.09, 0.08, 0.06}},
}
for _, test := range cases {
adjusted := holm(test.raw)
for index, want := range test.expected {
if math.Abs(adjusted[index]-want) > 1e-12 {
t.Errorf("holm(%v)[%d] = %v, want %v", test.raw, index, adjusted[index], want)
}
}
}
}
func TestHolm_CapsAtOneAndKeepsOrder(t *testing.T) {
adjusted := holm([]float64{0.4, 0.5, 0.9})
for index, value := range adjusted {
if value != 1 {
t.Errorf("adjusted[%d] = %v, want 1", index, value)
}
}
single := holm([]float64{0.03})
if len(single) != 1 || single[0] != 0.03 {
t.Errorf("single comparison adjusted to %v, want 0.03 unchanged", single)
}
if got := holm(nil); len(got) != 0 {
t.Errorf("holm(nil) = %v, want empty", got)
}
}