mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-04 20:17:09 +00:00
feat(analyze): survival analysis over campaign directories
Steps to first violation with clean runs right-censored at the budget, since per-run yield is a binary at 11 to 45 percent and separating two arms on it would need roughly 80 runs per arm. Kaplan-Meier, log-rank, Wilcoxon rank-sum with Vargha-Delaney A12, Holm within each family. A hand-rolled log-rank that is subtly wrong is a silent-wrong-number generator and would be believed, so every statistic is validated against a published worked example with the source named in the test: R survdiff on aml, Freireich 6-MP, Hollander and Wolfe 1973 for the rank sum, printed p.adjust output for Holm. Two could not be: the k>2 log-rank, guarded by calibration instead, and the tie-corrected variance, checked against an exact permutation variance. Failed and timed-out runs are excluded as missing data and counted by reason, never treated as censored observations, which would bias the result. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
71dffef2f2
commit
019d608f65
16 files changed
+2595
No files matched your search
@@ -0,0 +1,54 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"math"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// Both cases are printed R output in Eve Slavich, "Four strategies for dealing
|
||||
// with multiple comparisons", UNSW Stats Central, slides 9 and 10:
|
||||
//
|
||||
// pValues = c(0.01, 0.2, 0.08, 0.03)
|
||||
// p.adjust(pValues, method = "holm")
|
||||
// ## [1] 0.04 0.20 0.16 0.09
|
||||
//
|
||||
// pValues = c(0.01, 0.2, 0.08, 0.03, 0.02, 0.01)
|
||||
// p.adjust(pValues, method = "holm")
|
||||
// ## [1] 0.06 0.20 0.16 0.09 0.08 0.06
|
||||
//
|
||||
// The second case exercises the monotonicity step: sorted p are
|
||||
// .01 .01 .02 .03 .08 .20, scaled by 6 5 4 3 2 1 to .06 .05 .08 .09 .16 .20,
|
||||
// and the running maximum lifts the second back to .06.
|
||||
func TestHolm_MatchesPublishedAdjustment(t *testing.T) {
|
||||
cases := []struct {
|
||||
raw []float64
|
||||
expected []float64
|
||||
}{
|
||||
{[]float64{0.01, 0.2, 0.08, 0.03}, []float64{0.04, 0.20, 0.16, 0.09}},
|
||||
{[]float64{0.01, 0.2, 0.08, 0.03, 0.02, 0.01}, []float64{0.06, 0.20, 0.16, 0.09, 0.08, 0.06}},
|
||||
}
|
||||
for _, test := range cases {
|
||||
adjusted := holm(test.raw)
|
||||
for index, want := range test.expected {
|
||||
if math.Abs(adjusted[index]-want) > 1e-12 {
|
||||
t.Errorf("holm(%v)[%d] = %v, want %v", test.raw, index, adjusted[index], want)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHolm_CapsAtOneAndKeepsOrder(t *testing.T) {
|
||||
adjusted := holm([]float64{0.4, 0.5, 0.9})
|
||||
for index, value := range adjusted {
|
||||
if value != 1 {
|
||||
t.Errorf("adjusted[%d] = %v, want 1", index, value)
|
||||
}
|
||||
}
|
||||
single := holm([]float64{0.03})
|
||||
if len(single) != 1 || single[0] != 0.03 {
|
||||
t.Errorf("single comparison adjusted to %v, want 0.03 unchanged", single)
|
||||
}
|
||||
if got := holm(nil); len(got) != 0 {
|
||||
t.Errorf("holm(nil) = %v, want empty", got)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user