Files
sanderling/cmd/internal-tools/analyze/distribution_test.go
T
pj 019d608f65 feat(analyze): survival analysis over campaign directories
Steps to first violation with clean runs right-censored at the budget, since
per-run yield is a binary at 11 to 45 percent and separating two arms on it
would need roughly 80 runs per arm. Kaplan-Meier, log-rank, Wilcoxon rank-sum
with Vargha-Delaney A12, Holm within each family.

A hand-rolled log-rank that is subtly wrong is a silent-wrong-number generator
and would be believed, so every statistic is validated against a published
worked example with the source named in the test: R survdiff on aml, Freireich
6-MP, Hollander and Wolfe 1973 for the rank sum, printed p.adjust output for
Holm. Two could not be: the k>2 log-rank, guarded by calibration instead, and
the tie-corrected variance, checked against an exact permutation variance.

Failed and timed-out runs are excluded as missing data and counted by reason,
never treated as censored observations, which would bias the result.

Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
2026-08-12 23:03:03 +05:30

75 lines
2.2 KiB
Go

package main
import (
"math"
"testing"
)
// Chi-square critical values are the standard published table entries: the
// upper-tail probability of each of these statistics is the stated alpha in any
// chi-square table, for example Pearson and Hartley, Biometrika Tables for
// Statisticians, Table 8.
func TestChiSquareUpperTail_MatchesPublishedCriticalValues(t *testing.T) {
cases := []struct {
statistic float64
degreesOfFreedom int
expected float64
}{
{3.841459, 1, 0.05},
{6.634897, 1, 0.01},
{10.827566, 1, 0.001},
{5.991465, 2, 0.05},
{9.210340, 2, 0.01},
{7.814728, 3, 0.05},
{11.344867, 3, 0.01},
{9.487729, 4, 0.05},
{18.307038, 10, 0.05},
}
for _, test := range cases {
got := chiSquareUpperTail(test.statistic, test.degreesOfFreedom)
if math.Abs(got-test.expected) > 1e-6 {
t.Errorf("chiSquareUpperTail(%v, %d) = %v, want %v", test.statistic, test.degreesOfFreedom, got, test.expected)
}
}
}
// For one degree of freedom the upper tail has the closed form erfc(sqrt(x/2)),
// which is an independent check on the incomplete gamma routine.
func TestChiSquareUpperTail_AgreesWithClosedFormAtOneDegreeOfFreedom(t *testing.T) {
for _, statistic := range []float64{0.1, 1, 3.4, 16.79, 40, 120} {
expected := math.Erfc(math.Sqrt(statistic / 2))
got := chiSquareUpperTail(statistic, 1)
if math.Abs(got-expected) > 1e-12*math.Max(1, expected) {
t.Errorf("chiSquareUpperTail(%v, 1) = %v, want %v", statistic, got, expected)
}
}
}
func TestChiSquareUpperTail_ZeroStatisticIsCertain(t *testing.T) {
if got := chiSquareUpperTail(0, 1); got != 1 {
t.Errorf("chiSquareUpperTail(0, 1) = %v, want 1", got)
}
}
// Standard normal quantiles from any published normal table.
func TestStandardNormalUpperTail_MatchesPublishedQuantiles(t *testing.T) {
cases := []struct {
z float64
expected float64
}{
{1.281552, 0.10},
{1.644854, 0.05},
{1.959964, 0.025},
{2.326348, 0.01},
{2.575829, 0.005},
{3.090232, 0.001},
{0, 0.5},
}
for _, test := range cases {
got := standardNormalUpperTail(test.z)
if math.Abs(got-test.expected) > 1e-6 {
t.Errorf("standardNormalUpperTail(%v) = %v, want %v", test.z, got, test.expected)
}
}
}