Files
sanderling/cmd/internal-tools/analyze/paired.go
T
pj 1008402558 fix(analyze): score a seed pair by which run outlived the other
The paired path had the same defect as the unpaired one: it subtracted two
step counts and handed the differences to the signed-rank test, so a pair
holding a run the wall clock stopped at step 12 entered as a difference
neither run supports. Twenty seeds where the first arm was still clean at
step 12 and the second violated at step 5 in six of them read sign -1 and
p 0.0011, pointing at the arm that never violated.

A pair is now scored the way the unpaired comparison scores one and tested by
the exact sign test over the pairs whose order censoring determines, which is
what the log-rank stratified by seed reduces to here. The signed-rank goes
with the differences it needed: a magnitude-based paired test wants a
difference from every pair, and the arms censor on different clocks. The
median difference stays, over the pairs where both runs violated, and says so.
2026-08-18 20:16:40 +05:30

175 lines
6.0 KiB
Go

package main
import (
"fmt"
"math"
"slices"
)
// signTest is the exact two-sided sign test over matched pairs. Under the null
// that neither arm reaches its first violation sooner, a pair whose order the
// censoring determines falls either way with probability one half, so the count
// is binomial and the two-sided p-value doubles the smaller tail. Pairs left
// with no order carry no information and are not trials.
//
// It is the seed-matched form of the comparison the unpaired test makes, and
// it is what the log-rank stratified by seed reduces to with one run per arm in
// each stratum. The magnitude-based alternatives are not available: a
// difference in steps needs both runs to have violated, and a paired test built
// on scores of censored times, the paired Prentice-Wilcoxon among them, is
// centred at zero under the null only when the two arms censor alike, which is
// exactly what the wall clock stops them from doing.
func signTest(favouringFirst, favouringSecond int) float64 {
trials := favouringFirst + favouringSecond
if trials == 0 {
return math.NaN()
}
tail := 0.0
for count := 0; count <= min(favouringFirst, favouringSecond); count++ {
tail += math.Exp(logBinomialCoefficient(trials, count) - float64(trials)*math.Ln2)
}
return math.Min(2*tail, 1)
}
func logBinomialCoefficient(trials, chosen int) float64 {
all, _ := math.Lgamma(float64(trials + 1))
picked, _ := math.Lgamma(float64(chosen + 1))
rest, _ := math.Lgamma(float64(trials-chosen) + 1)
return all - picked - rest
}
// pairedComparison is the seed-matched contrast the actuation ablation reports.
// A pair is scored the way the unpaired comparison scores one, by which run
// outlived the other, so Sign is +1 when the second arm is the one seen to
// violate sooner across the pairs whose order censoring determines.
type pairedComparison struct {
First string `json:"first"`
Second string `json:"second"`
Pairs int `json:"pairs"`
UnpairedSeeds []int64 `json:"unpaired_seeds,omitempty"`
Sign int `json:"sign"`
FirstSooner int `json:"first_sooner"`
SecondSooner int `json:"second_sooner"`
// Unordered is the pairs the censoring leaves in no order, either because
// both runs ended clean or because the run that stopped first stopped before
// the other violated. They are not evidence either way and are not trials.
Unordered int `json:"unordered_pairs"`
// MedianDifference is in steps and is undefined unless some pair has both
// runs violating, which is the only shape a difference in steps can be read
// off. BothViolated says how many pairs it summarizes, because it describes
// those pairs and not the sample.
MedianDifference *float64 `json:"median_step_difference"`
BothViolated int `json:"both_violated_pairs"`
// A12 is the within-pair form of the Vargha-Delaney effect size, the share
// of matched seeds on which the first arm took more steps, an unordered pair
// counting as half. A matched design has no reason to compare the two arms
// as pooled bags of runs when each seed has a partner.
A12 float64 `json:"a12_within_pairs"`
PValue float64 `json:"p_value"`
HolmPValue float64 `json:"holm_p_value"`
}
// pairArms matches the two arms by seed and contrasts them pair by pair. A seed
// usable in one arm and not the other is named rather than dropped silently,
// because that is a host that lost a run and it is what the campaign manifest
// exists to make visible.
func pairArms(first, second arm) (pairedComparison, error) {
firstBySeed, err := usableBySeed(first)
if err != nil {
return pairedComparison{}, err
}
secondBySeed, err := usableBySeed(second)
if err != nil {
return pairedComparison{}, err
}
comparison := pairedComparison{
First: first.Name,
Second: second.Name,
A12: math.NaN(),
PValue: math.NaN(),
HolmPValue: math.NaN(),
}
var differences []float64
for _, seed := range sortedSeeds(firstBySeed, secondBySeed) {
left, inFirst := firstBySeed[seed]
right, inSecond := secondBySeed[seed]
if !inFirst || !inSecond {
comparison.UnpairedSeeds = append(comparison.UnpairedSeeds, seed)
continue
}
leftRun := observationOf(left, first.Budget)
rightRun := observationOf(right, second.Budget)
comparison.Pairs++
switch outlives(leftRun, rightRun) {
case 1:
comparison.SecondSooner++
case -1:
comparison.FirstSooner++
default:
comparison.Unordered++
}
if leftRun.Event && rightRun.Event {
comparison.BothViolated++
differences = append(differences, leftRun.Steps-rightRun.Steps)
}
}
if comparison.Pairs == 0 {
return comparison, nil
}
if len(differences) > 0 {
median := medianOf(differences)
comparison.MedianDifference = &median
}
switch {
case comparison.SecondSooner > comparison.FirstSooner:
comparison.Sign = 1
case comparison.FirstSooner > comparison.SecondSooner:
comparison.Sign = -1
}
comparison.A12 = (float64(comparison.SecondSooner) + 0.5*float64(comparison.Unordered)) / float64(comparison.Pairs)
comparison.PValue = signTest(comparison.FirstSooner, comparison.SecondSooner)
return comparison, nil
}
func usableBySeed(current arm) (map[int64]classifiedRun, error) {
bySeed := map[int64]classifiedRun{}
for _, item := range current.Runs {
if item.ExcludedBecause != "" {
continue
}
if _, repeated := bySeed[item.Seed]; repeated {
return nil, fmt.Errorf("arm %q has more than one usable run for seed %d: a seed-matched "+
"comparison cannot choose between them", current.Name, item.Seed)
}
bySeed[item.Seed] = item
}
return bySeed, nil
}
func sortedSeeds(sets ...map[int64]classifiedRun) []int64 {
var seeds []int64
seen := map[int64]bool{}
for _, set := range sets {
for seed := range set {
if seen[seed] {
continue
}
seen[seed] = true
seeds = append(seeds, seed)
}
}
slices.Sort(seeds)
return seeds
}
func medianOf(values []float64) float64 {
sorted := slices.Sorted(slices.Values(values))
middle := len(sorted) / 2
if len(sorted)%2 == 1 {
return sorted[middle]
}
return (sorted[middle-1] + sorted[middle]) / 2
}