mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-04 20:17:09 +00:00
feat(analyze): survival analysis over campaign directories
Steps to first violation with clean runs right-censored at the budget, since per-run yield is a binary at 11 to 45 percent and separating two arms on it would need roughly 80 runs per arm. Kaplan-Meier, log-rank, Wilcoxon rank-sum with Vargha-Delaney A12, Holm within each family. A hand-rolled log-rank that is subtly wrong is a silent-wrong-number generator and would be believed, so every statistic is validated against a published worked example with the source named in the test: R survdiff on aml, Freireich 6-MP, Hollander and Wolfe 1973 for the rank sum, printed p.adjust output for Holm. Two could not be: the k>2 log-rank, guarded by calibration instead, and the tie-corrected variance, checked against an exact permutation variance. Failed and timed-out runs are excluded as missing data and counted by reason, never treated as censored observations, which would bias the result. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
71dffef2f2
commit
019d608f65
16 files changed
+2595
No files matched your search
@@ -0,0 +1,192 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"math"
|
||||
"slices"
|
||||
"time"
|
||||
)
|
||||
|
||||
type armSummary struct {
|
||||
Arm string `json:"arm"`
|
||||
Generator string `json:"generator,omitempty"`
|
||||
Platform string `json:"platform,omitempty"`
|
||||
StepBudget int `json:"step_budget"`
|
||||
Directories []string `json:"directories"`
|
||||
Recorded int `json:"recorded_runs"`
|
||||
Usable int `json:"usable_runs"`
|
||||
Violated int `json:"violated_runs"`
|
||||
Censored int `json:"censored_runs"`
|
||||
Excluded int `json:"excluded_runs"`
|
||||
ExcludedByReason map[string]int `json:"excluded_by_reason,omitempty"`
|
||||
MissingSeeds []int64 `json:"missing_seeds,omitempty"`
|
||||
EventsHeldAtBudget int `json:"events_held_at_budget"`
|
||||
MedianStepsToFirstViolation *float64 `json:"median_steps_to_first_violation"`
|
||||
SurvivalCurve []survivalPoint `json:"survival_curve,omitempty"`
|
||||
ViolationRate *float64 `json:"violation_rate"`
|
||||
TotalActions int `json:"total_actions"`
|
||||
TotalRunHours float64 `json:"total_run_hours"`
|
||||
Detections int `json:"detections"`
|
||||
DefectsPerThousandActions *float64 `json:"defects_per_thousand_actions"`
|
||||
DefectsPerHour *float64 `json:"defects_per_hour"`
|
||||
DistinctDefects int `json:"distinct_defects"`
|
||||
SingletonDefects int `json:"singleton_defects"`
|
||||
SingletonFraction *float64 `json:"singleton_fraction"`
|
||||
DefectRunCounts map[string]int `json:"defect_run_counts,omitempty"`
|
||||
}
|
||||
|
||||
type pairwiseResult struct {
|
||||
First string `json:"first"`
|
||||
Second string `json:"second"`
|
||||
FirstSize int `json:"first_size"`
|
||||
SecondSize int `json:"second_size"`
|
||||
Statistic float64 `json:"mann_whitney_u"`
|
||||
A12 float64 `json:"a12"`
|
||||
PValue float64 `json:"p_value"`
|
||||
HolmPValue float64 `json:"holm_p_value"`
|
||||
Exact bool `json:"exact"`
|
||||
}
|
||||
|
||||
type analysis struct {
|
||||
GeneratedAt time.Time `json:"generated_at"`
|
||||
Outcome string `json:"outcome"`
|
||||
Arms []armSummary `json:"arms"`
|
||||
LogRank *logRankResult `json:"log_rank"`
|
||||
Pairwise []pairwiseResult `json:"pairwise"`
|
||||
Notes []string `json:"notes,omitempty"`
|
||||
}
|
||||
|
||||
const outcomeDescription = "steps to first violation, right-censored at the step budget"
|
||||
|
||||
func analyse(arms []arm, now time.Time) analysis {
|
||||
result := analysis{GeneratedAt: now, Outcome: outcomeDescription}
|
||||
for _, current := range arms {
|
||||
result.Arms = append(result.Arms, summarize(current))
|
||||
}
|
||||
|
||||
var testable []arm
|
||||
for _, current := range arms {
|
||||
if len(current.observations()) > 0 {
|
||||
testable = append(testable, current)
|
||||
}
|
||||
}
|
||||
if len(testable) < len(arms) {
|
||||
result.Notes = append(result.Notes,
|
||||
"arms with no usable runs are reported but left out of the log-rank test and the pairwise comparisons")
|
||||
}
|
||||
if len(testable) >= 2 {
|
||||
names := make([]string, len(testable))
|
||||
groups := make([][]observation, len(testable))
|
||||
for index, current := range testable {
|
||||
names[index] = current.Name
|
||||
groups[index] = current.observations()
|
||||
}
|
||||
test := logRank(names, groups)
|
||||
result.LogRank = &test
|
||||
result.Pairwise = comparePairs(testable)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func comparePairs(arms []arm) []pairwiseResult {
|
||||
var pairs []pairwiseResult
|
||||
for first := 0; first < len(arms); first++ {
|
||||
for second := first + 1; second < len(arms); second++ {
|
||||
test := rankSum(arms[first].stepTimes(), arms[second].stepTimes())
|
||||
pairs = append(pairs, pairwiseResult{
|
||||
First: arms[first].Name,
|
||||
Second: arms[second].Name,
|
||||
FirstSize: test.FirstSize,
|
||||
SecondSize: test.SecondSize,
|
||||
Statistic: test.Statistic,
|
||||
A12: test.A12,
|
||||
PValue: test.PValue,
|
||||
HolmPValue: math.NaN(),
|
||||
Exact: test.Exact,
|
||||
})
|
||||
}
|
||||
}
|
||||
// Holm runs over this one family of comparisons. A comparison whose p-value
|
||||
// could not be computed is not part of the family and does not shrink the
|
||||
// correction the others receive.
|
||||
var family []int
|
||||
var raw []float64
|
||||
for index, pair := range pairs {
|
||||
if math.IsNaN(pair.PValue) {
|
||||
continue
|
||||
}
|
||||
family = append(family, index)
|
||||
raw = append(raw, pair.PValue)
|
||||
}
|
||||
for position, adjusted := range holm(raw) {
|
||||
pairs[family[position]].HolmPValue = adjusted
|
||||
}
|
||||
return pairs
|
||||
}
|
||||
|
||||
func summarize(current arm) armSummary {
|
||||
summary := armSummary{
|
||||
Arm: current.Name,
|
||||
Generator: current.Generator,
|
||||
Platform: current.Platform,
|
||||
StepBudget: current.Budget,
|
||||
Directories: current.Directories,
|
||||
Recorded: len(current.Runs),
|
||||
MissingSeeds: current.MissingSeeds,
|
||||
}
|
||||
runsPerDefect := map[string]int{}
|
||||
for _, item := range current.Runs {
|
||||
if item.ExcludedBecause != "" {
|
||||
summary.Excluded++
|
||||
if summary.ExcludedByReason == nil {
|
||||
summary.ExcludedByReason = map[string]int{}
|
||||
}
|
||||
summary.ExcludedByReason[item.ExcludedBecause]++
|
||||
continue
|
||||
}
|
||||
summary.Usable++
|
||||
summary.TotalActions += item.Steps
|
||||
summary.TotalRunHours += float64(item.DurationMillis) / float64(time.Hour/time.Millisecond)
|
||||
if item.ClampedToBudget {
|
||||
summary.EventsHeldAtBudget++
|
||||
}
|
||||
if item.Violated {
|
||||
summary.Violated++
|
||||
} else {
|
||||
summary.Censored++
|
||||
}
|
||||
distinct := slices.Compact(slices.Sorted(slices.Values(item.ViolatedProperties)))
|
||||
summary.Detections += len(distinct)
|
||||
for _, property := range distinct {
|
||||
runsPerDefect[property]++
|
||||
}
|
||||
}
|
||||
|
||||
summary.SurvivalCurve = kaplanMeier(current.observations())
|
||||
if median, ok := medianSurvival(summary.SurvivalCurve); ok {
|
||||
summary.MedianStepsToFirstViolation = &median
|
||||
}
|
||||
if summary.Usable > 0 {
|
||||
rate := float64(summary.Violated) / float64(summary.Usable)
|
||||
summary.ViolationRate = &rate
|
||||
}
|
||||
if summary.TotalActions > 0 {
|
||||
perThousand := 1000 * float64(summary.Detections) / float64(summary.TotalActions)
|
||||
summary.DefectsPerThousandActions = &perThousand
|
||||
}
|
||||
if summary.TotalRunHours > 0 {
|
||||
perHour := float64(summary.Detections) / summary.TotalRunHours
|
||||
summary.DefectsPerHour = &perHour
|
||||
}
|
||||
if len(runsPerDefect) > 0 {
|
||||
summary.DefectRunCounts = runsPerDefect
|
||||
summary.DistinctDefects = len(runsPerDefect)
|
||||
for _, count := range runsPerDefect {
|
||||
if count == 1 {
|
||||
summary.SingletonDefects++
|
||||
}
|
||||
}
|
||||
fraction := float64(summary.SingletonDefects) / float64(summary.DistinctDefects)
|
||||
summary.SingletonFraction = &fraction
|
||||
}
|
||||
return summary
|
||||
}
|
||||
Reference in new issue
Block a user