feat(analyze): survival analysis over campaign directories

Steps to first violation with clean runs right-censored at the budget, since
per-run yield is a binary at 11 to 45 percent and separating two arms on it
would need roughly 80 runs per arm. Kaplan-Meier, log-rank, Wilcoxon rank-sum
with Vargha-Delaney A12, Holm within each family.

A hand-rolled log-rank that is subtly wrong is a silent-wrong-number generator
and would be believed, so every statistic is validated against a published
worked example with the source named in the test: R survdiff on aml, Freireich
6-MP, Hollander and Wolfe 1973 for the rank sum, printed p.adjust output for
Holm. Two could not be: the k>2 log-rank, guarded by calibration instead, and
the tie-corrected variance, checked against an exact permutation variance.

Failed and timed-out runs are excluded as missing data and counted by reason,
never treated as censored observations, which would bias the result.

Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
pj committed 2026-08-12 23:03:03 +05:30
1 parent 71dffef2f2
commit 019d608f65
16 files changed
+2595

No files matched your search

+102
View File
@@ -0,0 +1,102 @@
// Command analyze reduces campaign directories to the statistics the
// evaluation reports. The primary outcome is steps to first violation, with
// clean runs right-censored at the step budget rather than discarded: defect
// yield per run is a binary that would need on the order of eighty runs an arm
// to separate, while survival analysis uses every run, including the clean ones.
package main
import (
"encoding/json"
"errors"
"flag"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"time"
)
const usage = `analyze reports the statistics of a sanderling evaluation from campaign directories.
Usage:
analyze [--json <path>] <campaign-dir> [<campaign-dir> ...]
Each directory is one produced by the campaign tool and must hold campaign.json
and runs.jsonl. Directories sharing an arm label are pooled and must agree on
the step budget.
`
type stringList []string
func (list *stringList) String() string { return strings.Join(*list, ",") }
func (list *stringList) Set(value string) error {
if strings.TrimSpace(value) == "" {
return errors.New("empty campaign directory")
}
*list = append(*list, value)
return nil
}
func run(arguments []string, stdout, stderr io.Writer) error {
flagSet := flag.NewFlagSet("analyze", flag.ContinueOnError)
flagSet.SetOutput(stderr)
flagSet.Usage = func() {
fmt.Fprint(stderr, usage)
flagSet.PrintDefaults()
}
var directories stringList
var jsonPath string
flagSet.Var(&directories, "campaign", "campaign directory to read; repeat for more, or pass them as arguments")
flagSet.StringVar(&jsonPath, "json", "", "write the machine-readable summary here, or - for stdout")
if err := flagSet.Parse(arguments); err != nil {
return err
}
directories = append(directories, flagSet.Args()...)
if len(directories) == 0 {
return errors.New("no campaign directories given")
}
seen := map[string]bool{}
for _, directory := range directories {
resolved, err := filepath.Abs(directory)
if err != nil {
return fmt.Errorf("resolve %s: %w", directory, err)
}
if seen[resolved] {
return fmt.Errorf("campaign directory %s given twice: its runs would be counted twice", directory)
}
seen[resolved] = true
}
arms, err := groupArms(directories)
if err != nil {
return err
}
result := analyse(arms, time.Now().UTC())
writeReport(result, stdout)
if jsonPath == "" {
return nil
}
body, err := json.MarshalIndent(result, "", " ")
if err != nil {
return fmt.Errorf("marshal summary: %w", err)
}
body = append(body, '\n')
if jsonPath == "-" {
_, err = stdout.Write(body)
return err
}
return os.WriteFile(jsonPath, body, 0o644)
}
func main() {
if err := run(os.Args[1:], os.Stdout, os.Stderr); err != nil {
if errors.Is(err, flag.ErrHelp) {
return
}
fmt.Fprintf(os.Stderr, "error: %v\n", err)
os.Exit(1)
}
}