mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-04 20:17:09 +00:00
fix(analyze): censor a clean run at the steps it ran, and refuse mismatched budgets
A run stops at whichever comes first, the step budget or --duration, so a clean run that reached the wall clock exited with fewer steps than the budget and was still credited with the whole of it. The model arm pays a network call and a screenshot per step, so it reaches the wall sooner and was handed exposure it never had. Nothing checked that two arms shared a budget either. Thirty identical clean runs under budgets of 400 and 100 read a12 0.000 and p 1.685e-14 from the rank-sum while the log-rank in the same report read p 1.0000. groupArms already refused this within one arm. The claims the old convention left in comments and report lines are corrected rather than left standing beside the new behaviour.
This commit is contained in:
1 parent
ff6c66a74b
commit
1da0c3e118
10 files changed
+141
-41
No files matched your search
@@ -1,8 +1,10 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"io"
|
||||
"math"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
@@ -227,11 +229,14 @@ func TestAnalyse_ExcludedRunsNeverBecomeObservations(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestAnalyse_ArmWithNoUsableRunsIsReportedButNotTested(t *testing.T) {
|
||||
result := analyse([]arm{
|
||||
result, err := analyse([]arm{
|
||||
{Name: "a", Budget: 30, Runs: []classifiedRun{violatingRun(1, 4, 4), violatingRun(2, 6, 6)}},
|
||||
{Name: "b", Budget: 30, Runs: []classifiedRun{cleanRun(1, 30), cleanRun(2, 30)}},
|
||||
{Name: "c", Budget: 30, Runs: []classifiedRun{{Seed: 1, ExcludedBecause: reasonNonzeroExit}}},
|
||||
}, time.Unix(0, 0).UTC())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if len(result.Arms) != 3 {
|
||||
t.Fatalf("%d arms reported, want all 3", len(result.Arms))
|
||||
@@ -250,9 +255,12 @@ func TestAnalyse_ArmWithNoUsableRunsIsReportedButNotTested(t *testing.T) {
|
||||
// With a single testable arm there is nothing to compare against, and the tool
|
||||
// must say so instead of producing a statistic.
|
||||
func TestAnalyse_SingleArmHasNoTests(t *testing.T) {
|
||||
result := analyse([]arm{
|
||||
result, err := analyse([]arm{
|
||||
{Name: "a", Budget: 30, Runs: []classifiedRun{violatingRun(1, 4, 4)}},
|
||||
}, time.Unix(0, 0).UTC())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if result.LogRank != nil || len(result.Pairwise) != 0 {
|
||||
t.Errorf("log-rank %+v pairwise %v, want neither", result.LogRank, result.Pairwise)
|
||||
}
|
||||
@@ -297,6 +305,53 @@ func TestComparePairs_A12DirectionFollowsStepCounts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func writeCleanCampaign(t *testing.T, directory, name string, budget, steps, runs int) {
|
||||
t.Helper()
|
||||
seeds := make([]int, 0, runs)
|
||||
records := make([]map[string]any, 0, runs)
|
||||
for seed := 1; seed <= runs; seed++ {
|
||||
seeds = append(seeds, seed)
|
||||
records = append(records, map[string]any{
|
||||
"seed": seed, "exit_code": 0, "steps": steps, "actions": steps, "monotonic_millis": 60_000,
|
||||
})
|
||||
}
|
||||
writeCampaign(t, directory, map[string]any{"arm": name, "max_steps": budget, "seeds": seeds}, records)
|
||||
}
|
||||
|
||||
// Arms censored at different budgets are not on the same clock: every clean run
|
||||
// of the wider arm outranks every clean run of the narrower one whatever the
|
||||
// app did, so the rank-sum and the paired test reach a foregone conclusion the
|
||||
// log-rank in the same report contradicts. groupArms already refuses this
|
||||
// within one arm, and comparing across arms is the same hazard.
|
||||
func TestRun_RefusesToCompareArmsCensoredAtDifferentBudgets(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
wideSteps int
|
||||
arguments []string
|
||||
}{
|
||||
{name: "identical runs under different budgets", wideSteps: 100},
|
||||
{name: "each arm run to its own budget", wideSteps: 400},
|
||||
{name: "paired", wideSteps: 400, arguments: []string{"--paired"}},
|
||||
}
|
||||
for _, test := range cases {
|
||||
root := t.TempDir()
|
||||
wide := filepath.Join(root, "wide")
|
||||
narrow := filepath.Join(root, "narrow")
|
||||
writeCleanCampaign(t, wide, "wide", 400, test.wideSteps, 30)
|
||||
writeCleanCampaign(t, narrow, "narrow", 100, 100, 30)
|
||||
|
||||
err := run(append(test.arguments, wide, narrow), io.Discard, io.Discard)
|
||||
if err == nil {
|
||||
t.Fatalf("%s: arms censored at 400 and at 100 steps were compared without complaint", test.name)
|
||||
}
|
||||
for _, fragment := range []string{"wide", "400", "narrow", "100", "different budgets"} {
|
||||
if !strings.Contains(err.Error(), fragment) {
|
||||
t.Errorf("%s: error %q is missing %q", test.name, err, fragment)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func manyRuns(count, originStep int) []classifiedRun {
|
||||
runs := make([]classifiedRun, 0, count)
|
||||
for index := 0; index < count; index++ {
|
||||
|
||||
Reference in new issue
Block a user