mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-04 12:07:09 +00:00
fix(analyze): censor a clean run at the steps it ran, and refuse mismatched budgets
A run stops at whichever comes first, the step budget or --duration, so a clean run that reached the wall clock exited with fewer steps than the budget and was still credited with the whole of it. The model arm pays a network call and a screenshot per step, so it reaches the wall sooner and was handed exposure it never had. Nothing checked that two arms shared a budget either. Thirty identical clean runs under budgets of 400 and 100 read a12 0.000 and p 1.685e-14 from the rank-sum while the log-rank in the same report read p 1.0000. groupArms already refused this within one arm. The claims the old convention left in comments and report lines are corrected rather than left standing beside the new behaviour.
This commit is contained in:
1 parent
ff6c66a74b
commit
1da0c3e118
10 files changed
+141
-41
No files matched your search
@@ -31,18 +31,32 @@ func TestClassify_FailedAndTimedOutRunsAreMissingDataNotCensored(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestClassify_CleanRunIsCensoredAtTheBudget(t *testing.T) {
|
||||
item := classify(runRecord{Seed: 4, Steps: 50, DurationMillis: 1000}, 50)
|
||||
if item.ExcludedBecause != "" {
|
||||
t.Fatalf("excluded because %q", item.ExcludedBecause)
|
||||
// A run also ends when the campaign's wall clock does, so a clean run can stop
|
||||
// well short of the budget. Censoring it at the budget would credit it with
|
||||
// steps it never ran, and a slower arm loses fewer steps in the same wall clock
|
||||
// than a fast one, so the credit does not cancel between arms.
|
||||
func TestClassify_CleanRunIsCensoredAtTheStepsItRan(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
steps int
|
||||
budget int
|
||||
censored float64
|
||||
}{
|
||||
{"stopped by the wall clock short of the budget", 12, 400, 12},
|
||||
{"ran the whole budget", 400, 400, 400},
|
||||
{"recorded more steps than the manifest budget", 420, 400, 400},
|
||||
}
|
||||
if item.Violated {
|
||||
t.Error("clean run marked as violated")
|
||||
}
|
||||
current := arm{Budget: 50, Runs: []classifiedRun{item}}
|
||||
observations := current.observations()
|
||||
if len(observations) != 1 || observations[0].Event || observations[0].Steps != 50 {
|
||||
t.Errorf("observations %+v, want one censored observation at 50", observations)
|
||||
for _, test := range cases {
|
||||
item := classify(runRecord{Seed: 4, Steps: test.steps, DurationMillis: 1000}, test.budget)
|
||||
if item.ExcludedBecause != "" || item.Violated {
|
||||
t.Fatalf("%s: run %+v, want a usable clean run", test.name, item)
|
||||
}
|
||||
current := arm{Budget: test.budget, Runs: []classifiedRun{item}}
|
||||
observations := current.observations()
|
||||
if len(observations) != 1 || observations[0].Event || observations[0].Steps != test.censored {
|
||||
t.Errorf("%s: observations %+v, want one censored observation at %v",
|
||||
test.name, observations, test.censored)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user