mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
fix(analyze): a run that never explored is excluded, not counted as evidence
The campaign has written precondition_failures since it learned to count them and neither reader declared it, so a run whose scope guard could not bring the app back entered the survival analysis with its whole budget as exposure the app survived. A strict majority of steps is the line: below it the failures were transient and the run still explored.
This commit is contained in:
1 parent
52625d95a7
commit
3e4633f588
2 files changed
+67
-5
No files matched your search
@@ -65,17 +65,30 @@ type runRecord struct {
|
|||||||
// field, and reading that silence as none would let a denominator of unknown
|
// field, and reading that silence as none would let a denominator of unknown
|
||||||
// provenance pass for one that excludes the setup's login.
|
// provenance pass for one that excludes the setup's login.
|
||||||
UnattributedActions *int `json:"unattributed_actions"`
|
UnattributedActions *int `json:"unattributed_actions"`
|
||||||
|
// PreconditionFailures is how many of the run's steps never had the app under
|
||||||
|
// test in front of them: the startup gate's verdict and every later step the
|
||||||
|
// scope guard could not bring the app back for. The campaign omits the field
|
||||||
|
// when it is zero, so absence and zero mean the same thing here.
|
||||||
|
PreconditionFailures int `json:"precondition_failures"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// Exclusion reasons. A run that failed or timed out is missing data, not a
|
// Exclusion reasons. A run that failed or timed out is missing data, not a
|
||||||
// censored observation: it broke off, so its step count is not exposure the app
|
// censored observation: it broke off, so its step count is not exposure the app
|
||||||
// survived and counting it as one would bias the survival estimate downward.
|
// survived and counting it as one would bias the survival estimate downward.
|
||||||
const (
|
const (
|
||||||
reasonLaunchError = "launch error"
|
reasonLaunchError = "launch error"
|
||||||
reasonTimedOut = "timed out"
|
reasonTimedOut = "timed out"
|
||||||
reasonNonzeroExit = "nonzero exit"
|
reasonNonzeroExit = "nonzero exit"
|
||||||
reasonTraceError = "unreadable trace"
|
reasonTraceError = "unreadable trace"
|
||||||
reasonMalformedStep = "violation step outside the budget"
|
// reasonPreconditionFailures is the run that exited cleanly having spent its
|
||||||
|
// budget failing preconditions. The scope guard records one on every step it
|
||||||
|
// could not bring the app back for and lets the run finish, so nothing else
|
||||||
|
// here separates it from a run that explored the whole budget and found
|
||||||
|
// nothing. A majority is the line: below it the failures are transient and
|
||||||
|
// the run still explored, above it the step count the analysis would censor
|
||||||
|
// at is mostly steps the app was never there for.
|
||||||
|
reasonPreconditionFailures = "precondition failures"
|
||||||
|
reasonMalformedStep = "violation step outside the budget"
|
||||||
)
|
)
|
||||||
|
|
||||||
type classifiedRun struct {
|
type classifiedRun struct {
|
||||||
@@ -201,6 +214,9 @@ func classify(record runRecord, budget int) classifiedRun {
|
|||||||
case record.TraceError != "":
|
case record.TraceError != "":
|
||||||
item.ExcludedBecause = reasonTraceError
|
item.ExcludedBecause = reasonTraceError
|
||||||
return item
|
return item
|
||||||
|
case record.PreconditionFailures*2 > record.Steps:
|
||||||
|
item.ExcludedBecause = reasonPreconditionFailures
|
||||||
|
return item
|
||||||
}
|
}
|
||||||
if record.FirstViolationOriginStep == nil {
|
if record.FirstViolationOriginStep == nil {
|
||||||
if len(record.ViolatedProperties) > 0 {
|
if len(record.ViolatedProperties) > 0 {
|
||||||
|
|||||||
@@ -20,6 +20,8 @@ func TestClassify_FailedAndTimedOutRunsAreMissingDataNotCensored(t *testing.T) {
|
|||||||
{"timed out", runRecord{TimedOut: true, ExitCode: -1}, reasonTimedOut},
|
{"timed out", runRecord{TimedOut: true, ExitCode: -1}, reasonTimedOut},
|
||||||
{"nonzero exit", runRecord{ExitCode: 3}, reasonNonzeroExit},
|
{"nonzero exit", runRecord{ExitCode: 3}, reasonNonzeroExit},
|
||||||
{"unreadable trace", runRecord{TraceError: "no run directory with meta.json"}, reasonTraceError},
|
{"unreadable trace", runRecord{TraceError: "no run directory with meta.json"}, reasonTraceError},
|
||||||
|
{"never started", runRecord{PreconditionFailures: 1}, reasonPreconditionFailures},
|
||||||
|
{"most steps never saw the app", runRecord{Steps: 40, PreconditionFailures: 21}, reasonPreconditionFailures},
|
||||||
{"violation at step zero", runRecord{FirstViolationOriginStep: stepPointer(0)}, reasonMalformedStep},
|
{"violation at step zero", runRecord{FirstViolationOriginStep: stepPointer(0)}, reasonMalformedStep},
|
||||||
{"violation without a step", runRecord{ViolatedProperties: []string{"cartTotal"}}, reasonMalformedStep},
|
{"violation without a step", runRecord{ViolatedProperties: []string{"cartTotal"}}, reasonMalformedStep},
|
||||||
}
|
}
|
||||||
@@ -148,6 +150,50 @@ func TestGroupArms_PoolsDirectoriesSharingAnArmAndReportsMissingSeeds(t *testing
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The scope guard records a precondition failure on every step it could not
|
||||||
|
// bring the app back for and still lets the run exit cleanly, so a run that
|
||||||
|
// spent its budget outside the app under test arrives at the analysis looking
|
||||||
|
// like one that explored the whole budget and found nothing.
|
||||||
|
func TestGroupArms_ExcludesRunsThatSpentTheBudgetFailingPreconditions(t *testing.T) {
|
||||||
|
root := t.TempDir()
|
||||||
|
directory := filepath.Join(root, "escaped")
|
||||||
|
writeCampaign(t, directory, map[string]any{"arm": "seeded", "max_steps": 40, "seeds": []int{1, 2, 3}},
|
||||||
|
[]map[string]any{
|
||||||
|
{"seed": 1, "exit_code": 0, "steps": 40, "actions": 40, "precondition_failures": 36},
|
||||||
|
{"seed": 2, "exit_code": 0, "steps": 40, "actions": 40, "precondition_failures": 20},
|
||||||
|
{"seed": 3, "exit_code": 0, "steps": 40, "actions": 40},
|
||||||
|
})
|
||||||
|
|
||||||
|
arms, err := groupArms([]string{directory})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(arms) != 1 {
|
||||||
|
t.Fatalf("%d arms, want 1", len(arms))
|
||||||
|
}
|
||||||
|
excludedBySeed := map[int64]string{}
|
||||||
|
for _, item := range arms[0].Runs {
|
||||||
|
excludedBySeed[item.Seed] = item.ExcludedBecause
|
||||||
|
}
|
||||||
|
if excludedBySeed[1] != reasonPreconditionFailures {
|
||||||
|
t.Errorf("the run that spent its budget failing preconditions is excluded as %q, want %q",
|
||||||
|
excludedBySeed[1], reasonPreconditionFailures)
|
||||||
|
}
|
||||||
|
if excludedBySeed[2] != "" || excludedBySeed[3] != "" {
|
||||||
|
t.Errorf("runs that explored are excluded as %q and %q, want both kept",
|
||||||
|
excludedBySeed[2], excludedBySeed[3])
|
||||||
|
}
|
||||||
|
observations := arms[0].observations()
|
||||||
|
if len(observations) != 2 {
|
||||||
|
t.Fatalf("%d observations, want only the 2 runs that explored", len(observations))
|
||||||
|
}
|
||||||
|
for _, item := range observations {
|
||||||
|
if item.Event || item.Steps != 40 {
|
||||||
|
t.Errorf("observation %+v, want a censored observation at 40", item)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestGroupArms_RejectsDisagreeingStepBudgets(t *testing.T) {
|
func TestGroupArms_RejectsDisagreeingStepBudgets(t *testing.T) {
|
||||||
root := t.TempDir()
|
root := t.TempDir()
|
||||||
writeCampaign(t, filepath.Join(root, "a"), map[string]any{"arm": "seeded", "max_steps": 40, "seeds": []int{1}},
|
writeCampaign(t, filepath.Join(root, "a"), map[string]any{"arm": "seeded", "max_steps": 40, "seeds": []int{1}},
|
||||||
|
|||||||
Reference in new issue
Block a user