mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 11:07:10 +00:00
fix(analyze): a run that never explored is excluded, not counted as evidence
The campaign has written precondition_failures since it learned to count them and neither reader declared it, so a run whose scope guard could not bring the app back entered the survival analysis with its whole budget as exposure the app survived. A strict majority of steps is the line: below it the failures were transient and the run still explored.
This commit is contained in:
1 parent
52625d95a7
commit
3e4633f588
2 files changed
+62
No files matched your search
@@ -65,6 +65,11 @@ type runRecord struct {
|
||||
// field, and reading that silence as none would let a denominator of unknown
|
||||
// provenance pass for one that excludes the setup's login.
|
||||
UnattributedActions *int `json:"unattributed_actions"`
|
||||
// PreconditionFailures is how many of the run's steps never had the app under
|
||||
// test in front of them: the startup gate's verdict and every later step the
|
||||
// scope guard could not bring the app back for. The campaign omits the field
|
||||
// when it is zero, so absence and zero mean the same thing here.
|
||||
PreconditionFailures int `json:"precondition_failures"`
|
||||
}
|
||||
|
||||
// Exclusion reasons. A run that failed or timed out is missing data, not a
|
||||
@@ -75,6 +80,14 @@ const (
|
||||
reasonTimedOut = "timed out"
|
||||
reasonNonzeroExit = "nonzero exit"
|
||||
reasonTraceError = "unreadable trace"
|
||||
// reasonPreconditionFailures is the run that exited cleanly having spent its
|
||||
// budget failing preconditions. The scope guard records one on every step it
|
||||
// could not bring the app back for and lets the run finish, so nothing else
|
||||
// here separates it from a run that explored the whole budget and found
|
||||
// nothing. A majority is the line: below it the failures are transient and
|
||||
// the run still explored, above it the step count the analysis would censor
|
||||
// at is mostly steps the app was never there for.
|
||||
reasonPreconditionFailures = "precondition failures"
|
||||
reasonMalformedStep = "violation step outside the budget"
|
||||
)
|
||||
|
||||
@@ -201,6 +214,9 @@ func classify(record runRecord, budget int) classifiedRun {
|
||||
case record.TraceError != "":
|
||||
item.ExcludedBecause = reasonTraceError
|
||||
return item
|
||||
case record.PreconditionFailures*2 > record.Steps:
|
||||
item.ExcludedBecause = reasonPreconditionFailures
|
||||
return item
|
||||
}
|
||||
if record.FirstViolationOriginStep == nil {
|
||||
if len(record.ViolatedProperties) > 0 {
|
||||
|
||||
@@ -20,6 +20,8 @@ func TestClassify_FailedAndTimedOutRunsAreMissingDataNotCensored(t *testing.T) {
|
||||
{"timed out", runRecord{TimedOut: true, ExitCode: -1}, reasonTimedOut},
|
||||
{"nonzero exit", runRecord{ExitCode: 3}, reasonNonzeroExit},
|
||||
{"unreadable trace", runRecord{TraceError: "no run directory with meta.json"}, reasonTraceError},
|
||||
{"never started", runRecord{PreconditionFailures: 1}, reasonPreconditionFailures},
|
||||
{"most steps never saw the app", runRecord{Steps: 40, PreconditionFailures: 21}, reasonPreconditionFailures},
|
||||
{"violation at step zero", runRecord{FirstViolationOriginStep: stepPointer(0)}, reasonMalformedStep},
|
||||
{"violation without a step", runRecord{ViolatedProperties: []string{"cartTotal"}}, reasonMalformedStep},
|
||||
}
|
||||
@@ -148,6 +150,50 @@ func TestGroupArms_PoolsDirectoriesSharingAnArmAndReportsMissingSeeds(t *testing
|
||||
}
|
||||
}
|
||||
|
||||
// The scope guard records a precondition failure on every step it could not
|
||||
// bring the app back for and still lets the run exit cleanly, so a run that
|
||||
// spent its budget outside the app under test arrives at the analysis looking
|
||||
// like one that explored the whole budget and found nothing.
|
||||
func TestGroupArms_ExcludesRunsThatSpentTheBudgetFailingPreconditions(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
directory := filepath.Join(root, "escaped")
|
||||
writeCampaign(t, directory, map[string]any{"arm": "seeded", "max_steps": 40, "seeds": []int{1, 2, 3}},
|
||||
[]map[string]any{
|
||||
{"seed": 1, "exit_code": 0, "steps": 40, "actions": 40, "precondition_failures": 36},
|
||||
{"seed": 2, "exit_code": 0, "steps": 40, "actions": 40, "precondition_failures": 20},
|
||||
{"seed": 3, "exit_code": 0, "steps": 40, "actions": 40},
|
||||
})
|
||||
|
||||
arms, err := groupArms([]string{directory})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(arms) != 1 {
|
||||
t.Fatalf("%d arms, want 1", len(arms))
|
||||
}
|
||||
excludedBySeed := map[int64]string{}
|
||||
for _, item := range arms[0].Runs {
|
||||
excludedBySeed[item.Seed] = item.ExcludedBecause
|
||||
}
|
||||
if excludedBySeed[1] != reasonPreconditionFailures {
|
||||
t.Errorf("the run that spent its budget failing preconditions is excluded as %q, want %q",
|
||||
excludedBySeed[1], reasonPreconditionFailures)
|
||||
}
|
||||
if excludedBySeed[2] != "" || excludedBySeed[3] != "" {
|
||||
t.Errorf("runs that explored are excluded as %q and %q, want both kept",
|
||||
excludedBySeed[2], excludedBySeed[3])
|
||||
}
|
||||
observations := arms[0].observations()
|
||||
if len(observations) != 2 {
|
||||
t.Fatalf("%d observations, want only the 2 runs that explored", len(observations))
|
||||
}
|
||||
for _, item := range observations {
|
||||
if item.Event || item.Steps != 40 {
|
||||
t.Errorf("observation %+v, want a censored observation at 40", item)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGroupArms_RejectsDisagreeingStepBudgets(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
writeCampaign(t, filepath.Join(root, "a"), map[string]any{"arm": "seeded", "max_steps": 40, "seeds": []int{1}},
|
||||
|
||||
Reference in new issue
Block a user