From 7880e3e70ff11d9aade1b25c6ebcf52be3b6735d Mon Sep 17 00:00:00 2001 From: PJ Date: Thu, 13 Aug 2026 00:44:00 +0530 Subject: [PATCH] fix(runner): abort on a candidate enumeration that refused Recorded as candidates_failed before the run stops, so the trace says why. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX --- .../runner/authored_action_parity_test.go | 13 +++- internal/runner/llm_source.go | 34 ++++++---- internal/runner/llm_source_test.go | 63 ++++++++++++++++--- internal/trace/llm_call.go | 3 + 4 files changed, 89 insertions(+), 24 deletions(-) diff --git a/internal/runner/authored_action_parity_test.go b/internal/runner/authored_action_parity_test.go index 3435463..f5dd162 100644 --- a/internal/runner/authored_action_parity_test.go +++ b/internal/runner/authored_action_parity_test.go @@ -11,13 +11,15 @@ import ( ) // authoredParityTreeJSON holds one target per authored action shape: a button to -// tap, a field to type into, and a scrollable container to scroll. +// tap, a field to type into, a scrollable container to scroll, and a disabled +// button, which is a target like any other here. const authoredParityTreeJSON = `{ "attributes": {"bounds": "[0,0,400,800]"}, "children": [ {"attributes": {"resource-id": "Save", "text": "Save", "bounds": "[0,0,200,60]"}, "clickable": true, "enabled": true, "children": []}, {"attributes": {"resource-id": "Amount", "class": "EditText", "hintText": "Amount", "bounds": "[0,100,400,160]"}, "enabled": true, "children": []}, - {"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []} + {"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []}, + {"attributes": {"resource-id": "Off", "text": "Off", "bounds": "[0,700,200,760]"}, "clickable": true, "enabled": false, "children": []} ] }` @@ -36,6 +38,11 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) { }{ {"tap an element", `const e = state.ax.find("id:Save"); return e ? [Tap({on: e})] : [];`}, {"tap a selector", `return [Tap({on: "id:Save"})];`}, + // Attempting a disabled control is a legitimate thing for a UI fuzzer to + // do and is exactly where boundary defects live, so neither policy may + // quietly refuse to offer it. + {"tap a disabled element", `const e = state.ax.find("id:Off"); return e ? [Tap({on: e})] : [];`}, + {"tap a disabled selector", `return [Tap({on: "id:Off"})];`}, {"double-tap an element", `const e = state.ax.find("id:Save"); return e ? [DoubleTap({on: e})] : [];`}, {"long-press an element", `const e = state.ax.find("id:Save"); return e ? [LongPress({on: e})] : [];`}, {"type into an element", `const e = state.ax.find("id:Amount"); return e ? [InputText({into: e, text: "42"})] : [];`}, @@ -60,7 +67,7 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) { } modelVerifier := loadAuthoredSpec(t, spec, tree) - candidates := modelVerifier.Candidates(verifier.LabelSourceVisibleText) + candidates := mustCandidates(t, modelVerifier, verifier.LabelSourceVisibleText) if len(candidates) != 1 { t.Fatalf("model was offered %d candidates, want the one authored action", len(candidates)) } diff --git a/internal/runner/llm_source.go b/internal/runner/llm_source.go index 69df7cf..65f7625 100644 --- a/internal/runner/llm_source.go +++ b/internal/runner/llm_source.go @@ -110,8 +110,11 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act return verifier.Action{}, err } - selection, call := s.selectViaLLM(ctx) + selection, call, err := s.selectViaLLM(ctx) s.record(stepIndex, call) + if err != nil { + return verifier.Action{}, err + } if call.Outcome != trace.LLMOutcomeSelected { // Every other outcome skips the step; the next step re-observes and // tries again. The record says which one it was. @@ -129,12 +132,21 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act // The returned record is complete whichever way the selection ended: its // Outcome is trace.LLMOutcomeSelected exactly when the returned selection is // usable. -func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall) { +// +// The error is the spec refusing this policy rather than a step going nowhere: +// every later step would refuse identically, so the run stops instead of +// recording two hundred skipped steps. +func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall, error) { call := trace.LLMCall{Timestamp: time.Now(), Model: s.model} - candidates := s.verifier.Candidates(s.labelSource) + candidates, err := s.verifier.Candidates(s.labelSource) + if err != nil { + call.Outcome = trace.LLMOutcomeCandidatesFailed + call.Error = err.Error() + return llmSelection{}, call, err + } if len(candidates) == 0 { call.Outcome = trace.LLMOutcomeNoCandidates - return llmSelection{}, call + return llmSelection{}, call, nil } call.Candidates = recordCandidates(candidates) @@ -149,7 +161,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa s.logger.Warn("llm action selection failed", "err", err) call.Outcome = trace.LLMOutcomeRequestFailed call.Error = err.Error() - return llmSelection{}, call + return llmSelection{}, call, nil } call.ServedModel = response.Model call.PromptTokens = response.Usage.PromptTokens @@ -158,7 +170,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa if len(response.Choices) == 0 { s.logger.Warn("llm returned no choices") call.Outcome = trace.LLMOutcomeNoChoices - return llmSelection{}, call + return llmSelection{}, call, nil } call.RawResponse = response.Choices[0].Message.Content @@ -167,7 +179,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa s.logger.Warn("llm output unusable", "err", err) call.Outcome = trace.LLMOutcomeUnparsableResponse call.Error = err.Error() - return llmSelection{}, call + return llmSelection{}, call, nil } call.Choice = output.Choice call.EchoedAction = output.ChosenAction @@ -176,7 +188,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa if output.Choice < 1 || output.Choice > len(candidates) { s.logger.Warn("llm choice out of range", "choice", output.Choice, "candidates", len(candidates)) call.Outcome = trace.LLMOutcomeChoiceOutOfRange - return llmSelection{}, call + return llmSelection{}, call, nil } candidate := candidates[output.Choice-1] // Strict skip: the echoed action must match the numbered entry, so a model @@ -187,14 +199,14 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa s.logger.Warn("llm chosen_action mismatch; skipping", "choice", output.Choice, "echoed", output.ChosenAction, "candidate", candidate.Description) call.Outcome = trace.LLMOutcomeEchoMismatch - return llmSelection{}, call + return llmSelection{}, call, nil } action, err := s.actionForCandidate(candidate, output.Text) if err != nil { s.logger.Warn("building action from candidate failed", "choice", output.Choice, "err", err) call.Outcome = trace.LLMOutcomeActionBuildFailed call.Error = err.Error() - return llmSelection{}, call + return llmSelection{}, call, nil } call.Outcome = trace.LLMOutcomeSelected return llmSelection{ @@ -202,7 +214,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa reasoning: output.Reasoning, choice: output.Choice, chosenAction: candidate.Description, - }, call + }, call, nil } // record stamps the step this selection belongs to and appends the record. A diff --git a/internal/runner/llm_source_test.go b/internal/runner/llm_source_test.go index 8e0d31e..1bd2a86 100644 --- a/internal/runner/llm_source_test.go +++ b/internal/runner/llm_source_test.go @@ -234,6 +234,17 @@ func lastCall(t *testing.T, source *llmSource) trace.LLMCall { return calls[len(calls)-1] } +// mustCandidates enumerates the model policy's list, failing the test on the +// refusal an authored multi-item sampler raises. +func mustCandidates(t *testing.T, v *verifier.Verifier, labelSource string) []verifier.ActionCandidate { + t.Helper() + candidates, err := v.Candidates(labelSource) + if err != nil { + t.Fatalf("Candidates: %v", err) + } + return candidates +} + func newLLMSource(t *testing.T, fake *fakeOpenRouter) (*llmSource, *verifier.Verifier) { t.Helper() return newLLMSourceWithSpec(t, fake, llmFixtureSpec) @@ -428,7 +439,7 @@ func TestLLMSourceRecordsTheLabelsTheModelSaw(t *testing.T) { source.labelSource = want.labelSource pushSnapshotTree(t, verifierInstance, labelSplitTreeJSON) - tap := candidateByKind(t, verifierInstance.Candidates(want.labelSource), verifier.ActionKindTap) + tap := candidateByKind(t, mustCandidates(t, verifierInstance, want.labelSource), verifier.ActionKindTap) fake.choice = tap.Index fake.chosenAction = tap.Description if _, err := source.NextAction(context.Background(), 0); err != nil { @@ -550,7 +561,7 @@ func TestLLMSourceDrivesExecutedActions(t *testing.T) { fake := newFakeOpenRouter(t) source, verifierInstance := newLLMSource(t, fake) pushLLMSnapshot(t, verifierInstance) - candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText) + candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) // Step 1: the model picks the Tap on Submit by its number, echoing its // description. @@ -629,7 +640,7 @@ func TestLLMSourceAcceptsEchoWithWeightSuffix(t *testing.T) { fake := newFakeOpenRouter(t) source, verifierInstance := newLLMSource(t, fake) pushLLMSnapshot(t, verifierInstance) - candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText) + candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) tap := candidateByKind(t, candidates, verifier.ActionKindTap) fake.choice = tap.Index @@ -664,7 +675,7 @@ func TestLLMSourceStrictSkipsOnEchoMismatch(t *testing.T) { fake := newFakeOpenRouter(t) source, verifierInstance := newLLMSource(t, fake) pushLLMSnapshot(t, verifierInstance) - candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText) + candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) // A valid number, but the echoed action disagrees with that numbered entry: // the model reasoned about one control and picked another's number. @@ -701,7 +712,7 @@ func TestLLMSourceEchoGuardAdmitsARepeatedDescription(t *testing.T) { pushSnapshotTree(t, verifierInstance, llmSharedLabelTreeJSON) var repeated []verifier.ActionCandidate - for _, candidate := range verifierInstance.Candidates(verifier.LabelSourceVisibleText) { + for _, candidate := range mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) { if candidate.Description == `Tap "Delete"` { repeated = append(repeated, candidate) } @@ -763,7 +774,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) { fake := newFakeOpenRouter(t) source, verifierInstance := newLLMSource(t, fake) pushLLMSnapshot(t, verifierInstance) - tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap) + tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap) fake.choice = tap.Index fake.chosenAction = `Tap "Something Else"` if _, err := source.NextAction(context.Background(), stepIndex); !errors.Is(err, verifier.ErrNoAction) { @@ -784,7 +795,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) { fake := newFakeOpenRouter(t) source, verifierInstance := newLLMSource(t, fake) pushLLMSnapshot(t, verifierInstance) - tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap) + tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap) fake.choice = tap.Index fake.chosenAction = tap.Description if _, err := source.NextAction(context.Background(), stepIndex); err != nil { @@ -832,7 +843,7 @@ func TestLLMCallRecordsCandidateListAsShown(t *testing.T) { source, verifierInstance := newLLMSource(t, fake) source.instructions = "hunt for double submits" pushLLMSnapshot(t, verifierInstance) - tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap) + tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap) fake.choice = tap.Index fake.chosenAction = tap.Description if _, err := source.NextAction(context.Background(), 1); err != nil { @@ -920,7 +931,7 @@ func TestLLMCallFileRecordsUsageLatencyAndScreenshot(t *testing.T) { source.recorder = writer pushLLMSnapshotAtStep(t, verifierInstance, 4) - tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap) + tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap) fake.choice = tap.Index fake.chosenAction = tap.Description if _, err := source.NextAction(context.Background(), 4); err != nil { @@ -960,7 +971,7 @@ func TestLLMCallScreenshotNamesObservedStep(t *testing.T) { fake := newFakeOpenRouter(t) source, verifierInstance := newLLMSource(t, fake) pushLLMSnapshotAtStep(t, verifierInstance, 4) - tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap) + tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap) fake.choice = tap.Index fake.chosenAction = tap.Description if _, err := source.NextAction(context.Background(), 6); err != nil { @@ -1030,3 +1041,35 @@ func tinyPNG(t *testing.T) []byte { } return buffer.Bytes() } + +// llmSamplerFixtureSpec drives the model policy over an authored leaf that +// samples one of three targets, which is the shape the seeded picker draws from +// and the model policy cannot. +const llmSamplerFixtureSpec = ` +import { actions, from, llm, Tap, always } from "@sanderling/spec"; +globalThis.properties = { ok: always(() => true) }; +const targets = from(["id:Submit", "id:Name"]); +globalThis.actions = actions(() => [Tap({ on: targets.generate() })]); +globalThis.generator = llm({ model: "test/model" }); +` + +// TestLLMSourceRefusesAMultiItemAuthoredSampler: the step must fail the run, not +// skip. A skip would leave the model quietly fuzzing a spec whose authored +// targets it can never reach past the first, which is the comparison the seeded +// arm is measured against. +func TestLLMSourceRefusesAMultiItemAuthoredSampler(t *testing.T) { + fake := newFakeOpenRouter(t) + source, verifierInstance := newLLMSourceWithSpec(t, fake, llmSamplerFixtureSpec) + pushLLMSnapshot(t, verifierInstance) + + _, err := source.NextAction(context.Background(), 1) + if err == nil || errors.Is(err, verifier.ErrNoAction) { + t.Fatalf("NextAction err = %v, want the run to stop on a sampler the model cannot draw", err) + } + if !strings.Contains(err.Error(), "targets.generate()") { + t.Errorf("error does not name the offending leaf: %v", err) + } + if outcome := lastCall(t, source).Outcome; outcome != trace.LLMOutcomeCandidatesFailed { + t.Errorf("recorded outcome = %q, want %q", outcome, trace.LLMOutcomeCandidatesFailed) + } +} diff --git a/internal/trace/llm_call.go b/internal/trace/llm_call.go index 4375883..fefbca1 100644 --- a/internal/trace/llm_call.go +++ b/internal/trace/llm_call.go @@ -35,6 +35,9 @@ const ( // LLMOutcomeNoCandidates: the action tree yielded nothing on this screen, so // no call was made. LLMOutcomeNoCandidates = "no_candidates" + // LLMOutcomeCandidatesFailed: the action tree cannot be enumerated for this + // policy at all (an authored leaf samples), which aborts the run. + LLMOutcomeCandidatesFailed = "candidates_failed" // LLMOutcomeRequestFailed: the provider call failed (transport, timeout, // non-2xx). LLMOutcomeRequestFailed = "request_failed"