fix(runner): abort on a candidate enumeration that refused

Recorded as candidates_failed before the run stops, so the trace says why.

Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
pj committed 2026-08-13 00:44:00 +05:30
1 parent 0094a7fc64
commit 7880e3e70f
4 files changed
+89 -24

No files matched your search

+10 -3
View File
@@ -11,13 +11,15 @@ import (
)
// authoredParityTreeJSON holds one target per authored action shape: a button to
// tap, a field to type into, and a scrollable container to scroll.
// tap, a field to type into, a scrollable container to scroll, and a disabled
// button, which is a target like any other here.
const authoredParityTreeJSON = `{
"attributes": {"bounds": "[0,0,400,800]"},
"children": [
{"attributes": {"resource-id": "Save", "text": "Save", "bounds": "[0,0,200,60]"}, "clickable": true, "enabled": true, "children": []},
{"attributes": {"resource-id": "Amount", "class": "EditText", "hintText": "Amount", "bounds": "[0,100,400,160]"}, "enabled": true, "children": []},
{"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []}
{"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []},
{"attributes": {"resource-id": "Off", "text": "Off", "bounds": "[0,700,200,760]"}, "clickable": true, "enabled": false, "children": []}
]
}`
@@ -36,6 +38,11 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) {
}{
{"tap an element", `const e = state.ax.find("id:Save"); return e ? [Tap({on: e})] : [];`},
{"tap a selector", `return [Tap({on: "id:Save"})];`},
// Attempting a disabled control is a legitimate thing for a UI fuzzer to
// do and is exactly where boundary defects live, so neither policy may
// quietly refuse to offer it.
{"tap a disabled element", `const e = state.ax.find("id:Off"); return e ? [Tap({on: e})] : [];`},
{"tap a disabled selector", `return [Tap({on: "id:Off"})];`},
{"double-tap an element", `const e = state.ax.find("id:Save"); return e ? [DoubleTap({on: e})] : [];`},
{"long-press an element", `const e = state.ax.find("id:Save"); return e ? [LongPress({on: e})] : [];`},
{"type into an element", `const e = state.ax.find("id:Amount"); return e ? [InputText({into: e, text: "42"})] : [];`},
@@ -60,7 +67,7 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) {
}
modelVerifier := loadAuthoredSpec(t, spec, tree)
candidates := modelVerifier.Candidates(verifier.LabelSourceVisibleText)
candidates := mustCandidates(t, modelVerifier, verifier.LabelSourceVisibleText)
if len(candidates) != 1 {
t.Fatalf("model was offered %d candidates, want the one authored action", len(candidates))
}
+23 -11
View File
@@ -110,8 +110,11 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
return verifier.Action{}, err
}
selection, call := s.selectViaLLM(ctx)
selection, call, err := s.selectViaLLM(ctx)
s.record(stepIndex, call)
if err != nil {
return verifier.Action{}, err
}
if call.Outcome != trace.LLMOutcomeSelected {
// Every other outcome skips the step; the next step re-observes and
// tries again. The record says which one it was.
@@ -129,12 +132,21 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
// The returned record is complete whichever way the selection ended: its
// Outcome is trace.LLMOutcomeSelected exactly when the returned selection is
// usable.
func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall) {
//
// The error is the spec refusing this policy rather than a step going nowhere:
// every later step would refuse identically, so the run stops instead of
// recording two hundred skipped steps.
func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall, error) {
call := trace.LLMCall{Timestamp: time.Now(), Model: s.model}
candidates := s.verifier.Candidates(s.labelSource)
candidates, err := s.verifier.Candidates(s.labelSource)
if err != nil {
call.Outcome = trace.LLMOutcomeCandidatesFailed
call.Error = err.Error()
return llmSelection{}, call, err
}
if len(candidates) == 0 {
call.Outcome = trace.LLMOutcomeNoCandidates
return llmSelection{}, call
return llmSelection{}, call, nil
}
call.Candidates = recordCandidates(candidates)
@@ -149,7 +161,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
s.logger.Warn("llm action selection failed", "err", err)
call.Outcome = trace.LLMOutcomeRequestFailed
call.Error = err.Error()
return llmSelection{}, call
return llmSelection{}, call, nil
}
call.ServedModel = response.Model
call.PromptTokens = response.Usage.PromptTokens
@@ -158,7 +170,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
if len(response.Choices) == 0 {
s.logger.Warn("llm returned no choices")
call.Outcome = trace.LLMOutcomeNoChoices
return llmSelection{}, call
return llmSelection{}, call, nil
}
call.RawResponse = response.Choices[0].Message.Content
@@ -167,7 +179,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
s.logger.Warn("llm output unusable", "err", err)
call.Outcome = trace.LLMOutcomeUnparsableResponse
call.Error = err.Error()
return llmSelection{}, call
return llmSelection{}, call, nil
}
call.Choice = output.Choice
call.EchoedAction = output.ChosenAction
@@ -176,7 +188,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
if output.Choice < 1 || output.Choice > len(candidates) {
s.logger.Warn("llm choice out of range", "choice", output.Choice, "candidates", len(candidates))
call.Outcome = trace.LLMOutcomeChoiceOutOfRange
return llmSelection{}, call
return llmSelection{}, call, nil
}
candidate := candidates[output.Choice-1]
// Strict skip: the echoed action must match the numbered entry, so a model
@@ -187,14 +199,14 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
s.logger.Warn("llm chosen_action mismatch; skipping",
"choice", output.Choice, "echoed", output.ChosenAction, "candidate", candidate.Description)
call.Outcome = trace.LLMOutcomeEchoMismatch
return llmSelection{}, call
return llmSelection{}, call, nil
}
action, err := s.actionForCandidate(candidate, output.Text)
if err != nil {
s.logger.Warn("building action from candidate failed", "choice", output.Choice, "err", err)
call.Outcome = trace.LLMOutcomeActionBuildFailed
call.Error = err.Error()
return llmSelection{}, call
return llmSelection{}, call, nil
}
call.Outcome = trace.LLMOutcomeSelected
return llmSelection{
@@ -202,7 +214,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
reasoning: output.Reasoning,
choice: output.Choice,
chosenAction: candidate.Description,
}, call
}, call, nil
}
// record stamps the step this selection belongs to and appends the record. A
+53 -10
View File
@@ -234,6 +234,17 @@ func lastCall(t *testing.T, source *llmSource) trace.LLMCall {
return calls[len(calls)-1]
}
// mustCandidates enumerates the model policy's list, failing the test on the
// refusal an authored multi-item sampler raises.
func mustCandidates(t *testing.T, v *verifier.Verifier, labelSource string) []verifier.ActionCandidate {
t.Helper()
candidates, err := v.Candidates(labelSource)
if err != nil {
t.Fatalf("Candidates: %v", err)
}
return candidates
}
func newLLMSource(t *testing.T, fake *fakeOpenRouter) (*llmSource, *verifier.Verifier) {
t.Helper()
return newLLMSourceWithSpec(t, fake, llmFixtureSpec)
@@ -428,7 +439,7 @@ func TestLLMSourceRecordsTheLabelsTheModelSaw(t *testing.T) {
source.labelSource = want.labelSource
pushSnapshotTree(t, verifierInstance, labelSplitTreeJSON)
tap := candidateByKind(t, verifierInstance.Candidates(want.labelSource), verifier.ActionKindTap)
tap := candidateByKind(t, mustCandidates(t, verifierInstance, want.labelSource), verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = tap.Description
if _, err := source.NextAction(context.Background(), 0); err != nil {
@@ -550,7 +561,7 @@ func TestLLMSourceDrivesExecutedActions(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSource(t, fake)
pushLLMSnapshot(t, verifierInstance)
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
// Step 1: the model picks the Tap on Submit by its number, echoing its
// description.
@@ -629,7 +640,7 @@ func TestLLMSourceAcceptsEchoWithWeightSuffix(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSource(t, fake)
pushLLMSnapshot(t, verifierInstance)
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
tap := candidateByKind(t, candidates, verifier.ActionKindTap)
fake.choice = tap.Index
@@ -664,7 +675,7 @@ func TestLLMSourceStrictSkipsOnEchoMismatch(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSource(t, fake)
pushLLMSnapshot(t, verifierInstance)
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
// A valid number, but the echoed action disagrees with that numbered entry:
// the model reasoned about one control and picked another's number.
@@ -701,7 +712,7 @@ func TestLLMSourceEchoGuardAdmitsARepeatedDescription(t *testing.T) {
pushSnapshotTree(t, verifierInstance, llmSharedLabelTreeJSON)
var repeated []verifier.ActionCandidate
for _, candidate := range verifierInstance.Candidates(verifier.LabelSourceVisibleText) {
for _, candidate := range mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) {
if candidate.Description == `Tap "Delete"` {
repeated = append(repeated, candidate)
}
@@ -763,7 +774,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSource(t, fake)
pushLLMSnapshot(t, verifierInstance)
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = `Tap "Something Else"`
if _, err := source.NextAction(context.Background(), stepIndex); !errors.Is(err, verifier.ErrNoAction) {
@@ -784,7 +795,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSource(t, fake)
pushLLMSnapshot(t, verifierInstance)
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = tap.Description
if _, err := source.NextAction(context.Background(), stepIndex); err != nil {
@@ -832,7 +843,7 @@ func TestLLMCallRecordsCandidateListAsShown(t *testing.T) {
source, verifierInstance := newLLMSource(t, fake)
source.instructions = "hunt for double submits"
pushLLMSnapshot(t, verifierInstance)
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = tap.Description
if _, err := source.NextAction(context.Background(), 1); err != nil {
@@ -920,7 +931,7 @@ func TestLLMCallFileRecordsUsageLatencyAndScreenshot(t *testing.T) {
source.recorder = writer
pushLLMSnapshotAtStep(t, verifierInstance, 4)
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = tap.Description
if _, err := source.NextAction(context.Background(), 4); err != nil {
@@ -960,7 +971,7 @@ func TestLLMCallScreenshotNamesObservedStep(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSource(t, fake)
pushLLMSnapshotAtStep(t, verifierInstance, 4)
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = tap.Description
if _, err := source.NextAction(context.Background(), 6); err != nil {
@@ -1030,3 +1041,35 @@ func tinyPNG(t *testing.T) []byte {
}
return buffer.Bytes()
}
// llmSamplerFixtureSpec drives the model policy over an authored leaf that
// samples one of three targets, which is the shape the seeded picker draws from
// and the model policy cannot.
const llmSamplerFixtureSpec = `
import { actions, from, llm, Tap, always } from "@sanderling/spec";
globalThis.properties = { ok: always(() => true) };
const targets = from(["id:Submit", "id:Name"]);
globalThis.actions = actions(() => [Tap({ on: targets.generate() })]);
globalThis.generator = llm({ model: "test/model" });
`
// TestLLMSourceRefusesAMultiItemAuthoredSampler: the step must fail the run, not
// skip. A skip would leave the model quietly fuzzing a spec whose authored
// targets it can never reach past the first, which is the comparison the seeded
// arm is measured against.
func TestLLMSourceRefusesAMultiItemAuthoredSampler(t *testing.T) {
fake := newFakeOpenRouter(t)
source, verifierInstance := newLLMSourceWithSpec(t, fake, llmSamplerFixtureSpec)
pushLLMSnapshot(t, verifierInstance)
_, err := source.NextAction(context.Background(), 1)
if err == nil || errors.Is(err, verifier.ErrNoAction) {
t.Fatalf("NextAction err = %v, want the run to stop on a sampler the model cannot draw", err)
}
if !strings.Contains(err.Error(), "targets.generate()") {
t.Errorf("error does not name the offending leaf: %v", err)
}
if outcome := lastCall(t, source).Outcome; outcome != trace.LLMOutcomeCandidatesFailed {
t.Errorf("recorded outcome = %q, want %q", outcome, trace.LLMOutcomeCandidatesFailed)
}
}
+3
View File
@@ -35,6 +35,9 @@ const (
// LLMOutcomeNoCandidates: the action tree yielded nothing on this screen, so
// no call was made.
LLMOutcomeNoCandidates = "no_candidates"
// LLMOutcomeCandidatesFailed: the action tree cannot be enumerated for this
// policy at all (an authored leaf samples), which aborts the run.
LLMOutcomeCandidatesFailed = "candidates_failed"
// LLMOutcomeRequestFailed: the provider call failed (transport, timeout,
// non-2xx).
LLMOutcomeRequestFailed = "request_failed"