mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
fix(runner): abort on a candidate enumeration that refused
Recorded as candidates_failed before the run stops, so the trace says why. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
0094a7fc64
commit
7880e3e70f
4 files changed
+89
-24
No files matched your search
@@ -11,13 +11,15 @@ import (
|
||||
)
|
||||
|
||||
// authoredParityTreeJSON holds one target per authored action shape: a button to
|
||||
// tap, a field to type into, and a scrollable container to scroll.
|
||||
// tap, a field to type into, a scrollable container to scroll, and a disabled
|
||||
// button, which is a target like any other here.
|
||||
const authoredParityTreeJSON = `{
|
||||
"attributes": {"bounds": "[0,0,400,800]"},
|
||||
"children": [
|
||||
{"attributes": {"resource-id": "Save", "text": "Save", "bounds": "[0,0,200,60]"}, "clickable": true, "enabled": true, "children": []},
|
||||
{"attributes": {"resource-id": "Amount", "class": "EditText", "hintText": "Amount", "bounds": "[0,100,400,160]"}, "enabled": true, "children": []},
|
||||
{"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []}
|
||||
{"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []},
|
||||
{"attributes": {"resource-id": "Off", "text": "Off", "bounds": "[0,700,200,760]"}, "clickable": true, "enabled": false, "children": []}
|
||||
]
|
||||
}`
|
||||
|
||||
@@ -36,6 +38,11 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) {
|
||||
}{
|
||||
{"tap an element", `const e = state.ax.find("id:Save"); return e ? [Tap({on: e})] : [];`},
|
||||
{"tap a selector", `return [Tap({on: "id:Save"})];`},
|
||||
// Attempting a disabled control is a legitimate thing for a UI fuzzer to
|
||||
// do and is exactly where boundary defects live, so neither policy may
|
||||
// quietly refuse to offer it.
|
||||
{"tap a disabled element", `const e = state.ax.find("id:Off"); return e ? [Tap({on: e})] : [];`},
|
||||
{"tap a disabled selector", `return [Tap({on: "id:Off"})];`},
|
||||
{"double-tap an element", `const e = state.ax.find("id:Save"); return e ? [DoubleTap({on: e})] : [];`},
|
||||
{"long-press an element", `const e = state.ax.find("id:Save"); return e ? [LongPress({on: e})] : [];`},
|
||||
{"type into an element", `const e = state.ax.find("id:Amount"); return e ? [InputText({into: e, text: "42"})] : [];`},
|
||||
@@ -60,7 +67,7 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) {
|
||||
}
|
||||
|
||||
modelVerifier := loadAuthoredSpec(t, spec, tree)
|
||||
candidates := modelVerifier.Candidates(verifier.LabelSourceVisibleText)
|
||||
candidates := mustCandidates(t, modelVerifier, verifier.LabelSourceVisibleText)
|
||||
if len(candidates) != 1 {
|
||||
t.Fatalf("model was offered %d candidates, want the one authored action", len(candidates))
|
||||
}
|
||||
|
||||
@@ -110,8 +110,11 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
|
||||
return verifier.Action{}, err
|
||||
}
|
||||
|
||||
selection, call := s.selectViaLLM(ctx)
|
||||
selection, call, err := s.selectViaLLM(ctx)
|
||||
s.record(stepIndex, call)
|
||||
if err != nil {
|
||||
return verifier.Action{}, err
|
||||
}
|
||||
if call.Outcome != trace.LLMOutcomeSelected {
|
||||
// Every other outcome skips the step; the next step re-observes and
|
||||
// tries again. The record says which one it was.
|
||||
@@ -129,12 +132,21 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
|
||||
// The returned record is complete whichever way the selection ended: its
|
||||
// Outcome is trace.LLMOutcomeSelected exactly when the returned selection is
|
||||
// usable.
|
||||
func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall) {
|
||||
//
|
||||
// The error is the spec refusing this policy rather than a step going nowhere:
|
||||
// every later step would refuse identically, so the run stops instead of
|
||||
// recording two hundred skipped steps.
|
||||
func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall, error) {
|
||||
call := trace.LLMCall{Timestamp: time.Now(), Model: s.model}
|
||||
candidates := s.verifier.Candidates(s.labelSource)
|
||||
candidates, err := s.verifier.Candidates(s.labelSource)
|
||||
if err != nil {
|
||||
call.Outcome = trace.LLMOutcomeCandidatesFailed
|
||||
call.Error = err.Error()
|
||||
return llmSelection{}, call, err
|
||||
}
|
||||
if len(candidates) == 0 {
|
||||
call.Outcome = trace.LLMOutcomeNoCandidates
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
call.Candidates = recordCandidates(candidates)
|
||||
|
||||
@@ -149,7 +161,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
||||
s.logger.Warn("llm action selection failed", "err", err)
|
||||
call.Outcome = trace.LLMOutcomeRequestFailed
|
||||
call.Error = err.Error()
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
call.ServedModel = response.Model
|
||||
call.PromptTokens = response.Usage.PromptTokens
|
||||
@@ -158,7 +170,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
||||
if len(response.Choices) == 0 {
|
||||
s.logger.Warn("llm returned no choices")
|
||||
call.Outcome = trace.LLMOutcomeNoChoices
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
call.RawResponse = response.Choices[0].Message.Content
|
||||
|
||||
@@ -167,7 +179,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
||||
s.logger.Warn("llm output unusable", "err", err)
|
||||
call.Outcome = trace.LLMOutcomeUnparsableResponse
|
||||
call.Error = err.Error()
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
call.Choice = output.Choice
|
||||
call.EchoedAction = output.ChosenAction
|
||||
@@ -176,7 +188,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
||||
if output.Choice < 1 || output.Choice > len(candidates) {
|
||||
s.logger.Warn("llm choice out of range", "choice", output.Choice, "candidates", len(candidates))
|
||||
call.Outcome = trace.LLMOutcomeChoiceOutOfRange
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
candidate := candidates[output.Choice-1]
|
||||
// Strict skip: the echoed action must match the numbered entry, so a model
|
||||
@@ -187,14 +199,14 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
||||
s.logger.Warn("llm chosen_action mismatch; skipping",
|
||||
"choice", output.Choice, "echoed", output.ChosenAction, "candidate", candidate.Description)
|
||||
call.Outcome = trace.LLMOutcomeEchoMismatch
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
action, err := s.actionForCandidate(candidate, output.Text)
|
||||
if err != nil {
|
||||
s.logger.Warn("building action from candidate failed", "choice", output.Choice, "err", err)
|
||||
call.Outcome = trace.LLMOutcomeActionBuildFailed
|
||||
call.Error = err.Error()
|
||||
return llmSelection{}, call
|
||||
return llmSelection{}, call, nil
|
||||
}
|
||||
call.Outcome = trace.LLMOutcomeSelected
|
||||
return llmSelection{
|
||||
@@ -202,7 +214,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
||||
reasoning: output.Reasoning,
|
||||
choice: output.Choice,
|
||||
chosenAction: candidate.Description,
|
||||
}, call
|
||||
}, call, nil
|
||||
}
|
||||
|
||||
// record stamps the step this selection belongs to and appends the record. A
|
||||
|
||||
@@ -234,6 +234,17 @@ func lastCall(t *testing.T, source *llmSource) trace.LLMCall {
|
||||
return calls[len(calls)-1]
|
||||
}
|
||||
|
||||
// mustCandidates enumerates the model policy's list, failing the test on the
|
||||
// refusal an authored multi-item sampler raises.
|
||||
func mustCandidates(t *testing.T, v *verifier.Verifier, labelSource string) []verifier.ActionCandidate {
|
||||
t.Helper()
|
||||
candidates, err := v.Candidates(labelSource)
|
||||
if err != nil {
|
||||
t.Fatalf("Candidates: %v", err)
|
||||
}
|
||||
return candidates
|
||||
}
|
||||
|
||||
func newLLMSource(t *testing.T, fake *fakeOpenRouter) (*llmSource, *verifier.Verifier) {
|
||||
t.Helper()
|
||||
return newLLMSourceWithSpec(t, fake, llmFixtureSpec)
|
||||
@@ -428,7 +439,7 @@ func TestLLMSourceRecordsTheLabelsTheModelSaw(t *testing.T) {
|
||||
source.labelSource = want.labelSource
|
||||
pushSnapshotTree(t, verifierInstance, labelSplitTreeJSON)
|
||||
|
||||
tap := candidateByKind(t, verifierInstance.Candidates(want.labelSource), verifier.ActionKindTap)
|
||||
tap := candidateByKind(t, mustCandidates(t, verifierInstance, want.labelSource), verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
fake.chosenAction = tap.Description
|
||||
if _, err := source.NextAction(context.Background(), 0); err != nil {
|
||||
@@ -550,7 +561,7 @@ func TestLLMSourceDrivesExecutedActions(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
|
||||
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
|
||||
|
||||
// Step 1: the model picks the Tap on Submit by its number, echoing its
|
||||
// description.
|
||||
@@ -629,7 +640,7 @@ func TestLLMSourceAcceptsEchoWithWeightSuffix(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
|
||||
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
|
||||
|
||||
tap := candidateByKind(t, candidates, verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
@@ -664,7 +675,7 @@ func TestLLMSourceStrictSkipsOnEchoMismatch(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
|
||||
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
|
||||
|
||||
// A valid number, but the echoed action disagrees with that numbered entry:
|
||||
// the model reasoned about one control and picked another's number.
|
||||
@@ -701,7 +712,7 @@ func TestLLMSourceEchoGuardAdmitsARepeatedDescription(t *testing.T) {
|
||||
pushSnapshotTree(t, verifierInstance, llmSharedLabelTreeJSON)
|
||||
|
||||
var repeated []verifier.ActionCandidate
|
||||
for _, candidate := range verifierInstance.Candidates(verifier.LabelSourceVisibleText) {
|
||||
for _, candidate := range mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) {
|
||||
if candidate.Description == `Tap "Delete"` {
|
||||
repeated = append(repeated, candidate)
|
||||
}
|
||||
@@ -763,7 +774,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
fake.chosenAction = `Tap "Something Else"`
|
||||
if _, err := source.NextAction(context.Background(), stepIndex); !errors.Is(err, verifier.ErrNoAction) {
|
||||
@@ -784,7 +795,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
fake.chosenAction = tap.Description
|
||||
if _, err := source.NextAction(context.Background(), stepIndex); err != nil {
|
||||
@@ -832,7 +843,7 @@ func TestLLMCallRecordsCandidateListAsShown(t *testing.T) {
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
source.instructions = "hunt for double submits"
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
fake.chosenAction = tap.Description
|
||||
if _, err := source.NextAction(context.Background(), 1); err != nil {
|
||||
@@ -920,7 +931,7 @@ func TestLLMCallFileRecordsUsageLatencyAndScreenshot(t *testing.T) {
|
||||
source.recorder = writer
|
||||
|
||||
pushLLMSnapshotAtStep(t, verifierInstance, 4)
|
||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
fake.chosenAction = tap.Description
|
||||
if _, err := source.NextAction(context.Background(), 4); err != nil {
|
||||
@@ -960,7 +971,7 @@ func TestLLMCallScreenshotNamesObservedStep(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSource(t, fake)
|
||||
pushLLMSnapshotAtStep(t, verifierInstance, 4)
|
||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||
fake.choice = tap.Index
|
||||
fake.chosenAction = tap.Description
|
||||
if _, err := source.NextAction(context.Background(), 6); err != nil {
|
||||
@@ -1030,3 +1041,35 @@ func tinyPNG(t *testing.T) []byte {
|
||||
}
|
||||
return buffer.Bytes()
|
||||
}
|
||||
|
||||
// llmSamplerFixtureSpec drives the model policy over an authored leaf that
|
||||
// samples one of three targets, which is the shape the seeded picker draws from
|
||||
// and the model policy cannot.
|
||||
const llmSamplerFixtureSpec = `
|
||||
import { actions, from, llm, Tap, always } from "@sanderling/spec";
|
||||
globalThis.properties = { ok: always(() => true) };
|
||||
const targets = from(["id:Submit", "id:Name"]);
|
||||
globalThis.actions = actions(() => [Tap({ on: targets.generate() })]);
|
||||
globalThis.generator = llm({ model: "test/model" });
|
||||
`
|
||||
|
||||
// TestLLMSourceRefusesAMultiItemAuthoredSampler: the step must fail the run, not
|
||||
// skip. A skip would leave the model quietly fuzzing a spec whose authored
|
||||
// targets it can never reach past the first, which is the comparison the seeded
|
||||
// arm is measured against.
|
||||
func TestLLMSourceRefusesAMultiItemAuthoredSampler(t *testing.T) {
|
||||
fake := newFakeOpenRouter(t)
|
||||
source, verifierInstance := newLLMSourceWithSpec(t, fake, llmSamplerFixtureSpec)
|
||||
pushLLMSnapshot(t, verifierInstance)
|
||||
|
||||
_, err := source.NextAction(context.Background(), 1)
|
||||
if err == nil || errors.Is(err, verifier.ErrNoAction) {
|
||||
t.Fatalf("NextAction err = %v, want the run to stop on a sampler the model cannot draw", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "targets.generate()") {
|
||||
t.Errorf("error does not name the offending leaf: %v", err)
|
||||
}
|
||||
if outcome := lastCall(t, source).Outcome; outcome != trace.LLMOutcomeCandidatesFailed {
|
||||
t.Errorf("recorded outcome = %q, want %q", outcome, trace.LLMOutcomeCandidatesFailed)
|
||||
}
|
||||
}
|
||||
@@ -35,6 +35,9 @@ const (
|
||||
// LLMOutcomeNoCandidates: the action tree yielded nothing on this screen, so
|
||||
// no call was made.
|
||||
LLMOutcomeNoCandidates = "no_candidates"
|
||||
// LLMOutcomeCandidatesFailed: the action tree cannot be enumerated for this
|
||||
// policy at all (an authored leaf samples), which aborts the run.
|
||||
LLMOutcomeCandidatesFailed = "candidates_failed"
|
||||
// LLMOutcomeRequestFailed: the provider call failed (transport, timeout,
|
||||
// non-2xx).
|
||||
LLMOutcomeRequestFailed = "request_failed"
|
||||
|
||||
Reference in new issue
Block a user