mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-04 12:07:09 +00:00
fix(runner): abort on a candidate enumeration that refused
Recorded as candidates_failed before the run stops, so the trace says why. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
0094a7fc64
commit
7880e3e70f
4 files changed
+89
-24
No files matched your search
@@ -11,13 +11,15 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
// authoredParityTreeJSON holds one target per authored action shape: a button to
|
// authoredParityTreeJSON holds one target per authored action shape: a button to
|
||||||
// tap, a field to type into, and a scrollable container to scroll.
|
// tap, a field to type into, a scrollable container to scroll, and a disabled
|
||||||
|
// button, which is a target like any other here.
|
||||||
const authoredParityTreeJSON = `{
|
const authoredParityTreeJSON = `{
|
||||||
"attributes": {"bounds": "[0,0,400,800]"},
|
"attributes": {"bounds": "[0,0,400,800]"},
|
||||||
"children": [
|
"children": [
|
||||||
{"attributes": {"resource-id": "Save", "text": "Save", "bounds": "[0,0,200,60]"}, "clickable": true, "enabled": true, "children": []},
|
{"attributes": {"resource-id": "Save", "text": "Save", "bounds": "[0,0,200,60]"}, "clickable": true, "enabled": true, "children": []},
|
||||||
{"attributes": {"resource-id": "Amount", "class": "EditText", "hintText": "Amount", "bounds": "[0,100,400,160]"}, "enabled": true, "children": []},
|
{"attributes": {"resource-id": "Amount", "class": "EditText", "hintText": "Amount", "bounds": "[0,100,400,160]"}, "enabled": true, "children": []},
|
||||||
{"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []}
|
{"attributes": {"resource-id": "List", "scrollable": "true", "bounds": "[0,300,400,700]"}, "children": []},
|
||||||
|
{"attributes": {"resource-id": "Off", "text": "Off", "bounds": "[0,700,200,760]"}, "clickable": true, "enabled": false, "children": []}
|
||||||
]
|
]
|
||||||
}`
|
}`
|
||||||
|
|
||||||
@@ -36,6 +38,11 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) {
|
|||||||
}{
|
}{
|
||||||
{"tap an element", `const e = state.ax.find("id:Save"); return e ? [Tap({on: e})] : [];`},
|
{"tap an element", `const e = state.ax.find("id:Save"); return e ? [Tap({on: e})] : [];`},
|
||||||
{"tap a selector", `return [Tap({on: "id:Save"})];`},
|
{"tap a selector", `return [Tap({on: "id:Save"})];`},
|
||||||
|
// Attempting a disabled control is a legitimate thing for a UI fuzzer to
|
||||||
|
// do and is exactly where boundary defects live, so neither policy may
|
||||||
|
// quietly refuse to offer it.
|
||||||
|
{"tap a disabled element", `const e = state.ax.find("id:Off"); return e ? [Tap({on: e})] : [];`},
|
||||||
|
{"tap a disabled selector", `return [Tap({on: "id:Off"})];`},
|
||||||
{"double-tap an element", `const e = state.ax.find("id:Save"); return e ? [DoubleTap({on: e})] : [];`},
|
{"double-tap an element", `const e = state.ax.find("id:Save"); return e ? [DoubleTap({on: e})] : [];`},
|
||||||
{"long-press an element", `const e = state.ax.find("id:Save"); return e ? [LongPress({on: e})] : [];`},
|
{"long-press an element", `const e = state.ax.find("id:Save"); return e ? [LongPress({on: e})] : [];`},
|
||||||
{"type into an element", `const e = state.ax.find("id:Amount"); return e ? [InputText({into: e, text: "42"})] : [];`},
|
{"type into an element", `const e = state.ax.find("id:Amount"); return e ? [InputText({into: e, text: "42"})] : [];`},
|
||||||
@@ -60,7 +67,7 @@ func TestPoliciesDispatchTheSameAuthoredAction(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
modelVerifier := loadAuthoredSpec(t, spec, tree)
|
modelVerifier := loadAuthoredSpec(t, spec, tree)
|
||||||
candidates := modelVerifier.Candidates(verifier.LabelSourceVisibleText)
|
candidates := mustCandidates(t, modelVerifier, verifier.LabelSourceVisibleText)
|
||||||
if len(candidates) != 1 {
|
if len(candidates) != 1 {
|
||||||
t.Fatalf("model was offered %d candidates, want the one authored action", len(candidates))
|
t.Fatalf("model was offered %d candidates, want the one authored action", len(candidates))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -110,8 +110,11 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
|
|||||||
return verifier.Action{}, err
|
return verifier.Action{}, err
|
||||||
}
|
}
|
||||||
|
|
||||||
selection, call := s.selectViaLLM(ctx)
|
selection, call, err := s.selectViaLLM(ctx)
|
||||||
s.record(stepIndex, call)
|
s.record(stepIndex, call)
|
||||||
|
if err != nil {
|
||||||
|
return verifier.Action{}, err
|
||||||
|
}
|
||||||
if call.Outcome != trace.LLMOutcomeSelected {
|
if call.Outcome != trace.LLMOutcomeSelected {
|
||||||
// Every other outcome skips the step; the next step re-observes and
|
// Every other outcome skips the step; the next step re-observes and
|
||||||
// tries again. The record says which one it was.
|
// tries again. The record says which one it was.
|
||||||
@@ -129,12 +132,21 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
|
|||||||
// The returned record is complete whichever way the selection ended: its
|
// The returned record is complete whichever way the selection ended: its
|
||||||
// Outcome is trace.LLMOutcomeSelected exactly when the returned selection is
|
// Outcome is trace.LLMOutcomeSelected exactly when the returned selection is
|
||||||
// usable.
|
// usable.
|
||||||
func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall) {
|
//
|
||||||
|
// The error is the spec refusing this policy rather than a step going nowhere:
|
||||||
|
// every later step would refuse identically, so the run stops instead of
|
||||||
|
// recording two hundred skipped steps.
|
||||||
|
func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCall, error) {
|
||||||
call := trace.LLMCall{Timestamp: time.Now(), Model: s.model}
|
call := trace.LLMCall{Timestamp: time.Now(), Model: s.model}
|
||||||
candidates := s.verifier.Candidates(s.labelSource)
|
candidates, err := s.verifier.Candidates(s.labelSource)
|
||||||
|
if err != nil {
|
||||||
|
call.Outcome = trace.LLMOutcomeCandidatesFailed
|
||||||
|
call.Error = err.Error()
|
||||||
|
return llmSelection{}, call, err
|
||||||
|
}
|
||||||
if len(candidates) == 0 {
|
if len(candidates) == 0 {
|
||||||
call.Outcome = trace.LLMOutcomeNoCandidates
|
call.Outcome = trace.LLMOutcomeNoCandidates
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
call.Candidates = recordCandidates(candidates)
|
call.Candidates = recordCandidates(candidates)
|
||||||
|
|
||||||
@@ -149,7 +161,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
|||||||
s.logger.Warn("llm action selection failed", "err", err)
|
s.logger.Warn("llm action selection failed", "err", err)
|
||||||
call.Outcome = trace.LLMOutcomeRequestFailed
|
call.Outcome = trace.LLMOutcomeRequestFailed
|
||||||
call.Error = err.Error()
|
call.Error = err.Error()
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
call.ServedModel = response.Model
|
call.ServedModel = response.Model
|
||||||
call.PromptTokens = response.Usage.PromptTokens
|
call.PromptTokens = response.Usage.PromptTokens
|
||||||
@@ -158,7 +170,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
|||||||
if len(response.Choices) == 0 {
|
if len(response.Choices) == 0 {
|
||||||
s.logger.Warn("llm returned no choices")
|
s.logger.Warn("llm returned no choices")
|
||||||
call.Outcome = trace.LLMOutcomeNoChoices
|
call.Outcome = trace.LLMOutcomeNoChoices
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
call.RawResponse = response.Choices[0].Message.Content
|
call.RawResponse = response.Choices[0].Message.Content
|
||||||
|
|
||||||
@@ -167,7 +179,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
|||||||
s.logger.Warn("llm output unusable", "err", err)
|
s.logger.Warn("llm output unusable", "err", err)
|
||||||
call.Outcome = trace.LLMOutcomeUnparsableResponse
|
call.Outcome = trace.LLMOutcomeUnparsableResponse
|
||||||
call.Error = err.Error()
|
call.Error = err.Error()
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
call.Choice = output.Choice
|
call.Choice = output.Choice
|
||||||
call.EchoedAction = output.ChosenAction
|
call.EchoedAction = output.ChosenAction
|
||||||
@@ -176,7 +188,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
|||||||
if output.Choice < 1 || output.Choice > len(candidates) {
|
if output.Choice < 1 || output.Choice > len(candidates) {
|
||||||
s.logger.Warn("llm choice out of range", "choice", output.Choice, "candidates", len(candidates))
|
s.logger.Warn("llm choice out of range", "choice", output.Choice, "candidates", len(candidates))
|
||||||
call.Outcome = trace.LLMOutcomeChoiceOutOfRange
|
call.Outcome = trace.LLMOutcomeChoiceOutOfRange
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
candidate := candidates[output.Choice-1]
|
candidate := candidates[output.Choice-1]
|
||||||
// Strict skip: the echoed action must match the numbered entry, so a model
|
// Strict skip: the echoed action must match the numbered entry, so a model
|
||||||
@@ -187,14 +199,14 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
|||||||
s.logger.Warn("llm chosen_action mismatch; skipping",
|
s.logger.Warn("llm chosen_action mismatch; skipping",
|
||||||
"choice", output.Choice, "echoed", output.ChosenAction, "candidate", candidate.Description)
|
"choice", output.Choice, "echoed", output.ChosenAction, "candidate", candidate.Description)
|
||||||
call.Outcome = trace.LLMOutcomeEchoMismatch
|
call.Outcome = trace.LLMOutcomeEchoMismatch
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
action, err := s.actionForCandidate(candidate, output.Text)
|
action, err := s.actionForCandidate(candidate, output.Text)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
s.logger.Warn("building action from candidate failed", "choice", output.Choice, "err", err)
|
s.logger.Warn("building action from candidate failed", "choice", output.Choice, "err", err)
|
||||||
call.Outcome = trace.LLMOutcomeActionBuildFailed
|
call.Outcome = trace.LLMOutcomeActionBuildFailed
|
||||||
call.Error = err.Error()
|
call.Error = err.Error()
|
||||||
return llmSelection{}, call
|
return llmSelection{}, call, nil
|
||||||
}
|
}
|
||||||
call.Outcome = trace.LLMOutcomeSelected
|
call.Outcome = trace.LLMOutcomeSelected
|
||||||
return llmSelection{
|
return llmSelection{
|
||||||
@@ -202,7 +214,7 @@ func (s *llmSource) selectViaLLM(ctx context.Context) (llmSelection, trace.LLMCa
|
|||||||
reasoning: output.Reasoning,
|
reasoning: output.Reasoning,
|
||||||
choice: output.Choice,
|
choice: output.Choice,
|
||||||
chosenAction: candidate.Description,
|
chosenAction: candidate.Description,
|
||||||
}, call
|
}, call, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// record stamps the step this selection belongs to and appends the record. A
|
// record stamps the step this selection belongs to and appends the record. A
|
||||||
|
|||||||
@@ -234,6 +234,17 @@ func lastCall(t *testing.T, source *llmSource) trace.LLMCall {
|
|||||||
return calls[len(calls)-1]
|
return calls[len(calls)-1]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// mustCandidates enumerates the model policy's list, failing the test on the
|
||||||
|
// refusal an authored multi-item sampler raises.
|
||||||
|
func mustCandidates(t *testing.T, v *verifier.Verifier, labelSource string) []verifier.ActionCandidate {
|
||||||
|
t.Helper()
|
||||||
|
candidates, err := v.Candidates(labelSource)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Candidates: %v", err)
|
||||||
|
}
|
||||||
|
return candidates
|
||||||
|
}
|
||||||
|
|
||||||
func newLLMSource(t *testing.T, fake *fakeOpenRouter) (*llmSource, *verifier.Verifier) {
|
func newLLMSource(t *testing.T, fake *fakeOpenRouter) (*llmSource, *verifier.Verifier) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
return newLLMSourceWithSpec(t, fake, llmFixtureSpec)
|
return newLLMSourceWithSpec(t, fake, llmFixtureSpec)
|
||||||
@@ -428,7 +439,7 @@ func TestLLMSourceRecordsTheLabelsTheModelSaw(t *testing.T) {
|
|||||||
source.labelSource = want.labelSource
|
source.labelSource = want.labelSource
|
||||||
pushSnapshotTree(t, verifierInstance, labelSplitTreeJSON)
|
pushSnapshotTree(t, verifierInstance, labelSplitTreeJSON)
|
||||||
|
|
||||||
tap := candidateByKind(t, verifierInstance.Candidates(want.labelSource), verifier.ActionKindTap)
|
tap := candidateByKind(t, mustCandidates(t, verifierInstance, want.labelSource), verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
fake.chosenAction = tap.Description
|
fake.chosenAction = tap.Description
|
||||||
if _, err := source.NextAction(context.Background(), 0); err != nil {
|
if _, err := source.NextAction(context.Background(), 0); err != nil {
|
||||||
@@ -550,7 +561,7 @@ func TestLLMSourceDrivesExecutedActions(t *testing.T) {
|
|||||||
fake := newFakeOpenRouter(t)
|
fake := newFakeOpenRouter(t)
|
||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
pushLLMSnapshot(t, verifierInstance)
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
|
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
|
||||||
|
|
||||||
// Step 1: the model picks the Tap on Submit by its number, echoing its
|
// Step 1: the model picks the Tap on Submit by its number, echoing its
|
||||||
// description.
|
// description.
|
||||||
@@ -629,7 +640,7 @@ func TestLLMSourceAcceptsEchoWithWeightSuffix(t *testing.T) {
|
|||||||
fake := newFakeOpenRouter(t)
|
fake := newFakeOpenRouter(t)
|
||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
pushLLMSnapshot(t, verifierInstance)
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
|
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
|
||||||
|
|
||||||
tap := candidateByKind(t, candidates, verifier.ActionKindTap)
|
tap := candidateByKind(t, candidates, verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
@@ -664,7 +675,7 @@ func TestLLMSourceStrictSkipsOnEchoMismatch(t *testing.T) {
|
|||||||
fake := newFakeOpenRouter(t)
|
fake := newFakeOpenRouter(t)
|
||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
pushLLMSnapshot(t, verifierInstance)
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
candidates := verifierInstance.Candidates(verifier.LabelSourceVisibleText)
|
candidates := mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText)
|
||||||
|
|
||||||
// A valid number, but the echoed action disagrees with that numbered entry:
|
// A valid number, but the echoed action disagrees with that numbered entry:
|
||||||
// the model reasoned about one control and picked another's number.
|
// the model reasoned about one control and picked another's number.
|
||||||
@@ -701,7 +712,7 @@ func TestLLMSourceEchoGuardAdmitsARepeatedDescription(t *testing.T) {
|
|||||||
pushSnapshotTree(t, verifierInstance, llmSharedLabelTreeJSON)
|
pushSnapshotTree(t, verifierInstance, llmSharedLabelTreeJSON)
|
||||||
|
|
||||||
var repeated []verifier.ActionCandidate
|
var repeated []verifier.ActionCandidate
|
||||||
for _, candidate := range verifierInstance.Candidates(verifier.LabelSourceVisibleText) {
|
for _, candidate := range mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText) {
|
||||||
if candidate.Description == `Tap "Delete"` {
|
if candidate.Description == `Tap "Delete"` {
|
||||||
repeated = append(repeated, candidate)
|
repeated = append(repeated, candidate)
|
||||||
}
|
}
|
||||||
@@ -763,7 +774,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) {
|
|||||||
fake := newFakeOpenRouter(t)
|
fake := newFakeOpenRouter(t)
|
||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
pushLLMSnapshot(t, verifierInstance)
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
fake.chosenAction = `Tap "Something Else"`
|
fake.chosenAction = `Tap "Something Else"`
|
||||||
if _, err := source.NextAction(context.Background(), stepIndex); !errors.Is(err, verifier.ErrNoAction) {
|
if _, err := source.NextAction(context.Background(), stepIndex); !errors.Is(err, verifier.ErrNoAction) {
|
||||||
@@ -784,7 +795,7 @@ func TestLLMCallRecordSeparatesGuardSkipFromDecline(t *testing.T) {
|
|||||||
fake := newFakeOpenRouter(t)
|
fake := newFakeOpenRouter(t)
|
||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
pushLLMSnapshot(t, verifierInstance)
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
fake.chosenAction = tap.Description
|
fake.chosenAction = tap.Description
|
||||||
if _, err := source.NextAction(context.Background(), stepIndex); err != nil {
|
if _, err := source.NextAction(context.Background(), stepIndex); err != nil {
|
||||||
@@ -832,7 +843,7 @@ func TestLLMCallRecordsCandidateListAsShown(t *testing.T) {
|
|||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
source.instructions = "hunt for double submits"
|
source.instructions = "hunt for double submits"
|
||||||
pushLLMSnapshot(t, verifierInstance)
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
fake.chosenAction = tap.Description
|
fake.chosenAction = tap.Description
|
||||||
if _, err := source.NextAction(context.Background(), 1); err != nil {
|
if _, err := source.NextAction(context.Background(), 1); err != nil {
|
||||||
@@ -920,7 +931,7 @@ func TestLLMCallFileRecordsUsageLatencyAndScreenshot(t *testing.T) {
|
|||||||
source.recorder = writer
|
source.recorder = writer
|
||||||
|
|
||||||
pushLLMSnapshotAtStep(t, verifierInstance, 4)
|
pushLLMSnapshotAtStep(t, verifierInstance, 4)
|
||||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
fake.chosenAction = tap.Description
|
fake.chosenAction = tap.Description
|
||||||
if _, err := source.NextAction(context.Background(), 4); err != nil {
|
if _, err := source.NextAction(context.Background(), 4); err != nil {
|
||||||
@@ -960,7 +971,7 @@ func TestLLMCallScreenshotNamesObservedStep(t *testing.T) {
|
|||||||
fake := newFakeOpenRouter(t)
|
fake := newFakeOpenRouter(t)
|
||||||
source, verifierInstance := newLLMSource(t, fake)
|
source, verifierInstance := newLLMSource(t, fake)
|
||||||
pushLLMSnapshotAtStep(t, verifierInstance, 4)
|
pushLLMSnapshotAtStep(t, verifierInstance, 4)
|
||||||
tap := candidateByKind(t, verifierInstance.Candidates(verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
tap := candidateByKind(t, mustCandidates(t, verifierInstance, verifier.LabelSourceVisibleText), verifier.ActionKindTap)
|
||||||
fake.choice = tap.Index
|
fake.choice = tap.Index
|
||||||
fake.chosenAction = tap.Description
|
fake.chosenAction = tap.Description
|
||||||
if _, err := source.NextAction(context.Background(), 6); err != nil {
|
if _, err := source.NextAction(context.Background(), 6); err != nil {
|
||||||
@@ -1030,3 +1041,35 @@ func tinyPNG(t *testing.T) []byte {
|
|||||||
}
|
}
|
||||||
return buffer.Bytes()
|
return buffer.Bytes()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// llmSamplerFixtureSpec drives the model policy over an authored leaf that
|
||||||
|
// samples one of three targets, which is the shape the seeded picker draws from
|
||||||
|
// and the model policy cannot.
|
||||||
|
const llmSamplerFixtureSpec = `
|
||||||
|
import { actions, from, llm, Tap, always } from "@sanderling/spec";
|
||||||
|
globalThis.properties = { ok: always(() => true) };
|
||||||
|
const targets = from(["id:Submit", "id:Name"]);
|
||||||
|
globalThis.actions = actions(() => [Tap({ on: targets.generate() })]);
|
||||||
|
globalThis.generator = llm({ model: "test/model" });
|
||||||
|
`
|
||||||
|
|
||||||
|
// TestLLMSourceRefusesAMultiItemAuthoredSampler: the step must fail the run, not
|
||||||
|
// skip. A skip would leave the model quietly fuzzing a spec whose authored
|
||||||
|
// targets it can never reach past the first, which is the comparison the seeded
|
||||||
|
// arm is measured against.
|
||||||
|
func TestLLMSourceRefusesAMultiItemAuthoredSampler(t *testing.T) {
|
||||||
|
fake := newFakeOpenRouter(t)
|
||||||
|
source, verifierInstance := newLLMSourceWithSpec(t, fake, llmSamplerFixtureSpec)
|
||||||
|
pushLLMSnapshot(t, verifierInstance)
|
||||||
|
|
||||||
|
_, err := source.NextAction(context.Background(), 1)
|
||||||
|
if err == nil || errors.Is(err, verifier.ErrNoAction) {
|
||||||
|
t.Fatalf("NextAction err = %v, want the run to stop on a sampler the model cannot draw", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "targets.generate()") {
|
||||||
|
t.Errorf("error does not name the offending leaf: %v", err)
|
||||||
|
}
|
||||||
|
if outcome := lastCall(t, source).Outcome; outcome != trace.LLMOutcomeCandidatesFailed {
|
||||||
|
t.Errorf("recorded outcome = %q, want %q", outcome, trace.LLMOutcomeCandidatesFailed)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -35,6 +35,9 @@ const (
|
|||||||
// LLMOutcomeNoCandidates: the action tree yielded nothing on this screen, so
|
// LLMOutcomeNoCandidates: the action tree yielded nothing on this screen, so
|
||||||
// no call was made.
|
// no call was made.
|
||||||
LLMOutcomeNoCandidates = "no_candidates"
|
LLMOutcomeNoCandidates = "no_candidates"
|
||||||
|
// LLMOutcomeCandidatesFailed: the action tree cannot be enumerated for this
|
||||||
|
// policy at all (an authored leaf samples), which aborts the run.
|
||||||
|
LLMOutcomeCandidatesFailed = "candidates_failed"
|
||||||
// LLMOutcomeRequestFailed: the provider call failed (transport, timeout,
|
// LLMOutcomeRequestFailed: the provider call failed (transport, timeout,
|
||||||
// non-2xx).
|
// non-2xx).
|
||||||
LLMOutcomeRequestFailed = "request_failed"
|
LLMOutcomeRequestFailed = "request_failed"
|
||||||
|
|||||||
Reference in new issue
Block a user