fix(testrun): the refusal asks whether the generator drove, not whether anything did

A dead provider against folio exited 0 on a real emulator: the login
setup dispatched three actions before the generator was consulted, so
DispatchedActions was 3 and the gate never fired while the generator
drove the app zero times across 83 steps. Any spec with a login setup
was immune, which is the normal case.

Summary counts generator actions separately and the refusal reads that.
NoActionsDispatchedError becomes NoGeneratorActionsError, because a run
that dispatched three login taps was lying in the old name.
This commit is contained in:
pj committed 2026-08-18 16:29:03 +05:30
1 parent 5a36b583ea
commit 58168a6a3e
6 files changed
+274 -29

No files matched your search

+16 -1
View File
@@ -64,11 +64,14 @@ type llmSource struct {
// can stamp the trace. lastSource is "llm" only when the LLM (not setup)
// chose the action; lastReasoning is the model's rationale. lastChoice is the
// 1-based number it picked and lastChosenAction the description it echoed, so
// the trace shows what the model believed it was doing.
// the trace shows what the model believed it was doing. lastFromSetup says
// the spec's setup produced the action, which is the app being put in
// position rather than the generator exploring it.
lastSource string
lastReasoning string
lastChoice int
lastChosenAction string
lastFromSetup bool
}
// llmSelection is the outcome of one LLM selection call.
@@ -95,12 +98,14 @@ func (s *llmSource) NextAction(ctx context.Context, stepIndex int) (verifier.Act
s.lastReasoning = ""
s.lastChoice = 0
s.lastChosenAction = ""
s.lastFromSetup = false
s.history.completeLast(s.verifier.CurrentScreen())
// Setup precedence only: the LLM replaces the seeded action root, so we run
// setup (e.g. login) first but never the weighted picker.
action, err := s.verifier.SetupAction()
if err == nil {
s.lastFromSetup = true
s.record(stepIndex, trace.LLMCall{Outcome: trace.LLMOutcomeSetupAction})
s.history.add(describeAction(action))
return action, nil
@@ -509,6 +514,16 @@ func stampActionSource(traceAction *trace.Action, source ActionSource) {
traceAction.LLMChosenAction = llm.lastChosenAction
}
// generatorChoseAction reports whether the action the source just returned came
// from the generator rather than from the spec's setup driving the app into
// position. Only the model source can tell them apart: the seeded picker
// resolves setup precedence inside the one JS call it makes, so everything it
// returns counts as the generator's.
func generatorChoseAction(source ActionSource) bool {
llm, ok := source.(*llmSource)
return !ok || !llm.lastFromSetup
}
// screenshotDataURL downscales the PNG and encodes it as a data URL for the
// image content part.
func screenshotDataURL(pngBytes []byte, maxEdge int) (string, bool) {
+162
View File
@@ -818,6 +818,168 @@ func TestRunner_EveryModelCallFailingIsNotACleanRun(t *testing.T) {
}
}
// llmLoginSetupSpec is the shape every spec with a login has: setup drives the
// app for its first steps and then yields nothing, leaving the rest of the run
// to the generator.
const llmLoginSetupSpec = `
import { llm, always, actions, taps, typing, weighted, Tap } from "@sanderling/spec";
globalThis.properties = { ok: always(() => true) };
let setupTapsLeft = 2;
globalThis.setup = actions(() => (setupTapsLeft-- > 0 ? [Tap({ on: "id:Submit" })] : []));
globalThis.actions = weighted([1, taps], [1, typing]);
globalThis.generator = llm({ model: "test/model" });
`
// TestRunner_SetupActionsAreNotTheGeneratorDrivingTheApp is the folio run: the
// spec's login setup dispatches the first steps, then every model call fails.
// Two actions reached the app and none of them explored it, and a run counted
// by dispatched actions alone reports that as a clean run.
func TestRunner_SetupActionsAreNotTheGeneratorDrivingTheApp(t *testing.T) {
fake := newFakeOpenRouter(t)
fake.server.Config.Handler = http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
w.WriteHeader(http.StatusTooManyRequests)
})
t.Setenv("OPENROUTER_API_KEY", "test-key")
t.Setenv("OPENROUTER_BASE_URL", fake.server.URL)
state := newHarnessWithSpec(t, llmLoginSetupSpec)
state.mock.HierarchyJSON = llmTreeJSON
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
summary, err := Run(ctx, Options{
Duration: 30 * time.Second,
IdleTimeout: 20 * time.Millisecond,
MaxSteps: 4,
Driver: state.mock,
Verifier: state.verifier,
TraceWriter: state.writer,
Generator: "llm",
LabelSource: verifier.LabelSourceVisibleText,
Logger: slog.New(slog.NewTextHandler(io.Discard, nil)),
})
if err != nil {
t.Fatalf("Run: %v", err)
}
wantOutcomes := []string{
trace.LLMOutcomeSetupAction, trace.LLMOutcomeSetupAction,
trace.LLMOutcomeRequestFailed, trace.LLMOutcomeRequestFailed,
}
calls := readLLMCalls(t, state.writer.Directory())
if len(calls) != len(wantOutcomes) {
t.Fatalf("recorded %d selection records, want %d", len(calls), len(wantOutcomes))
}
for index, call := range calls {
if call.Outcome != wantOutcomes[index] {
t.Fatalf("call at step %d ended %q, want %q", call.Step, call.Outcome, wantOutcomes[index])
}
}
lines := readTraceLines(t, state.writer.Directory())
if len(lines) != 4 {
t.Fatalf("wrote %d trace lines, want 4", len(lines))
}
for _, line := range lines[:2] {
if line.NextAction == nil || line.ActionSkipped != "" {
t.Errorf("step %d = %+v, want setup's action dispatched", line.Step, line)
}
}
for _, line := range lines[2:] {
if line.ActionSkipped != string(actionSkippedNoActionProduced) {
t.Errorf("step %d action_skipped = %q, want %q",
line.Step, line.ActionSkipped, actionSkippedNoActionProduced)
}
}
if summary.DispatchedActions != 2 {
t.Errorf("DispatchedActions = %d, want 2: setup drove the app twice",
summary.DispatchedActions)
}
if summary.GeneratorActions != 0 {
t.Errorf("GeneratorActions = %d, want 0: every model call failed",
summary.GeneratorActions)
}
if got := summary.SkippedActions[string(actionSkippedNoActionProduced)]; got != 2 {
t.Errorf("summary counted %d step(s) as %q, want 2: %v",
got, actionSkippedNoActionProduced, summary.SkippedActions)
}
taps := 0
for _, action := range state.mock.Actions() {
if action.Kind == mockdriver.ActionTap || action.Kind == mockdriver.ActionTapSelector {
taps++
}
}
if taps != 2 {
t.Errorf("the run drove the app %d time(s), want the 2 setup taps only", taps)
}
}
// TestRunner_SetupAndGeneratorBothDrivingIsAHealthyRun is the same spec with a
// provider that answers: setup drives its steps and the model drives the rest,
// which is what an ordinary login-fronted run looks like. Counting only the
// generator's actions must not turn it red.
func TestRunner_SetupAndGeneratorBothDrivingIsAHealthyRun(t *testing.T) {
fake := newFakeOpenRouter(t)
t.Setenv("OPENROUTER_API_KEY", "test-key")
t.Setenv("OPENROUTER_BASE_URL", fake.server.URL)
state := newHarnessWithSpec(t, llmLoginSetupSpec)
state.mock.HierarchyJSON = llmTreeJSON
pushSnapshotTree(t, state.verifier, llmTreeJSON)
tap := candidateByKind(t,
mustCandidates(t, state.verifier, verifier.LabelSourceVisibleText),
verifier.ActionKindTap)
fake.choice = tap.Index
fake.chosenAction = tap.Description
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
summary, err := Run(ctx, Options{
Duration: 30 * time.Second,
IdleTimeout: 20 * time.Millisecond,
MaxSteps: 4,
Driver: state.mock,
Verifier: state.verifier,
TraceWriter: state.writer,
Generator: "llm",
LabelSource: verifier.LabelSourceVisibleText,
Logger: slog.New(slog.NewTextHandler(io.Discard, nil)),
})
if err != nil {
t.Fatalf("Run: %v", err)
}
if summary.DispatchedActions != 4 {
t.Errorf("DispatchedActions = %d, want 4: every step drove the app",
summary.DispatchedActions)
}
if summary.GeneratorActions != 2 {
t.Errorf("GeneratorActions = %d, want 2: the model drove the two steps setup left it",
summary.GeneratorActions)
}
lines := readTraceLines(t, state.writer.Directory())
if len(lines) != 4 {
t.Fatalf("wrote %d trace lines, want 4", len(lines))
}
for _, line := range lines {
if line.NextAction == nil || line.ActionSkipped != "" {
t.Fatalf("step %d = %+v, want an action dispatched", line.Step, line)
}
}
for _, line := range lines[:2] {
if line.NextAction.Source != "" {
t.Errorf("step %d action source = %q, want none: setup chose it",
line.Step, line.NextAction.Source)
}
}
for _, line := range lines[2:] {
if line.NextAction.Source != "llm" {
t.Errorf("step %d action source = %q, want llm", line.Step, line.NextAction.Source)
}
}
}
// llmSetupFixtureSpec drives the first action from setup, so the model is never
// consulted for that step.
const llmSetupFixtureSpec = `
+10
View File
@@ -79,6 +79,13 @@ type Summary struct {
// A run at zero never touched the app, whatever its step count says, so its
// empty violation list is the reading of an instrument that measured nothing.
DispatchedActions int
// GeneratorActions counts the dispatched actions the generator chose. The
// spec's setup drives the app into its starting position before the
// generator is consulted, so a run at zero here explored nothing however
// many actions its login fired. Only the model generator separates the two:
// the seeded picker resolves setup precedence inside the one JS call it
// makes, so everything it returns counts as the generator's.
GeneratorActions int
}
type ViolationRecord struct {
@@ -499,6 +506,9 @@ func Run(ctx context.Context, options Options) (Summary, error) {
}
if nextErr == nil && !applySkipped {
summary.DispatchedActions++
if generatorChoseAction(actionSource) {
summary.GeneratorActions++
}
}
summary.Steps = stepIndex
if len(violations) > 0 {