diff --git a/examples/folio/README.md b/examples/folio/README.md index 7587a6c..8e539ce 100644 --- a/examples/folio/README.md +++ b/examples/folio/README.md @@ -78,24 +78,26 @@ DURATION=5m Traces land in `./sanderling/runs//`. -## Run with the LLM action backend +## Run with the LLM action generator -`sanderling/spec-llm.ts` reuses the same properties and login setup but swaps the -seeded fuzzer for `llm({ model })`: a vision model chooses which of the -already-enumerated candidates to act on each step, from the screenshot. +The same `sanderling/spec.ts` runs under either generator: `--generator seeded` +(the default weighted fuzzer) or `--generator llm`, where a vision model picks +from the SAME weighted candidate set — reading the screenshot plus a numbered, +weight-annotated list of concrete actions — and returns one number. The spec's +`generator = llm({ model, instructions })` export configures it. ```sh export OPENROUTER_API_KEY=sk-or-... # or OPENAI_API_KEY=sk-... for OpenAI direct -sanderling test --spec sanderling/spec-llm.ts --bundle-id app.folio --duration 2m +just test-llm # or: sanderling test --generator llm --spec sanderling/spec.ts --bundle-id app.folio ``` OpenRouter wins when both keys are set. With a plain OpenAI key, drop the vendor -prefix from the model id in `spec-llm.ts` (`gpt-5.4-nano`, not +prefix from the model id in `spec.ts` (`gpt-5.4-nano`, not `openai/gpt-5.4-nano`). The model must support image input **and** strict `json_schema` structured outputs. Each step is one multimodal call, so keep the -duration / step budget modest. The trace records the model's reasoning and -`source: "llm"` on each chosen action, so the replay UI shows why each pick was -made. +duration / step budget modest. The trace records the model's reasoning, the +chosen number, and `source: "llm"` on each action, so the replay UI shows why +each pick was made. ## Run a sanderling test (iOS) diff --git a/examples/folio/justfile b/examples/folio/justfile index df14b86..4328766 100644 --- a/examples/folio/justfile +++ b/examples/folio/justfile @@ -176,6 +176,30 @@ test: install --seed "{{seed}}" \ --output "{{output}}" +# Run 'sanderling test' with the LLM action generator instead of the seeded +# fuzzer. Needs OPENROUTER_API_KEY (or OPENAI_API_KEY) in the environment; the +# model is configured by generator = llm({...}) in spec.ts. +test-llm: install + #!/usr/bin/env bash + set -euo pipefail + avd_flag=() + if [[ -n "{{avd}}" ]]; then + avd_flag=(--avd "{{avd}}") + fi + if [[ -n "{{android_device}}" ]]; then + avd_flag+=(--device "{{android_device}}") + fi + apk="{{justfile_directory()}}/app/androidApp/build/outputs/apk/debug/androidApp-debug.apk" + "{{sanderling}}" test \ + --spec "{{justfile_directory()}}/sanderling/spec.ts" \ + --bundle-id app.folio \ + --generator llm \ + "${avd_flag[@]}" \ + --android-app-path "$apk" \ + --duration "{{duration}}" \ + --seed "{{seed}}" \ + --output "{{output}}" + # Serve the wasmJs web app from a webpack dev server with COOP/COEP headers. web: #!/usr/bin/env bash diff --git a/examples/folio/sanderling/spec-llm.ts b/examples/folio/sanderling/spec-llm.ts deleted file mode 100644 index af0671a..0000000 --- a/examples/folio/sanderling/spec-llm.ts +++ /dev/null @@ -1,29 +0,0 @@ -// LLM action-backend variant of spec.ts. -// -// The properties and the login `setup` are reused verbatim; only the action -// generator changes. Instead of the seeded fuzzer drawing a random candidate, -// `llm({ model })` hands selection to a vision model: each step it sees the -// screenshot plus the candidate list the system already enumerates and returns -// which candidate to act on. The candidate set, the typed input values, action -// execution, and the trace are all identical to the seeded run. -// -// Requirements: -// - OPENROUTER_API_KEY (OpenRouter) or OPENAI_API_KEY (OpenAI) in the -// environment; OpenRouter wins when both are set. With a plain OpenAI key, -// drop the vendor prefix from the model id ("gpt-5.4-nano"). -// - A model that supports image input AND strict json_schema structured -// outputs. A model lacking either fails clearly. -import { llm } from "@sanderling/spec"; - -export { properties, setup } from "./spec"; - -// instructions only describe WHAT the app is — its purpose and features. They -// say nothing about HOW to test it: no bug, no technique, no "try to break it" -// (the base prompt already carries the bug-finding goal). The model figures out -// how to test entirely on its own. If we encoded the answer here, a "pass" -// would prove nothing and the feature would be worse than useless. -export const actionsRoot = llm({ - model: "gpt-5.4-nano", - instructions: - "Folio is a personal-finance ledger app. After signing in, the home screen lists accounts, each with a balance. You can create accounts, open an account to see its ledger, and add transactions; each transaction has an amount and changes that account's balance and the overall total.", -}); diff --git a/examples/folio/sanderling/spec.ts b/examples/folio/sanderling/spec.ts index 649238a..4316613 100644 --- a/examples/folio/sanderling/spec.ts +++ b/examples/folio/sanderling/spec.ts @@ -6,6 +6,7 @@ import { extract, from, integers, + llm, next, weighted, whenRoute, @@ -187,3 +188,15 @@ export const actionsRoot = weighted( [5, doubleTaps], [25, defaultActions], ); + +// The LLM generator is orthogonal to actionsRoot: with `--generator llm` a model +// picks from the SAME weighted candidate set above, reading the screenshot and a +// numbered, weight-annotated list; the default `--generator seeded` ignores it. +// instructions describe only WHAT the app is, never HOW to test it — the model +// figures out how to surface bugs on its own. With a plain OpenAI key, drop the +// vendor prefix from the model id. +export const generator = llm({ + model: "gpt-5.4-nano", + instructions: + "Folio is a personal-finance ledger app. After signing in, the home screen lists accounts, each with a balance. You can create accounts, open an account to see its ledger, and add transactions; each transaction has an amount and changes that account's balance and the overall total.", +});