mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 11:07:10 +00:00
feat(folio): drive spec.ts under --generator llm; drop spec-llm.ts
This commit is contained in:
1 parent
98483926a6
commit
9845b89173
4 files changed
+48
-38
No files matched your search
@@ -78,24 +78,26 @@ DURATION=5m
|
||||
|
||||
Traces land in `./sanderling/runs/<timestamp>/`.
|
||||
|
||||
## Run with the LLM action backend
|
||||
## Run with the LLM action generator
|
||||
|
||||
`sanderling/spec-llm.ts` reuses the same properties and login setup but swaps the
|
||||
seeded fuzzer for `llm({ model })`: a vision model chooses which of the
|
||||
already-enumerated candidates to act on each step, from the screenshot.
|
||||
The same `sanderling/spec.ts` runs under either generator: `--generator seeded`
|
||||
(the default weighted fuzzer) or `--generator llm`, where a vision model picks
|
||||
from the SAME weighted candidate set — reading the screenshot plus a numbered,
|
||||
weight-annotated list of concrete actions — and returns one number. The spec's
|
||||
`generator = llm({ model, instructions })` export configures it.
|
||||
|
||||
```sh
|
||||
export OPENROUTER_API_KEY=sk-or-... # or OPENAI_API_KEY=sk-... for OpenAI direct
|
||||
sanderling test --spec sanderling/spec-llm.ts --bundle-id app.folio --duration 2m
|
||||
just test-llm # or: sanderling test --generator llm --spec sanderling/spec.ts --bundle-id app.folio
|
||||
```
|
||||
|
||||
OpenRouter wins when both keys are set. With a plain OpenAI key, drop the vendor
|
||||
prefix from the model id in `spec-llm.ts` (`gpt-5.4-nano`, not
|
||||
prefix from the model id in `spec.ts` (`gpt-5.4-nano`, not
|
||||
`openai/gpt-5.4-nano`). The model must support image input **and** strict
|
||||
`json_schema` structured outputs. Each step is one multimodal call, so keep the
|
||||
duration / step budget modest. The trace records the model's reasoning and
|
||||
`source: "llm"` on each chosen action, so the replay UI shows why each pick was
|
||||
made.
|
||||
duration / step budget modest. The trace records the model's reasoning, the
|
||||
chosen number, and `source: "llm"` on each action, so the replay UI shows why
|
||||
each pick was made.
|
||||
|
||||
## Run a sanderling test (iOS)
|
||||
|
||||
|
||||
@@ -176,6 +176,30 @@ test: install
|
||||
--seed "{{seed}}" \
|
||||
--output "{{output}}"
|
||||
|
||||
# Run 'sanderling test' with the LLM action generator instead of the seeded
|
||||
# fuzzer. Needs OPENROUTER_API_KEY (or OPENAI_API_KEY) in the environment; the
|
||||
# model is configured by generator = llm({...}) in spec.ts.
|
||||
test-llm: install
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
avd_flag=()
|
||||
if [[ -n "{{avd}}" ]]; then
|
||||
avd_flag=(--avd "{{avd}}")
|
||||
fi
|
||||
if [[ -n "{{android_device}}" ]]; then
|
||||
avd_flag+=(--device "{{android_device}}")
|
||||
fi
|
||||
apk="{{justfile_directory()}}/app/androidApp/build/outputs/apk/debug/androidApp-debug.apk"
|
||||
"{{sanderling}}" test \
|
||||
--spec "{{justfile_directory()}}/sanderling/spec.ts" \
|
||||
--bundle-id app.folio \
|
||||
--generator llm \
|
||||
"${avd_flag[@]}" \
|
||||
--android-app-path "$apk" \
|
||||
--duration "{{duration}}" \
|
||||
--seed "{{seed}}" \
|
||||
--output "{{output}}"
|
||||
|
||||
# Serve the wasmJs web app from a webpack dev server with COOP/COEP headers.
|
||||
web:
|
||||
#!/usr/bin/env bash
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
// LLM action-backend variant of spec.ts.
|
||||
//
|
||||
// The properties and the login `setup` are reused verbatim; only the action
|
||||
// generator changes. Instead of the seeded fuzzer drawing a random candidate,
|
||||
// `llm({ model })` hands selection to a vision model: each step it sees the
|
||||
// screenshot plus the candidate list the system already enumerates and returns
|
||||
// which candidate to act on. The candidate set, the typed input values, action
|
||||
// execution, and the trace are all identical to the seeded run.
|
||||
//
|
||||
// Requirements:
|
||||
// - OPENROUTER_API_KEY (OpenRouter) or OPENAI_API_KEY (OpenAI) in the
|
||||
// environment; OpenRouter wins when both are set. With a plain OpenAI key,
|
||||
// drop the vendor prefix from the model id ("gpt-5.4-nano").
|
||||
// - A model that supports image input AND strict json_schema structured
|
||||
// outputs. A model lacking either fails clearly.
|
||||
import { llm } from "@sanderling/spec";
|
||||
|
||||
export { properties, setup } from "./spec";
|
||||
|
||||
// instructions only describe WHAT the app is — its purpose and features. They
|
||||
// say nothing about HOW to test it: no bug, no technique, no "try to break it"
|
||||
// (the base prompt already carries the bug-finding goal). The model figures out
|
||||
// how to test entirely on its own. If we encoded the answer here, a "pass"
|
||||
// would prove nothing and the feature would be worse than useless.
|
||||
export const actionsRoot = llm({
|
||||
model: "gpt-5.4-nano",
|
||||
instructions:
|
||||
"Folio is a personal-finance ledger app. After signing in, the home screen lists accounts, each with a balance. You can create accounts, open an account to see its ledger, and add transactions; each transaction has an amount and changes that account's balance and the overall total.",
|
||||
});
|
||||
@@ -6,6 +6,7 @@ import {
|
||||
extract,
|
||||
from,
|
||||
integers,
|
||||
llm,
|
||||
next,
|
||||
weighted,
|
||||
whenRoute,
|
||||
@@ -187,3 +188,15 @@ export const actionsRoot = weighted(
|
||||
[5, doubleTaps],
|
||||
[25, defaultActions],
|
||||
);
|
||||
|
||||
// The LLM generator is orthogonal to actionsRoot: with `--generator llm` a model
|
||||
// picks from the SAME weighted candidate set above, reading the screenshot and a
|
||||
// numbered, weight-annotated list; the default `--generator seeded` ignores it.
|
||||
// instructions describe only WHAT the app is, never HOW to test it — the model
|
||||
// figures out how to surface bugs on its own. With a plain OpenAI key, drop the
|
||||
// vendor prefix from the model id.
|
||||
export const generator = llm({
|
||||
model: "gpt-5.4-nano",
|
||||
instructions:
|
||||
"Folio is a personal-finance ledger app. After signing in, the home screen lists accounts, each with a balance. You can create accounts, open an account to see its ledger, and add transactions; each transaction has an amount and changes that account's balance and the overall total.",
|
||||
});
|
||||
Reference in new issue
Block a user