mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
feat(folio): drive spec.ts under --generator llm; drop spec-llm.ts
This commit is contained in:
1 parent
98483926a6
commit
9845b89173
4 files changed
+48
-38
No files matched your search
@@ -78,24 +78,26 @@ DURATION=5m
|
|||||||
|
|
||||||
Traces land in `./sanderling/runs/<timestamp>/`.
|
Traces land in `./sanderling/runs/<timestamp>/`.
|
||||||
|
|
||||||
## Run with the LLM action backend
|
## Run with the LLM action generator
|
||||||
|
|
||||||
`sanderling/spec-llm.ts` reuses the same properties and login setup but swaps the
|
The same `sanderling/spec.ts` runs under either generator: `--generator seeded`
|
||||||
seeded fuzzer for `llm({ model })`: a vision model chooses which of the
|
(the default weighted fuzzer) or `--generator llm`, where a vision model picks
|
||||||
already-enumerated candidates to act on each step, from the screenshot.
|
from the SAME weighted candidate set — reading the screenshot plus a numbered,
|
||||||
|
weight-annotated list of concrete actions — and returns one number. The spec's
|
||||||
|
`generator = llm({ model, instructions })` export configures it.
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
export OPENROUTER_API_KEY=sk-or-... # or OPENAI_API_KEY=sk-... for OpenAI direct
|
export OPENROUTER_API_KEY=sk-or-... # or OPENAI_API_KEY=sk-... for OpenAI direct
|
||||||
sanderling test --spec sanderling/spec-llm.ts --bundle-id app.folio --duration 2m
|
just test-llm # or: sanderling test --generator llm --spec sanderling/spec.ts --bundle-id app.folio
|
||||||
```
|
```
|
||||||
|
|
||||||
OpenRouter wins when both keys are set. With a plain OpenAI key, drop the vendor
|
OpenRouter wins when both keys are set. With a plain OpenAI key, drop the vendor
|
||||||
prefix from the model id in `spec-llm.ts` (`gpt-5.4-nano`, not
|
prefix from the model id in `spec.ts` (`gpt-5.4-nano`, not
|
||||||
`openai/gpt-5.4-nano`). The model must support image input **and** strict
|
`openai/gpt-5.4-nano`). The model must support image input **and** strict
|
||||||
`json_schema` structured outputs. Each step is one multimodal call, so keep the
|
`json_schema` structured outputs. Each step is one multimodal call, so keep the
|
||||||
duration / step budget modest. The trace records the model's reasoning and
|
duration / step budget modest. The trace records the model's reasoning, the
|
||||||
`source: "llm"` on each chosen action, so the replay UI shows why each pick was
|
chosen number, and `source: "llm"` on each action, so the replay UI shows why
|
||||||
made.
|
each pick was made.
|
||||||
|
|
||||||
## Run a sanderling test (iOS)
|
## Run a sanderling test (iOS)
|
||||||
|
|
||||||
|
|||||||
@@ -176,6 +176,30 @@ test: install
|
|||||||
--seed "{{seed}}" \
|
--seed "{{seed}}" \
|
||||||
--output "{{output}}"
|
--output "{{output}}"
|
||||||
|
|
||||||
|
# Run 'sanderling test' with the LLM action generator instead of the seeded
|
||||||
|
# fuzzer. Needs OPENROUTER_API_KEY (or OPENAI_API_KEY) in the environment; the
|
||||||
|
# model is configured by generator = llm({...}) in spec.ts.
|
||||||
|
test-llm: install
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
avd_flag=()
|
||||||
|
if [[ -n "{{avd}}" ]]; then
|
||||||
|
avd_flag=(--avd "{{avd}}")
|
||||||
|
fi
|
||||||
|
if [[ -n "{{android_device}}" ]]; then
|
||||||
|
avd_flag+=(--device "{{android_device}}")
|
||||||
|
fi
|
||||||
|
apk="{{justfile_directory()}}/app/androidApp/build/outputs/apk/debug/androidApp-debug.apk"
|
||||||
|
"{{sanderling}}" test \
|
||||||
|
--spec "{{justfile_directory()}}/sanderling/spec.ts" \
|
||||||
|
--bundle-id app.folio \
|
||||||
|
--generator llm \
|
||||||
|
"${avd_flag[@]}" \
|
||||||
|
--android-app-path "$apk" \
|
||||||
|
--duration "{{duration}}" \
|
||||||
|
--seed "{{seed}}" \
|
||||||
|
--output "{{output}}"
|
||||||
|
|
||||||
# Serve the wasmJs web app from a webpack dev server with COOP/COEP headers.
|
# Serve the wasmJs web app from a webpack dev server with COOP/COEP headers.
|
||||||
web:
|
web:
|
||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
|
|||||||
@@ -1,29 +0,0 @@
|
|||||||
// LLM action-backend variant of spec.ts.
|
|
||||||
//
|
|
||||||
// The properties and the login `setup` are reused verbatim; only the action
|
|
||||||
// generator changes. Instead of the seeded fuzzer drawing a random candidate,
|
|
||||||
// `llm({ model })` hands selection to a vision model: each step it sees the
|
|
||||||
// screenshot plus the candidate list the system already enumerates and returns
|
|
||||||
// which candidate to act on. The candidate set, the typed input values, action
|
|
||||||
// execution, and the trace are all identical to the seeded run.
|
|
||||||
//
|
|
||||||
// Requirements:
|
|
||||||
// - OPENROUTER_API_KEY (OpenRouter) or OPENAI_API_KEY (OpenAI) in the
|
|
||||||
// environment; OpenRouter wins when both are set. With a plain OpenAI key,
|
|
||||||
// drop the vendor prefix from the model id ("gpt-5.4-nano").
|
|
||||||
// - A model that supports image input AND strict json_schema structured
|
|
||||||
// outputs. A model lacking either fails clearly.
|
|
||||||
import { llm } from "@sanderling/spec";
|
|
||||||
|
|
||||||
export { properties, setup } from "./spec";
|
|
||||||
|
|
||||||
// instructions only describe WHAT the app is — its purpose and features. They
|
|
||||||
// say nothing about HOW to test it: no bug, no technique, no "try to break it"
|
|
||||||
// (the base prompt already carries the bug-finding goal). The model figures out
|
|
||||||
// how to test entirely on its own. If we encoded the answer here, a "pass"
|
|
||||||
// would prove nothing and the feature would be worse than useless.
|
|
||||||
export const actionsRoot = llm({
|
|
||||||
model: "gpt-5.4-nano",
|
|
||||||
instructions:
|
|
||||||
"Folio is a personal-finance ledger app. After signing in, the home screen lists accounts, each with a balance. You can create accounts, open an account to see its ledger, and add transactions; each transaction has an amount and changes that account's balance and the overall total.",
|
|
||||||
});
|
|
||||||
@@ -6,6 +6,7 @@ import {
|
|||||||
extract,
|
extract,
|
||||||
from,
|
from,
|
||||||
integers,
|
integers,
|
||||||
|
llm,
|
||||||
next,
|
next,
|
||||||
weighted,
|
weighted,
|
||||||
whenRoute,
|
whenRoute,
|
||||||
@@ -187,3 +188,15 @@ export const actionsRoot = weighted(
|
|||||||
[5, doubleTaps],
|
[5, doubleTaps],
|
||||||
[25, defaultActions],
|
[25, defaultActions],
|
||||||
);
|
);
|
||||||
|
|
||||||
|
// The LLM generator is orthogonal to actionsRoot: with `--generator llm` a model
|
||||||
|
// picks from the SAME weighted candidate set above, reading the screenshot and a
|
||||||
|
// numbered, weight-annotated list; the default `--generator seeded` ignores it.
|
||||||
|
// instructions describe only WHAT the app is, never HOW to test it — the model
|
||||||
|
// figures out how to surface bugs on its own. With a plain OpenAI key, drop the
|
||||||
|
// vendor prefix from the model id.
|
||||||
|
export const generator = llm({
|
||||||
|
model: "gpt-5.4-nano",
|
||||||
|
instructions:
|
||||||
|
"Folio is a personal-finance ledger app. After signing in, the home screen lists accounts, each with a balance. You can create accounts, open an account to see its ledger, and add transactions; each transaction has an amount and changes that account's balance and the overall total.",
|
||||||
|
});
|
||||||
Reference in new issue
Block a user