feat(runner): llmSource selects actions via OpenRouter

This commit is contained in:
pj committed 2026-06-12 10:33:20 +05:30
1 parent bae2ca6e8d
commit ad6b78c9bf
1 file changed
+394
+394
View File
@@ -0,0 +1,394 @@
package runner
import (
"bytes"
"context"
"encoding/base64"
"encoding/json"
"errors"
"fmt"
"image"
"image/color"
"image/png"
"log/slog"
"strings"
"github.com/priyanshujain/sanderling/internal/openrouter"
"github.com/priyanshujain/sanderling/internal/trace"
"github.com/priyanshujain/sanderling/internal/verifier"
)
const (
// llmMaxImageEdge downscales the screenshot's long edge to bound the payload
// while keeping the UI legible.
llmMaxImageEdge = 1024
// llmHistorySize is how many recent actions (and the screen each led to) the
// prompt carries to discourage loops.
llmHistorySize = 5
// llmMaxRanked caps the ranked-index list the model returns.
llmMaxRanked = 5
// swipeMinMagnitude is the floor for an LLM-chosen swipe distance, matching
// the seeded swipe builder's minimum.
swipeMinMagnitude = 200
)
// llmSystemPrompt frames the selection task. The model only ranks the numbered
// candidates the system already enumerated; it never invents actions.
const llmSystemPrompt = "You are exploring this app to surface bugs. Choose the most useful next action from the numbered candidates. Avoid repeating recent actions; prefer progress into new screens. Return only your ranked choices."
// llmSource selects each step's action with an OpenRouter model instead of the
// seeded random pick. It replaces ONLY the pick: the candidate list, the input
// values, and action execution are all reused unchanged. The spec's JS setup
// still runs first each tick (setup precedence), and the LLM drives once setup
// yields nothing.
type llmSource struct {
verifier *verifier.Verifier
client *openrouter.Client
model string
logger *slog.Logger
history *actionHistory
// lastSource/lastReasoning describe the most recent NextAction so the runner
// can stamp the trace. lastSource is "llm" only when the LLM (not setup)
// chose the action; lastReasoning is the model's rationale.
lastSource string
lastReasoning string
}
// NextAction returns the step's action. Setup precedence is preserved by
// running the JS path first (the llm marker is inert there, so a null result
// means setup yielded nothing); the LLM selection then takes over.
func (s *llmSource) NextAction(ctx context.Context) (verifier.Action, error) {
s.lastSource = ""
s.lastReasoning = ""
s.history.completeLast(s.verifier.CurrentScreen())
action, err := s.verifier.NextAction()
if err == nil {
s.history.add(describeAction(action))
return action, nil
}
if !errors.Is(err, verifier.ErrNoAction) {
return verifier.Action{}, err
}
action, reasoning, ok := s.selectViaLLM(ctx)
if !ok {
// Any failure (HTTP error, unusable output, no valid index) skips the
// step; the next step re-observes and tries again. No backend mixing.
return verifier.Action{}, verifier.ErrNoAction
}
s.lastSource = "llm"
s.lastReasoning = reasoning
s.history.add(describeAction(action))
return action, nil
}
// selectViaLLM runs one multimodal call and maps the first valid ranked index
// to an action. It returns ok=false on any error/empty/invalid output, logging
// the cause; the caller turns that into a skipped step.
func (s *llmSource) selectViaLLM(ctx context.Context) (verifier.Action, string, bool) {
candidates := s.verifier.AllCandidates()
if len(candidates) == 0 {
return verifier.Action{}, "", false
}
response, err := s.client.ChatCompletion(ctx, s.buildRequest(candidates))
if err != nil {
s.logger.Warn("llm action selection failed", "err", err)
return verifier.Action{}, "", false
}
if len(response.Choices) == 0 {
s.logger.Warn("llm returned no choices")
return verifier.Action{}, "", false
}
ranked, reasoning, err := parseRanked(response.Choices[0].Message.Content)
if err != nil {
s.logger.Warn("llm output unusable", "err", err)
return verifier.Action{}, "", false
}
for _, index := range ranked {
if index < 0 || index >= len(candidates) {
continue
}
action, err := actionFromCandidate(candidates[index], s.verifier.SampleInput)
if err != nil {
s.logger.Warn("building action from candidate failed", "index", index, "err", err)
continue
}
return action, reasoning, true
}
s.logger.Warn("llm returned no valid candidate index", "ranked", ranked, "candidates", len(candidates))
return verifier.Action{}, "", false
}
// buildRequest assembles the one-shot multimodal request: a system frame, the
// numbered candidate list plus recent-action memory, and the downscaled
// screenshot. The strict json_schema response format pins the ranked output.
func (s *llmSource) buildRequest(candidates []verifier.ActionCandidate) openrouter.Request {
userParts := []openrouter.ContentPart{openrouter.TextPart(s.userPrompt(candidates))}
if screenshot := s.verifier.Screenshot(); len(screenshot) > 0 {
if dataURL, ok := screenshotDataURL(screenshot, llmMaxImageEdge); ok {
userParts = append(userParts, openrouter.ImagePart(dataURL))
}
}
return openrouter.Request{
Model: s.model,
Messages: []openrouter.Message{
{Role: "system", Content: []openrouter.ContentPart{openrouter.TextPart(llmSystemPrompt)}},
{Role: "user", Content: userParts},
},
ResponseFormat: rankedResponseFormat(),
}
}
// userPrompt renders the numbered candidate list and the recent-action memory.
func (s *llmSource) userPrompt(candidates []verifier.ActionCandidate) string {
var builder strings.Builder
builder.WriteString("Candidate actions on the current screen:\n")
for _, candidate := range candidates {
fmt.Fprintf(&builder, "#%d %s %q\n", candidate.Index, candidate.Kind, candidate.Label)
}
if recent := s.history.recent(); len(recent) > 0 {
builder.WriteString("\nYour recent actions (oldest first) and the screen each led to:\n")
for _, entry := range recent {
screen := entry.screen
if screen == "" {
screen = "(current screen)"
}
fmt.Fprintf(&builder, "- %s -> %s\n", entry.action, screen)
}
}
builder.WriteString("\nReturn your ranked candidate indices, most useful first.")
return builder.String()
}
// rankedResponseFormat is the strict structured-output schema: a short
// reasoning string and a ranked list of candidate indices.
func rankedResponseFormat() *openrouter.ResponseFormat {
return &openrouter.ResponseFormat{
Type: "json_schema",
JSONSchema: openrouter.JSONSchema{
Name: "ranked_actions",
Strict: true,
Schema: map[string]any{
"type": "object",
"properties": map[string]any{
"reasoning": map[string]any{
"type": "string",
"description": "One short sentence on why the top choice is most useful.",
},
"ranked": map[string]any{
"type": "array",
"items": map[string]any{"type": "integer"},
"minItems": 1,
"maxItems": llmMaxRanked,
},
},
"required": []string{"reasoning", "ranked"},
"additionalProperties": false,
},
},
}
}
// rankedOutput is the model's structured response.
type rankedOutput struct {
Reasoning string `json:"reasoning"`
Ranked []int `json:"ranked"`
}
// parseRanked decodes the model's JSON content into ranked indices + reasoning.
func parseRanked(content string) ([]int, string, error) {
content = strings.TrimSpace(content)
if content == "" {
return nil, "", errors.New("empty content")
}
var out rankedOutput
if err := json.Unmarshal([]byte(content), &out); err != nil {
return nil, "", err
}
if len(out.Ranked) == 0 {
return nil, "", errors.New("no ranked indices")
}
return out.Ranked, out.Reasoning, nil
}
// actionFromCandidate maps a chosen candidate to a concrete action, reusing the
// corpus sampler for InputText text and the seeded gesture geometry for
// swipe/scroll. sampleInput is verifier.SampleInput, injected for testability.
func actionFromCandidate(candidate verifier.ActionCandidate, sampleInput func() (string, error)) (verifier.Action, error) {
action := verifier.Action{Kind: candidate.Kind, On: candidate.Selector, X: candidate.X, Y: candidate.Y}
switch candidate.Kind {
case verifier.ActionKindInputText:
text, err := sampleInput()
if err != nil {
return verifier.Action{}, err
}
action.Text = text
case verifier.ActionKindScroll:
action.Direction = "down"
// Leave endpoints zero so the runner derives the gesture from the target
// bounds (scrollEndpoints), exactly as for an authored Scroll.
action.X, action.Y = 0, 0
case verifier.ActionKindSwipe:
// A vertical drag upward from the center reveals lower content, sized off
// the element height like the seeded swipe builder.
magnitude := max(swipeMinMagnitude, candidate.Height*4/10)
action.FromX, action.FromY = candidate.X, candidate.Y
action.ToX = candidate.X
action.ToY = max(0, candidate.Y-magnitude)
action.X, action.Y = 0, 0
}
return action, nil
}
// describeAction renders a short action summary for the recent-action memory.
func describeAction(action verifier.Action) string {
switch action.Kind {
case verifier.ActionKindInputText:
return fmt.Sprintf("InputText %s = %q", actionTarget(action), action.Text)
case verifier.ActionKindScroll:
return fmt.Sprintf("Scroll %s %s", action.Direction, action.On)
case verifier.ActionKindSwipe:
return "Swipe"
case verifier.ActionKindPressKey:
return "PressKey " + action.Key
case verifier.ActionKindWait:
return "Wait"
default:
return fmt.Sprintf("%s %s", action.Kind, actionTarget(action))
}
}
func actionTarget(action verifier.Action) string {
if action.On != "" {
return action.On
}
return fmt.Sprintf("(%d,%d)", action.X, action.Y)
}
// historyEntry records one performed action and the screen it led to (filled on
// the following step, once that screen is observed).
type historyEntry struct {
action string
screen string
}
// actionHistory is a bounded ring of recent actions for the prompt.
type actionHistory struct {
entries []historyEntry
size int
}
func newActionHistory(size int) *actionHistory {
return &actionHistory{size: size}
}
// completeLast fills the most recent action's led-to screen with the
// just-observed screen, if it was still pending.
func (h *actionHistory) completeLast(screen string) {
if n := len(h.entries); n > 0 && h.entries[n-1].screen == "" {
h.entries[n-1].screen = screen
}
}
// add appends an action (its led-to screen pending) and trims to size.
func (h *actionHistory) add(action string) {
h.entries = append(h.entries, historyEntry{action: action})
if len(h.entries) > h.size {
h.entries = h.entries[len(h.entries)-h.size:]
}
}
func (h *actionHistory) recent() []historyEntry {
return h.entries
}
// stampActionSource records the backend that chose an action on the trace.
// Only an LLM-selected action (not a setup action the JS path produced) carries
// source="llm" and the model's reasoning.
func stampActionSource(traceAction *trace.Action, source ActionSource) {
if traceAction == nil {
return
}
llm, ok := source.(*llmSource)
if !ok || llm.lastSource == "" {
return
}
traceAction.Source = llm.lastSource
traceAction.LLMReasoning = llm.lastReasoning
}
// screenshotDataURL downscales the PNG and encodes it as a data URL for the
// image content part.
func screenshotDataURL(pngBytes []byte, maxEdge int) (string, bool) {
scaled := downscalePNG(pngBytes, maxEdge)
if len(scaled) == 0 {
return "", false
}
return "data:image/png;base64," + base64.StdEncoding.EncodeToString(scaled), true
}
// downscalePNG shrinks the image so its long edge is at most maxEdge, returning
// the original bytes when it is already small enough and nil on decode failure.
func downscalePNG(pngBytes []byte, maxEdge int) []byte {
source, err := png.Decode(bytes.NewReader(pngBytes))
if err != nil {
return nil
}
bounds := source.Bounds()
width, height := bounds.Dx(), bounds.Dy()
if width <= 0 || height <= 0 {
return nil
}
longEdge := max(width, height)
if longEdge <= maxEdge {
return pngBytes
}
scale := float64(maxEdge) / float64(longEdge)
newWidth := max(1, int(float64(width)*scale))
newHeight := max(1, int(float64(height)*scale))
var buffer bytes.Buffer
if err := png.Encode(&buffer, boxDownscale(source, newWidth, newHeight)); err != nil {
return nil
}
return buffer.Bytes()
}
// boxDownscale averages each destination pixel over its source box, a cheap
// dependency-free downscale that keeps text legible enough for the model.
func boxDownscale(source image.Image, newWidth, newHeight int) image.Image {
bounds := source.Bounds()
width, height := bounds.Dx(), bounds.Dy()
dest := image.NewRGBA(image.Rect(0, 0, newWidth, newHeight))
for dy := range newHeight {
sy0 := dy * height / newHeight
sy1 := max((dy+1)*height/newHeight, sy0+1)
for dx := range newWidth {
sx0 := dx * width / newWidth
sx1 := max((dx+1)*width/newWidth, sx0+1)
var r, g, b, a, count uint64
for sy := sy0; sy < sy1; sy++ {
for sx := sx0; sx < sx1; sx++ {
pr, pg, pb, pa := source.At(bounds.Min.X+sx, bounds.Min.Y+sy).RGBA()
r += uint64(pr)
g += uint64(pg)
b += uint64(pb)
a += uint64(pa)
count++
}
}
count = max(1, count)
dest.Set(dx, dy, color.RGBA64{
R: uint16(r / count),
G: uint16(g / count),
B: uint16(b / count),
A: uint16(a / count),
})
}
}
return dest
}