Files
sanderling/internal/trace/writer.go
T
pj 7343085614 llm action-selection backend (#68)
* feat(spec): add llm() action-backend marker

* feat(spec): make llm marker inert on the JS picker

* feat(spec): expose __sanderlingSampleInput__ corpus draw

* feat(openrouter): minimal chat-completions client

* test(openrouter): cover request shape, parse, and errors

* feat(verifier): thread screenshot + capture corpus sampler

* feat(verifier): LLM accessors — candidates, config, sampler

* test(verifier): cover AllCandidates, LLMConfig, SampleInput

* feat(trace): record action Source and LLMReasoning

* feat(runner): thread step screenshot into PushSnapshot

* feat(runner): llmSource selects actions via OpenRouter

* feat(runner): wire llmSource selection and trace stamping

* test(runner): cover llmSource selection, mapping, downscale

* docs(folio): add llm action-backend example spec

* docs(folio): document the LLM action backend run

* feat(llmclient): support OPENAI_API_KEY, openrouter wins

* refactor(runner): rename openrouter package to llmclient

* docs: both api keys, example model gpt-5.4-nano

* docs: add pr style rules to claude.md

* fix(runner): explain action kinds in llm prompt to stop swipe loops

* feat(trace): record llm ranked list and chosen rank

* feat(runner): stamp llm ranked list and chosen rank on trace

* fix(runner): tap by selector to survive layout shift after observe

* revert(runner): drop selector-first tap; broke path/testTag selectors

* feat(spec): llm() accepts optional instructions

* feat(verifier): read llm instructions off config

* feat(runner): append spec instructions to llm system prompt

* docs(folio): describe app in llm spec instructions

* feat(bundler): map generator export to globalThis.generator

* feat(verifier): read llm config off globalThis.generator

* feat(runner): gate llm source on --generator flag

* feat(cmd): add --generator llm|seeded flag

* test: cover --generator flag parsing and pickSources gating

* feat(verifier): enumerate llm candidates by walking actionsRoot

collect-walk the weighted action tree: recurse weighted branches
accumulating selection probability, call authored leaves once for
concrete actions, enumerate builtins per element. label controls by
visible text (borrowing descendant text), fold gestures into directional
scrolls over scrollable containers, drop disabled, dedup descriptions.

* test(verifier): cover candidate enumeration walk

* feat(verifier): add SetupAction to walk setup without the seeded root

* test(verifier): cover SetupAction setup-only precedence

* refactor(llmclient): make JSONSchema.Schema raw json for pinned field order

* feat(trace): record llm choice number and chosen_action echo

* feat(runner): llm picks one number from weighted candidates

drop the seeded-root call for a setup-only precedence path, render a
numbered weighted candidate list, pin a reasoning-first choice schema,
strict-skip when chosen_action does not echo the numbered entry, and let
the model supply typed values (corpus fallback when empty).

* test(runner): cover choice schema, strict-skip, and setup precedence

* refactor(verifier): drop the superseded AllCandidates enumeration

* feat(folio): drive spec.ts under --generator llm; drop spec-llm.ts

* fix(verifier): label editable fields by hint, not the typed value

an editable field's own text is its transient content; prefer the hint
so the field is named by purpose and the label stays stable.

* test(runner): cover weight-suffixed echo and stripWeightSuffix

* fix(runner): accept chosen_action echo that carries the weight suffix

real runs showed the model copies the whole numbered line including the
trailing (w34) weight annotation, so strict-skip rejected ~91% of picks
and the llm was paralyzed. strip the weight suffix before comparing. also
nudge the prompt to stress-test repeated submissions (idempotency).

* fix(verifier): skip llm enumeration on cross-fade frames

a navhost mid-transition carries >1 route *Screen in a collapsed
coordinate space; acting on it taps garbage (soft keyboard). real runs
showed the llm acting on 44% of steps being such frames. skip them so the
llm re-observes a settled frame next step.

* feat(folio): show current balance on the add-transaction screen

renders the account's balance (testTag TxnCurrentBalance) below the
account name, above the credit/debit toggle, so before/after screenshots
carry comparison data.

* fix(replay): derive device space from screen extent, not first node

the first positive-bounds element is often a short status-bar node
(320x24 on android); using it gave a 320/24 aspect ratio that squashed
the screenshot overlay into a grey horizontal band. use the max extent
across elements (like the runner's screenBounds) instead.

* fix(folio): show balance as a compact one-line label

per review: one line, account-name-sized, e.g. "Balance: $0.00"
instead of a large balance card.

* fix(folio): move balance into the header, one compact line under the account name

* fix(replay): attribute deferred violations to the causing step, not detection

* fix(replay): show a step's own violations in both panels, no next-step bleed

* refactor(hierarchy): one Tree.Transitional, drop the duplicated cross-fade check

* chore: ignore .playwright-mcp scratch output

* docs: document the llm generator and --generator flag

* docs(spec): correct the llm() comment; config reads off globalThis.generator

* docs: add pr description rules
2026-07-31 21:12:00 +05:30

201 lines
7.3 KiB
Go

// Package trace records each run's steps, snapshots, and violations to disk for later inspection.
package trace
import (
"encoding/json"
"fmt"
"io"
"os"
"path/filepath"
"sync"
"time"
"github.com/priyanshujain/sanderling/internal/hierarchy"
)
type Step struct {
Index int `json:"step"`
Timestamp time.Time `json:"timestamp"`
Screen string `json:"screen,omitempty"`
Snapshots map[string]json.RawMessage `json:"snapshots,omitempty"`
// NextAction is the action chosen for the next iteration based on observing this step.
NextAction *Action `json:"next_action,omitempty"`
Exceptions []Exception `json:"exceptions,omitempty"`
Violations []string `json:"violations,omitempty"`
Hierarchy *hierarchy.Tree `json:"hierarchy,omitempty"`
Residuals map[string]json.RawMessage `json:"residuals,omitempty"`
Metrics *Metrics `json:"metrics,omitempty"`
ExtractorChanges map[string]ExtractorChange `json:"extractor_changes,omitempty"`
// Transitional marks a step whose hierarchy still showed a NavHost
// cross-fade (multiple route-level *Screen ids) after the runner's
// retry budget. The verifier is skipped for these steps so transient
// state does not poison the previous/current extractor advance.
Transitional bool `json:"transitional,omitempty"`
// SkippedVerification is set true exactly when the verifier was skipped
// for this step, so downstream tooling can tell a deliberately-skipped
// step from one that was verified and came back clean.
SkippedVerification bool `json:"skipped_verification,omitempty"`
// Witnesses records the violation witness for each property that newly
// violated at this step: the cause and the extractor values at onset.
Witnesses map[string]Witness `json:"witnesses,omitempty"`
}
// Witness is the trace-side record of a property violation: why it fired and a
// snapshot of every extractor's value at the violating step.
type Witness struct {
Reason string `json:"reason,omitempty"`
IsError bool `json:"is_error,omitempty"`
// Step is the step the failed obligation originated at: the step that
// caused the violation. For a deferred obligation (a next, an eventually)
// this is earlier than the step whose record carries the witness, which is
// where the failure was detected.
Step int `json:"step,omitempty"`
Extractors map[string]json.RawMessage `json:"extractors,omitempty"`
}
// ExtractorChange records the prev/curr JSON values of an extractor whose
// observation differed between two consecutive steps. Surfaced under
// violation rows in the replay UI as a "what changed at this step"
// breadcrumb.
type ExtractorChange struct {
Prev json.RawMessage `json:"prev"`
Curr json.RawMessage `json:"curr"`
}
type Metrics struct {
CPUPercent float64 `json:"cpu_percent"`
HeapBytes int64 `json:"heap_bytes,omitempty"`
TotalMemoryBytes int64 `json:"total_memory_bytes,omitempty"`
}
type Action struct {
Kind string `json:"kind"`
X int `json:"x,omitempty"`
Y int `json:"y,omitempty"`
FromX int `json:"from_x,omitempty"`
FromY int `json:"from_y,omitempty"`
ToX int `json:"to_x,omitempty"`
ToY int `json:"to_y,omitempty"`
Key string `json:"key,omitempty"`
Text string `json:"text,omitempty"`
DurationMillis int `json:"duration_millis,omitempty"`
Selector string `json:"selector,omitempty"`
ResolvedBounds *BoundsRecord `json:"resolved_bounds,omitempty"`
TapPoint *PointRecord `json:"tap_point,omitempty"`
// Source names the backend that chose this action: "llm" when the LLM
// action backend selected it, empty for the seeded picker. LLMReasoning is
// the model's short rationale, shown by the replay UI to explain the pick.
Source string `json:"source,omitempty"`
LLMReasoning string `json:"llm_reasoning,omitempty"`
// LLMChoice is the 1-based number the model picked from the candidate list;
// LLMChosenAction is the action description it echoed for that number. The
// runner strict-skips when the echo disagrees with the numbered entry, so on
// a recorded action the two always agree — the replay UI shows them to
// confirm the reasoning matched the executed action.
LLMChoice int `json:"llm_choice,omitempty"`
LLMChosenAction string `json:"llm_chosen_action,omitempty"`
}
type BoundsRecord struct {
X int `json:"x"`
Y int `json:"y"`
Width int `json:"width"`
Height int `json:"height"`
}
type PointRecord struct {
X int `json:"x"`
Y int `json:"y"`
}
type Exception struct {
Class string `json:"class"`
Message string `json:"message,omitempty"`
StackTrace string `json:"stack_trace,omitempty"`
UnixMillis int64 `json:"unix_millis,omitempty"`
}
type Meta struct {
Seed int64 `json:"seed"`
SpecPath string `json:"spec_path"`
BundleSHA256 string `json:"bundle_sha256"`
Platform string `json:"platform"`
BundleID string `json:"bundle_id"`
StartedAt time.Time `json:"started_at"`
EndedAt *time.Time `json:"ended_at,omitempty"`
SanderlingVersion string `json:"sanderling_version"`
}
type Writer struct {
directory string
mutex sync.Mutex
file io.WriteCloser
encoder *json.Encoder
}
// NewWriter ensures `directory` exists and opens trace.jsonl for append.
// meta.json is written separately via WriteMeta. Caller must Close.
func NewWriter(directory string) (*Writer, error) {
if err := os.MkdirAll(directory, 0o755); err != nil {
return nil, fmt.Errorf("mkdir: %w", err)
}
file, err := os.OpenFile(
filepath.Join(directory, "trace.jsonl"),
os.O_CREATE|os.O_WRONLY|os.O_APPEND,
0o644,
)
if err != nil {
return nil, fmt.Errorf("open trace.jsonl: %w", err)
}
encoder := json.NewEncoder(file)
return &Writer{directory: directory, file: file, encoder: encoder}, nil
}
func (w *Writer) Directory() string { return w.directory }
func (w *Writer) WriteMeta(meta Meta) error {
body, err := json.MarshalIndent(meta, "", " ")
if err != nil {
return fmt.Errorf("marshal meta: %w", err)
}
return os.WriteFile(filepath.Join(w.directory, "meta.json"), body, 0o644)
}
func (w *Writer) WriteStep(step Step) error {
w.mutex.Lock()
defer w.mutex.Unlock()
if w.file == nil {
return fmt.Errorf("trace: writer is closed")
}
return w.encoder.Encode(step)
}
// WriteScreenshot is lock-free: each call writes a distinct, uniquely-named
// file via os.WriteFile and touches no field of Writer, so concurrent calls
// never contend.
func (w *Writer) WriteScreenshot(stepIndex int, png []byte) error {
return w.writePNG(fmt.Sprintf("step-%05d.png", stepIndex), png)
}
func (w *Writer) writePNG(name string, png []byte) error {
if len(png) == 0 {
return nil
}
directory := filepath.Join(w.directory, "screenshots")
if err := os.MkdirAll(directory, 0o755); err != nil {
return fmt.Errorf("mkdir screenshots: %w", err)
}
return os.WriteFile(filepath.Join(directory, name), png, 0o644)
}
func (w *Writer) Close() error {
w.mutex.Lock()
defer w.mutex.Unlock()
if w.file == nil {
return nil
}
err := w.file.Close()
w.file = nil
return err
}