with no select(), every
+// InputText appended to the last, and a fuzzer typing twice into one field built
+// up text it could never clear (observed on the folio wasm build as
+// "0.0000001" -> "0.0000001\t-1").
+func TestInputText_ReplacesTextInsideAShadowRoot(t *testing.T) {
+ const page = `
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ if err := d.Tap(ctx, 40, 40); err != nil {
+ t.Fatalf("Tap: %v", err)
+ }
+ if err := d.InputText(ctx, "alpha"); err != nil {
+ t.Fatalf("InputText: %v", err)
+ }
+ if err := d.InputText(ctx, "beta"); err != nil {
+ t.Fatalf("InputText: %v", err)
+ }
+
+ var shown string
+ script := `document.getElementById("app").shadowRoot.getElementById("field").textContent`
+ if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &shown)); err != nil {
+ t.Fatalf("read field: %v", err)
+ }
+ if shown != "beta" {
+ t.Errorf("field holds %q, want %q; the second InputText appended instead of replacing", shown, "beta")
+ }
+
+ if err := d.EraseText(ctx, len("beta")); err != nil {
+ t.Fatalf("EraseText: %v", err)
+ }
+ if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &shown)); err != nil {
+ t.Fatalf("read field: %v", err)
+ }
+ if shown != "" {
+ t.Errorf("field holds %q after EraseText, want empty", shown)
+ }
+}
+
+// TestHierarchy_ScreenFallsBackToThePathname pins the route the goja host reads
+// off the dump. Reading location.hash alone reported "/" on every step of a
+// path-routed SPA (react-router's BrowserRouter, which the replay UI itself
+// uses), so every screen looked like the same screen and no route-scoped
+// property or action could tell them apart.
+func TestHierarchy_ScreenFallsBackToThePathname(t *testing.T) {
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(`
app
`))
+ }))
+ defer server.Close()
+
+ for _, testCase := range []struct {
+ name string
+ path string
+ want string
+ }{
+ {"path-routed", "/runs/20260101-120000/steps/7", "/runs/20260101-120000/steps/7"},
+ {"hash wins when present", "/runs/1#/detail", "/detail"},
+ {"root", "/", "/"},
+ } {
+ t.Run(testCase.name, func(t *testing.T) {
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL+testCase.path, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ dump, err := d.Hierarchy(ctx)
+ if err != nil {
+ t.Fatalf("Hierarchy: %v", err)
+ }
+ var root struct {
+ Attributes map[string]string `json:"attributes"`
+ }
+ if err := json.Unmarshal([]byte(dump), &root); err != nil {
+ t.Fatalf("unmarshal hierarchy: %v", err)
+ }
+ if got := root.Attributes["sanderling-screen"]; got != testCase.want {
+ t.Errorf("sanderling-screen: got %q, want %q", got, testCase.want)
+ }
+ })
+ }
+}
+
+// TestWaitForIdle_WaitsForWorkTheActionKickedOff pins the settle the runner
+// relies on between acting and observing. WaitForIdle used to return the moment
+// existed, which is true before the app has reacted at all: measured on
+// the folio wasm build, Compose's accessibility DOM lands ~136 ms after an
+// InputText, so the next step read the pre-action text and typed into a field
+// it believed was still empty.
+func TestWaitForIdle_WaitsForWorkTheActionKickedOff(t *testing.T) {
+ const page = `
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ if err := d.Tap(ctx, 40, 40); err != nil {
+ t.Fatalf("Tap: %v", err)
+ }
+ if err := d.WaitForIdle(ctx, time.Second); err != nil {
+ t.Fatalf("WaitForIdle: %v", err)
+ }
+
+ var shown string
+ script := `document.getElementById("app").shadowRoot.getElementById("out").textContent`
+ if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &shown)); err != nil {
+ t.Fatalf("read: %v", err)
+ }
+ if shown != "settled" {
+ t.Errorf("observed %q; WaitForIdle returned before the tap's own work landed", shown)
+ }
+}
+
+// TestWaitForIdle_ReturnsOnABusyPage is the other half: a page that never stops
+// mutating (an animation, a polling widget) must not hold the step loop open.
+func TestWaitForIdle_ReturnsOnABusyPage(t *testing.T) {
+ const page = `
0
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ start := time.Now()
+ if err := d.WaitForIdle(ctx, time.Second); err != nil {
+ t.Fatalf("WaitForIdle: %v", err)
+ }
+ if elapsed := time.Since(start); elapsed > 2*time.Second {
+ t.Errorf("WaitForIdle took %s on a busy page; it must return inside its budget", elapsed)
+ }
+}
+
+// TestWaitForIdle_WaitsOutARouteTransition covers the settle case a mutation
+// observer cannot see. A canvas app splices the incoming screen's
+// accessibility nodes in when its cross-fade STARTS and drops the outgoing
+// screen's when it ends; between those two mutations the DOM is quiet with both
+// routes live. Returning there hands the next step a tree naming the screen the
+// app is leaving, which on the folio wasm build recorded a submit that had
+// landed on Home as still being on the transaction screen: the route gate of an
+// action-gated property then skipped the very step the action landed on.
+//
+// The page keeps mutating for 300 ms after the route splice, and the settle
+// runs on the timeout production hands it (MinIdleTimeout). Both details are
+// load-bearing. A quiet page reaches the transition check immediately, so it
+// passes whether the transition window is anchored at the script start or at
+// the end of the quiet period; churn is what pushes the check past a
+// start-anchored deadline, which then finishes at once with two live screens.
+// And a caller timeout below MinIdleTimeout cuts the whole settle off before
+// the transition window can be spent, which is the same bug from the other end.
+func TestWaitForIdle_WaitsOutARouteTransition(t *testing.T) {
+ const page = `
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ if err := d.Tap(ctx, 40, 40); err != nil {
+ t.Fatalf("Tap: %v", err)
+ }
+ if err := d.WaitForIdle(ctx, d.MinIdleTimeout()); err != nil {
+ t.Fatalf("WaitForIdle: %v", err)
+ }
+
+ var live []string
+ script := `Array.from(document.getElementById("app").shadowRoot
+ .querySelectorAll('[id$="Screen"]')).map(e => e.id)`
+ if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &live)); err != nil {
+ t.Fatalf("read: %v", err)
+ }
+ if len(live) != 1 || live[0] != "HomeScreen" {
+ t.Errorf("live screens after the settle = %v, want [HomeScreen]; WaitForIdle "+
+ "returned mid-transition, so the next step verifies the outgoing route", live)
+ }
+}
+
+// TestWaitForIdle_BoundsTheTransitionWait is the other half of the transition
+// wait: a page that shows two *Screen ids at rest is not mid-transition, it
+// just matches the heuristic, and it must cost one bounded wait rather than the
+// whole step budget on every step.
+func TestWaitForIdle_BoundsTheTransitionWait(t *testing.T) {
+ const page = `
home
ledger
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ start := time.Now()
+ if err := d.WaitForIdle(ctx, 10*time.Second); err != nil {
+ t.Fatalf("WaitForIdle: %v", err)
+ }
+ if elapsed := time.Since(start); elapsed > transitionSettlePeriod+time.Second {
+ t.Errorf("WaitForIdle took %s on a page with two resting screens; the "+
+ "transition wait must be bounded by %s", elapsed, transitionSettlePeriod)
+ }
+}
+
+// TestEvaluateExtractors_WaitsOutARouteTransition covers the other sampler. A
+// step reads the page twice: the hierarchy dump (which re-fetches while the
+// tree looks transitional) and the spec's own extractors in V8. Sampling the
+// extractors mid cross-fade reports the route the app is leaving, and an
+// action-gated property then skips the one step its action can be judged on:
+// on the folio wasm build a double-submit that landed on Home was extracted as
+// still being on the transaction screen.
+func TestEvaluateExtractors_WaitsOutARouteTransition(t *testing.T) {
+ const page = `
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(`window.startTransition()`, nil)); err != nil {
+ t.Fatalf("start transition: %v", err)
+ }
+
+ values, err := d.EvaluateExtractors(ctx)
+ if err != nil {
+ t.Fatalf("EvaluateExtractors: %v", err)
+ }
+ if got := string(values[0]); got != `"HomeScreen"` {
+ t.Errorf("extractor read %s, want \"HomeScreen\"; the extractors sampled "+
+ "mid-transition, so the spec sees the route the app is leaving", got)
+ }
+}
+
+// TestSetLastAction_ReportsAPageThatCannotTakeIt covers the install the whole
+// web path's action-gated properties hang off. A page without the setter is
+// reachable: internal/testrun resolves the web runtime from
+// node_modules/@sanderling/spec when no sibling checkout is present, and an
+// older published runtime does not define it. Guarded as
+// `setter && setter(...)`, that page returns undefined and chromedp reports
+// success, so every step silently no-ops and every property gated on the last
+// action goes vacuously true - a green run that checked nothing.
+func TestSetLastAction_ReportsAPageThatCannotTakeIt(t *testing.T) {
+ const withSetter = ``
+ const withoutSetter = `
no sanderling runtime here
`
+ pages := map[string]string{"/with": withSetter, "/without": withoutSetter}
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(pages[r.URL.Path]))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+
+ if err := d.Launch(ctx, server.URL+"/with", false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ action := json.RawMessage(`{"kind":"Tap","on":"id:TxnSubmit"}`)
+ if err := d.SetLastAction(ctx, action); err != nil {
+ t.Fatalf("SetLastAction on a page that defines the setter: %v", err)
+ }
+ var seen map[string]string
+ if err := chromedp.Run(d.tabCtx,
+ chromedp.Evaluate(`window.__lastActionSeen`, &seen)); err != nil {
+ t.Fatalf("read installed action: %v", err)
+ }
+ if seen["on"] != "id:TxnSubmit" {
+ t.Errorf("the page received %v, want the action the runner applied", seen)
+ }
+
+ if err := d.Launch(ctx, server.URL+"/without", false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ if err := d.SetLastAction(ctx, action); err == nil {
+ t.Error("SetLastAction reported success on a page with no setter; " +
+ "a runtime that cannot take lastAction is indistinguishable from one that did")
+ }
+}
+
+// TestEvaluateExtractors_ReportsAMissingTable is the same failure on the other
+// sampler. An empty override map is what a spec with no extractors returns, so
+// treating a missing table as {} makes "this page has no sanderling runtime"
+// read as an ordinary step - and the verifier then judges the run on goja's
+// dump-derived values while believing they came from the page.
+func TestEvaluateExtractors_ReportsAMissingTable(t *testing.T) {
+ const page = `
no sanderling runtime here
`
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+ values, err := d.EvaluateExtractors(ctx)
+ if err == nil {
+ t.Errorf("EvaluateExtractors returned %v and no error on a page with no "+
+ "extractor table; a page that cannot be read must not read as empty", values)
+ }
+}
+
+// TestEvaluateExtractors_KeepsUndefinedApartFromNull covers the wire the page's
+// readings cross. JSON has no undefined, so the web runtime wraps each reading
+// in a {value} envelope: written straight into the table, an extractor that
+// returned undefined lost its whole index to JSON.stringify and the host kept
+// goja's dump-derived reading for it while the rest held the page's. An absent
+// value has to arrive as an empty payload, which is what makes the verifier
+// record undefined (the value the native host records for the same getter);
+// arriving as JSON null would claim the getter returned null.
+func TestEvaluateExtractors_KeepsUndefinedApartFromNull(t *testing.T) {
+ const page = ``
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+
+ values, err := d.EvaluateExtractors(ctx)
+ if err != nil {
+ t.Fatalf("EvaluateExtractors: %v", err)
+ }
+ if len(values) != 3 {
+ t.Fatalf("the page reported 3 readings, %d survived the wire: %v", len(values), values)
+ }
+ if got := values[0]; len(got) != 0 {
+ t.Errorf("the undefined reading arrived as %s, want an empty payload; "+
+ "the verifier records anything else as a value the getter never returned", got)
+ }
+ if got := string(values[1]); got != "null" {
+ t.Errorf("the null reading arrived as %s, want null", got)
+ }
+ if got := string(values[2]); got != `{"balance":7}` {
+ t.Errorf("the object reading arrived as %s, want {\"balance\":7}", got)
+ }
+}
+
+// TestEvaluateExtractors_RejectsAnUnenvelopedReading is the loud failure a page
+// running an older @sanderling/spec produces. Its readings are bare values, and
+// a bare value is indistinguishable from a reading whose getter returned that
+// value, so accepting them silently puts the two engines on different bundles.
+func TestEvaluateExtractors_RejectsAnUnenvelopedReading(t *testing.T) {
+ const page = ``
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ _, _ = w.Write([]byte(page))
+ }))
+ defer server.Close()
+
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
+
+ values, err := d.EvaluateExtractors(ctx)
+ if err == nil {
+ t.Fatalf("EvaluateExtractors accepted %v from a page whose readings are not "+
+ "enveloped; the page and the host are running different bundles", values)
+ }
+ if !strings.Contains(err.Error(), "different bundles") {
+ t.Errorf("EvaluateExtractors failed with %q, want it to name the bundle mismatch", err)
+ }
+}
diff --git a/internal/driver/chrome/fact_parity_test.go b/internal/driver/chrome/fact_parity_test.go
index c680e97..c4021be 100644
--- a/internal/driver/chrome/fact_parity_test.go
+++ b/internal/driver/chrome/fact_parity_test.go
@@ -60,30 +60,41 @@ type factRow struct {
facts elementFacts
}
+// parityPages are the pages both producers are compared on. Each one must
+// exercise every fact both ways on its own (requireBothPolarities), so adding a
+// page never weakens the comparison. fact-parity-shadow.html is shaped like a
+// Compose for Web app: the whole UI lives inside a shadow root, which both
+// producers have to descend into or they enumerate one node for an entire app.
+var parityPages = []string{"fact-parity.html", "fact-parity-shadow.html"}
+
func TestHierarchy_DerivesTheSameFactsAsTheWebRuntime(t *testing.T) {
server := httptest.NewServer(http.FileServer(http.Dir("testdata")))
defer server.Close()
- d := New()
- defer d.Terminate(context.Background())
- ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
- defer cancel()
- if err := d.Launch(ctx, server.URL+"/fact-parity.html", false, nil); err != nil {
- t.Fatalf("Launch: %v", err)
- }
+ for _, page := range parityPages {
+ t.Run(page, func(t *testing.T) {
+ d := New()
+ defer d.Terminate(context.Background())
+ ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
+ defer cancel()
+ if err := d.Launch(ctx, server.URL+"/"+page, false, nil); err != nil {
+ t.Fatalf("Launch: %v", err)
+ }
- dump, err := d.Hierarchy(ctx)
- if err != nil {
- t.Fatalf("Hierarchy: %v", err)
- }
- fromDump := factsFromHierarchyDump(t, dump)
- fromWebRuntime := factsFromWebRuntime(ctx, t, d)
+ dump, err := d.Hierarchy(ctx)
+ if err != nil {
+ t.Fatalf("Hierarchy: %v", err)
+ }
+ fromDump := factsFromHierarchyDump(t, dump)
+ fromWebRuntime := factsFromWebRuntime(ctx, t, d)
- requireEveryElementNamed(t, "the hierarchy dump", fromDump)
- requireEveryElementNamed(t, "the web runtime", fromWebRuntime)
- requireBothPolarities(t, fromWebRuntime)
- compareEnumeratedElements(t, fromDump, fromWebRuntime)
- compareDerivedFacts(t, fromDump, fromWebRuntime)
+ requireEveryElementNamed(t, "the hierarchy dump", fromDump)
+ requireEveryElementNamed(t, "the web runtime", fromWebRuntime)
+ requireBothPolarities(t, fromWebRuntime)
+ compareEnumeratedElements(t, fromDump, fromWebRuntime)
+ compareDerivedFacts(t, fromDump, fromWebRuntime)
+ })
+ }
}
// factsFromHierarchyDump reads the dump the way the goja host does: parse it,
diff --git a/internal/driver/chrome/testdata/fact-parity-shadow.html b/internal/driver/chrome/testdata/fact-parity-shadow.html
new file mode 100644
index 0000000..9529c50
--- /dev/null
+++ b/internal/driver/chrome/testdata/fact-parity-shadow.html
@@ -0,0 +1,34 @@
+
+
+
+
+
fact parity: shadow dom
+
+
+
+
+
light child, no slot to render into
+
+
+
+
diff --git a/internal/driver/chrome/testdata/fact-parity.html b/internal/driver/chrome/testdata/fact-parity.html
index dba230a..c378ec9 100644
--- a/internal/driver/chrome/testdata/fact-parity.html
+++ b/internal/driver/chrome/testdata/fact-parity.html
@@ -56,6 +56,23 @@
bio
+
+
link
+
checkbox
+
radio
+
switch
+
tab
+
+
+
+
tree item
+
+ option
+ disabled option
+
attribute click
delegating root
plain
diff --git a/internal/driver/ioscompanion/device.go b/internal/driver/ioscompanion/device.go
index 9dd7766..67536a7 100644
--- a/internal/driver/ioscompanion/device.go
+++ b/internal/driver/ioscompanion/device.go
@@ -109,6 +109,13 @@ func NewDevice(ctx context.Context, options DeviceOptions) (*Driver, error) {
d.restart = d.respawnDevice
d.processContext, d.processCancel = context.WithCancel(ctx)
+ lock, err := acquireDeviceLock(d.udid)
+ if err != nil {
+ d.processCancel()
+ return nil, err
+ }
+ d.deviceLock = lock
+
if err := d.bringUpDevice(ctx); err != nil {
d.Close()
return nil, err
diff --git a/internal/driver/ioscompanion/driver.go b/internal/driver/ioscompanion/driver.go
index 373dbbd..78a2ada 100644
--- a/internal/driver/ioscompanion/driver.go
+++ b/internal/driver/ioscompanion/driver.go
@@ -42,6 +42,17 @@ const runnerStartupTimeout = 120 * time.Second
// before it is killed. A variable so the kill-escalation test can shrink it.
var shutdownGrace = 15 * time.Second
+// launchTimeout bounds a single app lifecycle RPC. The runner serves lifecycle
+// inside its XCTest session, and a launch the simulator rejects sends that
+// session down a recovery chain (a 120s accessibility wait, a spindump, then an
+// idle wait) that answers minutes late or never. Callers reach Launch with an
+// undeadlined context, since it runs before the run's duration clock starts, so
+// the bound has to come from here or a wedged session hangs the run with no
+// trace, no error, and no end. Kept under runnerStartupTimeout: launching an
+// app inside a live session must cost less than cold-starting that session.
+// A variable so the timeout test can shrink it.
+var launchTimeout = 90 * time.Second
+
// longPressHoldMilliseconds is how long LongPress holds the finger down.
const longPressHoldMilliseconds = 600
@@ -141,6 +152,34 @@ type Driver struct {
// the moment startup finishes.
processContext context.Context
processCancel context.CancelFunc
+
+ // deviceLock is the exclusive claim on the target, held for the driver's
+ // whole life and released by Close.
+ deviceLock io.Closer
+}
+
+// acquireDeviceLock takes an exclusive advisory lock on the target so only one
+// run drives it at a time. Two runs on one device interleave app lifecycle: the
+// second run's uninstall and reinstall land under the first's live automation
+// session, leaving its app proxies bound to a bundle the simulator no longer
+// knows, and every later snapshot and launch on that session stalls. Failing
+// fast beats recovering silently, since the other run owns the device and would
+// be corrupted either way. The lock lives on the file descriptor, so a crashed
+// run's claim is released by the kernel and never strands the device.
+func acquireDeviceLock(udid string) (io.Closer, error) {
+ path := filepath.Join(os.TempDir(), "sanderling-ios-"+udid+".lock")
+ file, err := os.OpenFile(path, os.O_CREATE|os.O_RDWR, 0o644)
+ if err != nil {
+ return nil, fmt.Errorf("open device lock %s: %w", path, err)
+ }
+ if err := syscall.Flock(int(file.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil {
+ file.Close()
+ return nil, fmt.Errorf(
+ "ios target %s is already driven by another sanderling run (lock %s); "+
+ "wait for that run to finish or point this one at a different device with --ios-device",
+ udid, path)
+ }
+ return file, nil
}
// New extracts the embedded companion, spawns it against the configured
@@ -199,7 +238,15 @@ func New(ctx context.Context, options Options) (*Driver, error) {
driverInstance.grantPaste = driverInstance.grantPasteboardAccess
driverInstance.processContext, driverInstance.processCancel = context.WithCancel(ctx)
+ lock, err := acquireDeviceLock(driverInstance.udid)
+ if err != nil {
+ driverInstance.processCancel()
+ return nil, err
+ }
+ driverInstance.deviceLock = lock
+
if err := driverInstance.bringUp(ctx); err != nil {
+ driverInstance.Close()
return nil, err
}
if driverInstance.hybrid {
@@ -408,7 +455,9 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e
// Terminate first so the launch is a clean cold start regardless of the
// app's prior state. A not-running app is not an error here.
- _ = d.withRecovery(ctx, func() error { return d.lifecycleCompanion().Terminate(ctx, d.bundleID) })
+ _ = d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error {
+ return companion.Terminate(callCtx, d.bundleID)
+ })
if clearState {
if err := d.clearAppState(ctx); err != nil {
@@ -428,14 +477,26 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e
}
}
- if err := d.withRecovery(ctx, func() error {
- return d.lifecycleCompanion().Launch(ctx, d.bundleID, true)
+ if err := d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error {
+ return companion.Launch(callCtx, d.bundleID, true)
}); err != nil {
return fmt.Errorf("launch %s: %w", d.bundleID, err)
}
return nil
}
+// lifecycleCall runs an app lifecycle RPC against lifecycleCompanion under a
+// launchTimeout-bounded context, with the usual one-restart recovery. The
+// companion is resolved inside the retry so a restart's replacement client
+// serves the second attempt.
+func (d *Driver) lifecycleCall(ctx context.Context, call func(context.Context, transport.Companion) error) error {
+ boundedCtx, cancel := context.WithTimeout(ctx, launchTimeout)
+ defer cancel()
+ return d.withRecovery(boundedCtx, func() error {
+ return call(boundedCtx, d.lifecycleCompanion())
+ })
+}
+
// lifecycleCompanion is the transport that owns app launch and terminate: the
// in-simulator runner when the hybrid is active, otherwise the legacy
// companion. Lifecycle performed outside the runner's automation session
@@ -528,7 +589,9 @@ func (d *Driver) resetDataContainer(ctx context.Context) error {
}
func (d *Driver) Terminate(ctx context.Context) error {
- return d.withRecovery(ctx, func() error { return d.lifecycleCompanion().Terminate(ctx, d.bundleID) })
+ return d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error {
+ return companion.Terminate(callCtx, d.bundleID)
+ })
}
func (d *Driver) Tap(ctx context.Context, x, y int) error {
@@ -999,6 +1062,10 @@ func (d *Driver) Close() {
if d.processCancel != nil {
d.processCancel()
}
+ if d.deviceLock != nil {
+ _ = d.deviceLock.Close()
+ d.deviceLock = nil
+ }
}
// stopTunnel closes the in-process usbmux forwarder on the device path. Closing
diff --git a/internal/driver/ioscompanion/driver_test.go b/internal/driver/ioscompanion/driver_test.go
index 32ff96e..a70ae6a 100644
--- a/internal/driver/ioscompanion/driver_test.go
+++ b/internal/driver/ioscompanion/driver_test.go
@@ -851,3 +851,175 @@ func (s *sequencedDumpCompanion) AccessibilityInfo(context.Context) (string, err
*s.reads++
return s.dumps[index], nil
}
+
+// wedgedLifecycleCompanion never answers a lifecycle RPC, standing in for a
+// runner whose XCTest session is stuck inside a rejected launch.
+type wedgedLifecycleCompanion struct {
+ fakeCompanion
+ release chan struct{}
+}
+
+func (w *wedgedLifecycleCompanion) block(ctx context.Context) error {
+ select {
+ case <-ctx.Done():
+ return ctx.Err()
+ case <-w.release:
+ return nil
+ }
+}
+
+func (w *wedgedLifecycleCompanion) Launch(ctx context.Context, _ string, _ bool) error {
+ return w.block(ctx)
+}
+
+func (w *wedgedLifecycleCompanion) Terminate(ctx context.Context, _ string) error {
+ return w.block(ctx)
+}
+
+// TestLaunchBoundsWedgedLifecycleRPC proves the launch path carries its own
+// deadline. Callers hand Launch an undeadlined context, so without one a runner
+// that never answers hangs the run forever with nothing printed.
+func TestLaunchBoundsWedgedLifecycleRPC(t *testing.T) {
+ previous := launchTimeout
+ launchTimeout = 100 * time.Millisecond
+ defer func() { launchTimeout = previous }()
+
+ companion := &wedgedLifecycleCompanion{release: make(chan struct{})}
+ defer close(companion.release)
+ d := newTestDriver(companion)
+
+ done := make(chan error, 1)
+ go func() { done <- d.Launch(context.Background(), "com.example.app", false, nil) }()
+
+ select {
+ case err := <-done:
+ if err == nil {
+ t.Fatal("wedged launch returned nil; a stuck runner must surface an error")
+ }
+ if !errors.Is(err, context.DeadlineExceeded) {
+ t.Fatalf("err = %v, want a deadline-exceeded error", err)
+ }
+ case <-time.After(10 * time.Second):
+ t.Fatal("Launch never returned: the lifecycle RPC is unbounded, so a stuck runner hangs the run forever")
+ }
+}
+
+// TestTerminateBoundsWedgedLifecycleRPC covers the same bound on the standalone
+// terminate, which the runner calls mid-run on an equally stuck session.
+func TestTerminateBoundsWedgedLifecycleRPC(t *testing.T) {
+ previous := launchTimeout
+ launchTimeout = 100 * time.Millisecond
+ defer func() { launchTimeout = previous }()
+
+ companion := &wedgedLifecycleCompanion{release: make(chan struct{})}
+ defer close(companion.release)
+ d := newTestDriver(companion)
+
+ done := make(chan error, 1)
+ go func() { done <- d.Terminate(context.Background()) }()
+
+ select {
+ case err := <-done:
+ if !errors.Is(err, context.DeadlineExceeded) {
+ t.Fatalf("err = %v, want a deadline-exceeded error", err)
+ }
+ case <-time.After(10 * time.Second):
+ t.Fatal("Terminate never returned: the lifecycle RPC is unbounded")
+ }
+}
+
+// TestLaunchLeavesATighterCallerDeadlineAlone confirms the bound narrows the
+// caller's context and never widens it, so a caller that wants to give up
+// sooner still does.
+func TestLaunchLeavesATighterCallerDeadlineAlone(t *testing.T) {
+ previous := launchTimeout
+ launchTimeout = 30 * time.Second
+ defer func() { launchTimeout = previous }()
+
+ companion := &wedgedLifecycleCompanion{release: make(chan struct{})}
+ defer close(companion.release)
+ d := newTestDriver(companion)
+
+ ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond)
+ defer cancel()
+ start := time.Now()
+ if err := d.Launch(ctx, "com.example.app", false, nil); !errors.Is(err, context.DeadlineExceeded) {
+ t.Fatalf("err = %v, want a deadline-exceeded error", err)
+ }
+ if elapsed := time.Since(start); elapsed > 5*time.Second {
+ t.Fatalf("Launch took %v; the driver's bound overrode the caller's tighter deadline", elapsed)
+ }
+}
+
+// newLockTestOptions builds New options that dial a seamed companion, so the
+// device-lock tests exercise New without spawning anything.
+func newLockTestOptions(t *testing.T, udid string) Options {
+ t.Helper()
+ listener, err := net.Listen("tcp", "127.0.0.1:0")
+ if err != nil {
+ t.Fatal(err)
+ }
+ t.Cleanup(func() { listener.Close() })
+ go func() {
+ for {
+ connection, acceptErr := listener.Accept()
+ if acceptErr != nil {
+ return
+ }
+ _ = connection.Close()
+ }
+ }()
+ return Options{
+ UniqueDeviceIdentifier: udid,
+ pickAddress: func() (string, error) { return listener.Addr().String(), nil },
+ spawnChild: func(context.Context, string) (*exec.Cmd, error) { return &exec.Cmd{}, nil },
+ dialCompanion: func(string) (transport.Companion, error) {
+ return &fakeCompanion{accessibilityJSON: "[]"}, nil
+ },
+ }
+}
+
+// TestNewRejectsConcurrentRunOnSameDevice proves a second run cannot claim a
+// device the first is driving. Two runs interleave app lifecycle on one
+// simulator: the second's reinstall lands under the first's automation session
+// and wedges it. Failing fast names the contended device; the claim is released
+// on Close so the next run is not locked out.
+func TestNewRejectsConcurrentRunOnSameDevice(t *testing.T) {
+ t.Setenv("SANDERLING_SIMULATOR_COMPANION", "legacy")
+ udid := "LOCK-TEST-" + t.Name()
+
+ first, err := New(context.Background(), newLockTestOptions(t, udid))
+ if err != nil {
+ t.Fatalf("first New: %v", err)
+ }
+
+ _, err = New(context.Background(), newLockTestOptions(t, udid))
+ if err == nil {
+ t.Fatal("second run claimed a device the first still drives; concurrent runs corrupt each other's session")
+ }
+ if !strings.Contains(err.Error(), udid) {
+ t.Fatalf("err = %v, want it to name the contended device %s", err, udid)
+ }
+
+ first.Close()
+ third, err := New(context.Background(), newLockTestOptions(t, udid))
+ if err != nil {
+ t.Fatalf("device stayed locked after Close: %v", err)
+ }
+ third.Close()
+}
+
+// TestAcquireDeviceLockKeepsDistinctDevicesIndependent guards against a lock
+// path that ignores the udid and serializes unrelated runs.
+func TestAcquireDeviceLockKeepsDistinctDevicesIndependent(t *testing.T) {
+ first, err := acquireDeviceLock("LOCK-TEST-DEVICE-A")
+ if err != nil {
+ t.Fatal(err)
+ }
+ defer first.Close()
+ second, err := acquireDeviceLock("LOCK-TEST-DEVICE-B")
+ if err != nil {
+ t.Fatalf("a second device was refused while another was locked: %v", err)
+ }
+ defer second.Close()
+}
diff --git a/internal/driver/ioscompanion/hierarchymap.go b/internal/driver/ioscompanion/hierarchymap.go
index 8b0db9d..3d5f0b5 100644
--- a/internal/driver/ioscompanion/hierarchymap.go
+++ b/internal/driver/ioscompanion/hierarchymap.go
@@ -129,6 +129,9 @@ func mapElement(element *rawElement) (treeNode, bool) {
attributes["hintText"] = label
} else if label != "" {
attributes["accessibilityText"] = label
+ if labelIsDisplayedText(element.Type) {
+ attributes["text"] = label
+ }
}
enabled := element.Enabled
@@ -150,6 +153,22 @@ func isEditable(elementType string) bool {
return elementType == "TextArea" || elementType == "TextField"
}
+// labelIsDisplayedText reports whether an element type's AXLabel is the string
+// drawn on screen rather than an accessibility annotation about it. Only
+// StaticText qualifies: a text element's label IS what it renders, so it
+// belongs in `text`, matching a TextView on Android and a text node on web.
+//
+// Buttons and images are deliberately excluded even though a titled button's
+// label is also its visible title. The snapshot cannot tell that button apart
+// from an icon-only one whose label exists purely for VoiceOver, nor from a
+// container whose label is a comma-joined reading of its children ("CH,
+// Checking, $0.00, 0 transactions"). Inventing `text` for those would put
+// strings in `text` that no user can read, and would diverge from Android,
+// which leaves `text` empty and reports a contentDescription as `description`.
+func labelIsDisplayedText(elementType string) bool {
+ return elementType == "StaticText"
+}
+
func stringValue(pointer *string) string {
if pointer == nil {
return ""
diff --git a/internal/driver/ioscompanion/hierarchymap_test.go b/internal/driver/ioscompanion/hierarchymap_test.go
index 7c488c0..1c7a938 100644
--- a/internal/driver/ioscompanion/hierarchymap_test.go
+++ b/internal/driver/ioscompanion/hierarchymap_test.go
@@ -132,6 +132,79 @@ func TestNonEmptyValueMapsToText(t *testing.T) {
}
}
+func TestStaticTextLabelMapsToText(t *testing.T) {
+ dump := `[{"type":"StaticText","frame":{"x":0,"y":0,"width":50,"height":18},
+ "AXLabel":"$239.00","AXValue":null,"enabled":true}]`
+ element := parseSingle(t, dump)
+ // A StaticText renders its label, so a spec reading .text must see it.
+ if element.Text != "$239.00" {
+ t.Fatalf("text = %q, want $239.00", element.Text)
+ }
+ // The raw label stays available: desc:/label:/content-desc: selectors and
+ // the settle hash read it on the iOS path.
+ if element.Description != "$239.00" {
+ t.Fatalf("description = %q, want $239.00", element.Description)
+ }
+}
+
+func TestNonTextLabelStaysDescriptionOnly(t *testing.T) {
+ // An icon-only button's label is a VoiceOver annotation, and a container's
+ // is a comma-joined reading of its children. Neither is on screen, so
+ // neither may become text; Android reports both as description too.
+ cases := []struct{ elementType, label string }{
+ {"Button", "Log out"},
+ {"Image", "Avatar"},
+ {"Other", "CH, Checking, $0.00, 0 transactions"},
+ }
+ for _, testCase := range cases {
+ dump := `[{"type":"` + testCase.elementType + `","frame":{"x":0,"y":0,"width":10,"height":10},
+ "AXLabel":"` + testCase.label + `","AXValue":null,"enabled":true}]`
+ element := parseSingle(t, dump)
+ if element.Text != "" {
+ t.Errorf("%s text = %q, want empty", testCase.elementType, element.Text)
+ }
+ if element.Description != testCase.label {
+ t.Errorf("%s description = %q, want %q", testCase.elementType, element.Description, testCase.label)
+ }
+ }
+}
+
+// Mirrors the folio Home screen as the companion actually dumps it: each
+// AccountCard is a Button carrying a merged VoiceOver label, with the name and
+// balance as StaticText siblings that the flat tree re-parents by containment.
+// Reading a card's balance is what folio's totalBalance extractor does, and it
+// read nothing on iOS until StaticText labels became text.
+func TestGoldenHomeCardBalancesAreReadableAsText(t *testing.T) {
+ tree := mapAndParse(t, readDump(t, "home-cards-describe.json"), 402, 874)
+
+ cards := tree.FindAllNodes("id:HomeScreen > id:AccountCard")
+ if len(cards) != 2 {
+ t.Fatalf("account cards = %d, want 2", len(cards))
+ }
+ want := []struct{ name, balance string }{{"Checking", "$12.34"}, {"Savings", "$500.00"}}
+ for i, card := range cards {
+ balance := card.Find("id:AccountBalance")
+ if balance == nil {
+ t.Fatalf("card %d: id:AccountBalance did not resolve", i)
+ }
+ if balance.Text != want[i].balance {
+ t.Errorf("card %d: balance text = %q, want %q", i, balance.Text, want[i].balance)
+ }
+ name := card.Find("id:AccountName")
+ if name == nil {
+ t.Fatalf("card %d: id:AccountName did not resolve", i)
+ }
+ if name.Text != want[i].name {
+ t.Errorf("card %d: name text = %q, want %q", i, name.Text, want[i].name)
+ }
+ // The card's own merged label is not visible text; the web fallback in
+ // the folio spec parses cardText and must not see a VoiceOver reading.
+ if card.Element.Text != "" {
+ t.Errorf("card %d: card text = %q, want empty", i, card.Element.Text)
+ }
+ }
+}
+
func TestEditableAndClickableFlags(t *testing.T) {
cases := []struct {
elementType string
diff --git a/internal/driver/ioscompanion/testdata/home-cards-describe.json b/internal/driver/ioscompanion/testdata/home-cards-describe.json
new file mode 100644
index 0000000..ffc342d
--- /dev/null
+++ b/internal/driver/ioscompanion/testdata/home-cards-describe.json
@@ -0,0 +1,16 @@
+[
+ {"type":"Application","frame":{"x":0,"y":0,"width":402,"height":874},"enabled":true,"AXLabel":"Folio","AXValue":null,"AXUniqueId":null},
+ {"type":"Other","frame":{"x":0,"y":62,"width":402,"height":778},"enabled":true,"AXLabel":null,"AXValue":null,"AXUniqueId":"HomeScreen"},
+ {"type":"StaticText","frame":{"x":20,"y":76,"width":97,"height":24},"enabled":true,"AXLabel":"Accounts","AXValue":null,"AXUniqueId":null},
+ {"type":"Button","frame":{"x":340,"y":71,"width":48,"height":48},"enabled":true,"AXLabel":"Log out","AXValue":null,"AXUniqueId":"LogoutButton"},
+ {"type":"Button","frame":{"x":20,"y":130,"width":362,"height":72},"enabled":true,"AXLabel":"CH, Checking, $12.34, 0 transactions","AXValue":null,"AXUniqueId":"AccountCard"},
+ {"type":"StaticText","frame":{"x":90,"y":150,"width":74,"height":18},"enabled":true,"AXLabel":"Checking","AXValue":null,"AXUniqueId":"AccountName"},
+ {"type":"StaticText","frame":{"x":313,"y":157,"width":53,"height":18},"enabled":true,"AXLabel":"$12.34","AXValue":null,"AXUniqueId":"AccountBalance"},
+ {"type":"Button","frame":{"x":20,"y":210,"width":362,"height":72},"enabled":true,"AXLabel":"SA, Savings, $500.00, 2 transactions","AXValue":null,"AXUniqueId":"AccountCard"},
+ {"type":"StaticText","frame":{"x":90,"y":230,"width":66,"height":18},"enabled":true,"AXLabel":"Savings","AXValue":null,"AXUniqueId":"AccountName"},
+ {"type":"StaticText","frame":{"x":306,"y":237,"width":60,"height":18},"enabled":true,"AXLabel":"$500.00","AXValue":null,"AXUniqueId":"AccountBalance"},
+ {"type":"StaticText","frame":{"x":20,"y":716,"width":106,"height":14},"enabled":true,"AXLabel":"TOTAL BALANCE","AXValue":null,"AXUniqueId":null},
+ {"type":"StaticText","frame":{"x":20,"y":731,"width":85,"height":33},"enabled":true,"AXLabel":"$512.34","AXValue":null,"AXUniqueId":null},
+ {"type":"Button","frame":{"x":20,"y":777,"width":362,"height":48},"enabled":true,"AXLabel":"+ Add account","AXValue":null,"AXUniqueId":"AddAccountButton"},
+ {"type":"StaticText","frame":{"x":141,"y":792,"width":121,"height":18},"enabled":true,"AXLabel":"+ Add account","AXValue":null,"AXUniqueId":null}
+]
diff --git a/internal/runner/runner.go b/internal/runner/runner.go
index cee2b6e..950c12c 100644
--- a/internal/runner/runner.go
+++ b/internal/runner/runner.go
@@ -31,6 +31,11 @@ type Options struct {
// positive value stops the loop once that many steps have run.
MaxSteps int
+ // StopOnViolation ends the step loop as soon as a step records a
+ // violation, so a run that exists to find one bug stops at the evidence
+ // instead of spending the rest of its budget past it.
+ StopOnViolation bool
+
BundleID string
Driver driver.DeviceDriver
Verifier *verifier.Verifier
@@ -74,6 +79,7 @@ func Run(ctx context.Context, options Options) (Summary, error) {
if logger == nil {
logger = slog.Default()
}
+ options.IdleTimeout = resolveIdleTimeout(options)
// Gate on the app actually being on top before acting, so the first
// action never fires against a leftover screen or a system dialog. Done
@@ -87,6 +93,7 @@ func Run(ctx context.Context, options Options) (Summary, error) {
if err != nil {
return Summary{}, err
}
+ _, pageExtractors := extractorSource.(webSource)
summary := Summary{StartTime: time.Now()}
deadline := summary.StartTime.Add(options.Duration)
@@ -122,9 +129,8 @@ func Run(ctx context.Context, options Options) (Summary, error) {
var logs []verifier.LogEntry
// gctx is bound to the errgroup so a returned error (or outer
- // cancellation) propagates to siblings - notably the V8 extractor
- // goroutine, whose CDP round-trip can otherwise outrun the step
- // budget on a hung tab.
+ // cancellation) propagates to every sibling read rather than leaving
+ // one blocked on a hung device.
g, gctx := errgroup.WithContext(ctx)
si := stepIndex
// fetchSyncedState issues a single Snapshot RPC so hierarchy and
@@ -143,16 +149,6 @@ func Run(ctx context.Context, options Options) (Summary, error) {
logs = collectLogs(gctx, options.Driver, logSince)
return nil
})
- var v8Overrides map[int]json.RawMessage
- g.Go(func() error {
- overrides, err := extractorSource.ExtractorOverrides(gctx)
- if err != nil {
- logger.Warn("v8 extractor evaluation failed", "step", si, "err", err)
- return nil
- }
- v8Overrides = overrides
- return nil
- })
// All goroutines write to local variables and return nil, so the Wait
// error is always nil; ignored intentionally.
_ = g.Wait()
@@ -195,6 +191,29 @@ func Run(ctx context.Context, options Options) (Summary, error) {
var witnesses map[string]trace.Witness
skippedVerification := false
if !transitional {
+ // The page-side extractors evaluate only on steps the verifier will
+ // accept, which is why this read waits for the tree instead of
+ // racing it. A spec's extractor getters carry state across steps
+ // (folio's last-seen Home total, its submit counters) and that state
+ // advances every time they run: evaluating them on a step whose
+ // values are then thrown away leaves the page one window ahead of
+ // the verifier, so the next accepted pair brackets two committed
+ // transactions while having counted one submit, and the property
+ // convicts a healthy app. It costs the latency the read used to hide
+ // behind the hierarchy fetch; the fetch is what decides whether this
+ // step counts at all, so it has to go first.
+ //
+ // lastAction is the same value PushSnapshot hands the goja state
+ // below: the two engines evaluate this step against one action.
+ v8Overrides, overridesErr := extractorSource.ExtractorOverrides(ctx, lastAction)
+ if overridesErr != nil {
+ // Not a warning. Without the page's values this step's
+ // extractors keep goja's dump-derived readings while the
+ // previous step holds the page's, and a delta property then
+ // compares two producers and fires on an app that did nothing
+ // wrong.
+ return summary, fmt.Errorf("step %d extractor overrides: %w", stepIndex, overridesErr)
+ }
if err := options.Verifier.PushSnapshot(verifier.SnapshotInput{
Tree: tree,
ScreenshotPNG: screenshotPNG,
@@ -206,13 +225,26 @@ func Run(ctx context.Context, options Options) (Summary, error) {
}); err != nil {
return summary, fmt.Errorf("step %d push: %w", stepIndex, err)
}
+ // Every failure below leaves some extractors holding the page's
+ // value and the rest holding goja's reading of the dump, and a
+ // property comparing previous to current across that split fires
+ // on a healthy app. Each also means the two engines loaded
+ // different bundles, which nothing downstream can reconcile.
+ if pageExtractors && len(v8Overrides) != options.Verifier.ExtractorCount() {
+ return summary, fmt.Errorf(
+ "step %d: the page reported values for %d of the spec's %d extractors; "+
+ "the page and the host are running different bundles",
+ stepIndex, len(v8Overrides), options.Verifier.ExtractorCount())
+ }
skipped, overrideErr := options.Verifier.OverrideExtractorValues(v8Overrides)
if overrideErr != nil {
- logger.Warn("v8 override apply failed", "step", stepIndex, "err", overrideErr)
+ return summary, fmt.Errorf("step %d apply extractor overrides: %w", stepIndex, overrideErr)
}
if skipped > 0 {
- logger.Warn("v8 override skipped out-of-range entries",
- "step", stepIndex, "skipped", skipped, "have", len(v8Overrides))
+ return summary, fmt.Errorf(
+ "step %d: %d of %d extractor overrides fell outside the spec's extractor list; "+
+ "the page and the host are running different bundles",
+ stepIndex, skipped, len(v8Overrides))
}
options.Verifier.EvaluateProperties()
violations = options.Verifier.NewlyViolatedProperties()
@@ -317,6 +349,12 @@ func Run(ctx context.Context, options Options) (Summary, error) {
summary.Steps = stepIndex
if len(violations) > 0 {
summary.Violations = append(summary.Violations, violationRecords(violations, witnesses, stepIndex)...)
+ // The step is already written, so the trace ends on the state that
+ // produced the violation. Finalize below still runs, so pending
+ // liveness obligations are reported alongside it.
+ if options.StopOnViolation {
+ break
+ }
}
// Wait actions are themselves a settling: skip the idle poll. Actions
// that mutate the UI fall through to WaitForIdle so the next step's
@@ -391,12 +429,36 @@ func validate(options Options) error {
if options.Duration <= 0 {
return errors.New("runner: Duration must be positive")
}
- if options.IdleTimeout <= 0 {
- options.IdleTimeout = 2 * time.Second
- }
return nil
}
+// defaultIdleTimeout is the settle budget a caller that names none gets.
+const defaultIdleTimeout = 2 * time.Second
+
+// idleTimeoutFloor is a driver that knows how long its own settle can take.
+// Declared here rather than in the driver package (like lastActionInstaller in
+// source.go) so the mobile drivers stay untouched.
+type idleTimeoutFloor interface {
+ MinIdleTimeout() time.Duration
+}
+
+// resolveIdleTimeout settles the per-step settle budget: the caller's value,
+// defaulted when unset, and raised to whatever the driver says its own settle
+// needs. The chrome driver's settle waits for the DOM to go quiet and only then
+// opens its route-transition window; handed less than their sum it is cut off
+// mid-transition, and the step samples the screen the app is leaving. A driver
+// that reports no floor keeps the caller's value exactly.
+func resolveIdleTimeout(options Options) time.Duration {
+ timeout := options.IdleTimeout
+ if timeout <= 0 {
+ timeout = defaultIdleTimeout
+ }
+ if floor, ok := options.Driver.(idleTimeoutFloor); ok {
+ timeout = max(timeout, floor.MinIdleTimeout())
+ }
+ return timeout
+}
+
// ensureForeground keeps the app under test in the foreground. When the driver
// can report the foreground app and it no longer matches the bundle under test,
// the app is relaunched. Returns true when a relaunch happened so the caller
diff --git a/internal/runner/runner_test.go b/internal/runner/runner_test.go
index 38db86b..1e1cd90 100644
--- a/internal/runner/runner_test.go
+++ b/internal/runner/runner_test.go
@@ -2343,3 +2343,99 @@ func TestRunner_SkipsActionWhenOverlayStealsFocusAtApplyTime(t *testing.T) {
t.Errorf("no step recorded action_skipped=%q, so the undispatched action looks executed", actionSkippedForeground)
}
}
+
+// TestRunner_StopOnViolationEndsAtTheFirstViolation pins the gate CI runs on:
+// a step budget of 8 against a spec that only violates on the third step must
+// end on step 3 and write nothing after it, so the trace's last state is the
+// one that produced the violation.
+func TestRunner_StopOnViolationEndsAtTheFirstViolation(t *testing.T) {
+ const thirdStepViolationSpec = `
+import { actions, always, extract } from "@sanderling/spec";
+let observed = 0;
+const tick = extract(() => ++observed);
+globalThis.properties = {
+ staysUnderThree: always(() => tick.current < 3),
+};
+globalThis.actions = actions(() => []);
+`
+ state := newHarnessWithSpec(t, thirdStepViolationSpec)
+
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ summary, err := Run(ctx, Options{
+ Duration: time.Hour,
+ IdleTimeout: 20 * time.Millisecond,
+ MaxSteps: 8,
+ StopOnViolation: true,
+ Driver: state.mock,
+ Verifier: state.verifier,
+ TraceWriter: state.writer,
+ })
+ if err != nil {
+ t.Fatalf("Run: %v", err)
+ }
+ if !containsProperty(summary.Violations, "staysUnderThree") {
+ t.Fatalf("expected staysUnderThree to fire, got %v", summary.Violations)
+ }
+ if summary.Steps != 3 {
+ t.Errorf("steps: got %d, want 3 (the run must stop at the violating step, not run the 8-step budget)",
+ summary.Steps)
+ }
+ for _, step := range traceStepIndices(t, state.writer.Directory()) {
+ if step > summary.Steps {
+ t.Errorf("trace kept stepping after the violation: found step %d past step %d",
+ step, summary.Steps)
+ }
+ }
+}
+
+// TestRunner_WithoutStopOnViolationRunsTheWholeBudget is the other half: the
+// default must stay a full-budget fuzz run, so turning the flag on is the only
+// thing that shortens a run.
+func TestRunner_WithoutStopOnViolationRunsTheWholeBudget(t *testing.T) {
+ state := newHarnessWithSpec(t, violationSpec)
+
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ summary, err := Run(ctx, Options{
+ Duration: time.Hour,
+ IdleTimeout: 20 * time.Millisecond,
+ MaxSteps: 4,
+ Driver: state.mock,
+ Verifier: state.verifier,
+ TraceWriter: state.writer,
+ })
+ if err != nil {
+ t.Fatalf("Run: %v", err)
+ }
+ if summary.Steps != 4 {
+ t.Errorf("steps: got %d, want 4; a violation must not shorten a default run", summary.Steps)
+ }
+}
+
+// traceStepIndices reads every step index the trace recorded, so a test can
+// assert on what the run actually wrote rather than on the summary alone.
+func traceStepIndices(t *testing.T, directory string) []int {
+ t.Helper()
+ file, err := os.Open(filepath.Join(directory, "trace.jsonl"))
+ if err != nil {
+ t.Fatal(err)
+ }
+ defer file.Close()
+ var steps []int
+ scanner := bufio.NewScanner(file)
+ scanner.Buffer(make([]byte, 0, 64*1024), 8*1024*1024)
+ for scanner.Scan() {
+ var line struct {
+ Step int `json:"step"`
+ }
+ if err := json.Unmarshal(scanner.Bytes(), &line); err != nil {
+ t.Fatalf("trace line decode: %v", err)
+ }
+ steps = append(steps, line.Step)
+ }
+ if err := scanner.Err(); err != nil {
+ t.Fatalf("scan trace: %v", err)
+ }
+ return steps
+}
diff --git a/internal/runner/source.go b/internal/runner/source.go
index bb66bac..c57acf3 100644
--- a/internal/runner/source.go
+++ b/internal/runner/source.go
@@ -25,8 +25,23 @@ type ActionSource interface {
// ExtractorSource yields per-step extractor overrides the runner applies after
// PushSnapshot. The mobile path has none (returns nil); the web path returns the
// values its extractors computed in V8 against the real DOM.
+//
+// lastAction is the action the previous step actually applied, the same value
+// PushSnapshot hands the goja state. The web path has to install it in the page
+// before its extractors run: a spec extractor reading state.lastAction runs in
+// V8 there, and V8 has no way to know what the runner dispatched.
type ExtractorSource interface {
- ExtractorOverrides(ctx context.Context) (map[int]json.RawMessage, error)
+ ExtractorOverrides(
+ ctx context.Context,
+ lastAction *verifier.Action,
+ ) (map[int]json.RawMessage, error)
+}
+
+// lastActionInstaller is the web driver's channel for the previous step's
+// action. It is declared here rather than folded into driver.WebDriver so the
+// mobile drivers stay untouched; every web driver must implement it.
+type lastActionInstaller interface {
+ SetLastAction(ctx context.Context, encoded json.RawMessage) error
}
// gojaSource drives both action selection and (trivially) extractor overrides
@@ -40,7 +55,10 @@ func (s gojaSource) NextAction(context.Context, int) (verifier.Action, error) {
return s.verifier.NextAction()
}
-func (gojaSource) ExtractorOverrides(context.Context) (map[int]json.RawMessage, error) {
+func (gojaSource) ExtractorOverrides(
+ context.Context,
+ *verifier.Action,
+) (map[int]json.RawMessage, error) {
return nil, nil
}
@@ -61,7 +79,24 @@ func (s webSource) NextAction(ctx context.Context, _ int) (verifier.Action, erro
return verifier.DecodeAction(raw)
}
-func (s webSource) ExtractorOverrides(ctx context.Context) (map[int]json.RawMessage, error) {
+// ExtractorOverrides installs the previous step's action in the page, then
+// reads back what the spec's extractors computed against the live DOM. The
+// install is not best-effort: a web driver that cannot take it leaves
+// state.lastAction null in V8, which silently turns every action-gated
+// property vacuously true, so it is reported as an error instead.
+func (s webSource) ExtractorOverrides(
+ ctx context.Context,
+ lastAction *verifier.Action,
+) (map[int]json.RawMessage, error) {
+ installer, ok := s.web.(lastActionInstaller)
+ if !ok {
+ return nil, fmt.Errorf(
+ "web driver %T cannot install state.lastAction; every property gated "+
+ "on the last action would be vacuously true", s.web)
+ }
+ if err := installer.SetLastAction(ctx, verifier.EncodeLastAction(lastAction)); err != nil {
+ return nil, fmt.Errorf("install last action: %w", err)
+ }
return s.web.EvaluateExtractors(ctx)
}
diff --git a/internal/runner/web_carrier_test.go b/internal/runner/web_carrier_test.go
new file mode 100644
index 0000000..d4776a5
--- /dev/null
+++ b/internal/runner/web_carrier_test.go
@@ -0,0 +1,196 @@
+package runner
+
+import (
+ "bytes"
+ "context"
+ "encoding/json"
+ "errors"
+ "fmt"
+ "os"
+ "path/filepath"
+ "strconv"
+ "testing"
+ "time"
+
+ "github.com/priyanshujain/sanderling/internal/driver"
+ mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock"
+ "github.com/priyanshujain/sanderling/internal/trace"
+)
+
+// carrierSpec registers one extractor whose value the page supplies. It stands
+// in for every spec whose getters carry state across steps (folio's last-seen
+// Home total, its submit counters): what matters is that ASKING the page for
+// the value is what advances it.
+const carrierSpec = `
+import { actions, extract } from "@sanderling/spec";
+const carrier = extract("carrier", () => 0);
+globalThis.properties = {};
+globalThis.actions = actions(() => []);
+`
+
+// carrierWebDriver is a web target that alternates between a cross-fading
+// hierarchy (which the runner discards as transitional) and a settled one, and
+// whose page-side extractor advances a counter on every evaluation - exactly
+// what a spec-authored carrier does in V8.
+type carrierWebDriver struct {
+ *mockdriver.Driver
+ transitional bool
+ snapshots int
+ reads int
+}
+
+func (d *carrierWebDriver) Snapshot(ctx context.Context) (string, driver.Image, error) {
+ _, image, err := d.Driver.Snapshot(ctx)
+ d.snapshots++
+ if !d.transitional {
+ return `{"attributes":{"resource-id":"HomeScreen"},"children":[]}`, image, err
+ }
+ // A genuine cross-fade: two live routes, and a tree that keeps changing
+ // between retries so the runner spends its whole retry budget on it.
+ return fmt.Sprintf(`{"attributes":{"resource-id":"root"},"children":[
+ {"attributes":{"resource-id":"HomeScreen","text":"frame-%d"},"children":[]},
+ {"attributes":{"resource-id":"LedgerScreen"},"children":[]}
+ ]}`, d.snapshots), image, err
+}
+
+func (d *carrierWebDriver) InstallBundle(context.Context, []byte) error { return nil }
+
+func (d *carrierWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) {
+ d.reads++
+ return map[int]json.RawMessage{0: json.RawMessage(strconv.Itoa(d.reads))}, nil
+}
+
+// NextActionFromV8 runs once per step, after the hierarchy fetch, so flipping
+// here makes every other step a cross-fade.
+func (d *carrierWebDriver) NextActionFromV8(context.Context) (json.RawMessage, error) {
+ d.transitional = !d.transitional
+ return json.RawMessage(`{"kind":"Tap","x":5,"y":5}`), nil
+}
+
+func (d *carrierWebDriver) SetLastAction(context.Context, json.RawMessage) error { return nil }
+
+// TestRunner_TransitionalStepNeverAdvancesThePageCarrier pins the ordering the
+// web path depends on. The page-side extractors must run only on steps the
+// verifier accepts: their getters advance spec state every time they evaluate,
+// so evaluating them on a step whose values are then discarded leaves the page
+// one window ahead of the verifier. The next accepted pair then brackets two
+// committed transactions while having counted one submit, and the property
+// convicts an app that did nothing wrong.
+func TestRunner_TransitionalStepNeverAdvancesThePageCarrier(t *testing.T) {
+ state := newHarnessWithSpec(t, carrierSpec)
+ web := &carrierWebDriver{Driver: state.mock}
+
+ ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
+ defer cancel()
+ summary, err := Run(ctx, Options{
+ Duration: 30 * time.Second,
+ IdleTimeout: 20 * time.Millisecond,
+ MaxSteps: 5,
+ Driver: web,
+ Verifier: state.verifier,
+ TraceWriter: state.writer,
+ })
+ if err != nil {
+ t.Fatalf("Run: %v", err)
+ }
+ if summary.Steps != 5 {
+ t.Fatalf("steps = %d, want 5", summary.Steps)
+ }
+
+ type traceLine struct {
+ Step int `json:"step"`
+ Transitional bool `json:"transitional"`
+ ExtractorChanges map[string]trace.ExtractorChange `json:"extractor_changes"`
+ }
+ body, err := os.ReadFile(filepath.Join(state.writer.Directory(), "trace.jsonl"))
+ if err != nil {
+ t.Fatal(err)
+ }
+ verified, transitional := 0, 0
+ previous := 0
+ for _, raw := range bytes.Split(bytes.TrimSpace(body), []byte("\n")) {
+ var line traceLine
+ if err := json.Unmarshal(raw, &line); err != nil {
+ t.Fatalf("decode trace line: %v", err)
+ }
+ if line.Transitional {
+ transitional++
+ continue
+ }
+ verified++
+ change, ok := line.ExtractorChanges["carrier"]
+ if !ok {
+ t.Fatalf("step %d: no carrier value reached the verifier", line.Step)
+ }
+ current, convErr := strconv.Atoi(string(change.Curr))
+ if convErr != nil {
+ t.Fatalf("step %d: carrier value %s: %v", line.Step, change.Curr, convErr)
+ }
+ if current != previous+1 {
+ t.Errorf("step %d: carrier went %d -> %d; the page advanced it on a "+
+ "step the verifier discarded, so the verifier's window is wider "+
+ "than the one the spec counted actions over",
+ line.Step, previous, current)
+ }
+ previous = current
+ }
+ if verified == 0 || transitional == 0 {
+ t.Fatalf("need both kinds of step to prove anything: %d verified, %d transitional",
+ verified, transitional)
+ }
+ if web.reads != verified {
+ t.Errorf("the page evaluated its extractors %d time(s) across %d verified step(s); "+
+ "every evaluation the verifier does not use still advances spec state",
+ web.reads, verified)
+ }
+}
+
+// installFailsWebDriver is a web target whose page cannot take the runner's
+// lastAction: an older published @sanderling/spec runtime, a bundle that never
+// installed, a tab that navigated away from it.
+type installFailsWebDriver struct {
+ *mockdriver.Driver
+}
+
+func (d *installFailsWebDriver) InstallBundle(context.Context, []byte) error { return nil }
+
+func (d *installFailsWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) {
+ return map[int]json.RawMessage{0: json.RawMessage(`1`)}, nil
+}
+
+func (d *installFailsWebDriver) NextActionFromV8(context.Context) (json.RawMessage, error) {
+ return json.RawMessage(`{"kind":"Tap","x":5,"y":5}`), nil
+}
+
+func (d *installFailsWebDriver) SetLastAction(context.Context, json.RawMessage) error {
+ return errors.New("__sanderlingSetLastAction__ is not a function")
+}
+
+// TestRunner_LastActionInstallFailureFailsTheRun covers the other half of the
+// same trust boundary. A run that cannot install lastAction in the page cannot
+// apply the page's extractor values either, so the step keeps goja's
+// dump-derived readings while the step before it holds the page's, and a delta
+// property compares two producers and fires. Downgraded to a warning that is a
+// green run reporting a violation nobody can reproduce.
+func TestRunner_LastActionInstallFailureFailsTheRun(t *testing.T) {
+ state := newHarnessWithSpec(t, carrierSpec)
+ web := &installFailsWebDriver{Driver: state.mock}
+
+ ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
+ defer cancel()
+ _, err := Run(ctx, Options{
+ Duration: 2 * time.Second,
+ IdleTimeout: 20 * time.Millisecond,
+ MaxSteps: 3,
+ Driver: web,
+ Verifier: state.verifier,
+ TraceWriter: state.writer,
+ })
+ if err == nil {
+ t.Fatal("Run succeeded with a page that cannot take lastAction; the run " +
+ "reported green while its extractor values came from two engines")
+ }
+ if !bytes.Contains([]byte(err.Error()), []byte("install last action")) {
+ t.Errorf("Run error = %v, want it to name the failed lastAction install", err)
+ }
+}
diff --git a/internal/runner/web_extractor_trace_test.go b/internal/runner/web_extractor_trace_test.go
index ae9a33e..b31733e 100644
--- a/internal/runner/web_extractor_trace_test.go
+++ b/internal/runner/web_extractor_trace_test.go
@@ -6,6 +6,7 @@ import (
"encoding/json"
"os"
"path/filepath"
+ "strings"
"testing"
"time"
@@ -44,6 +45,8 @@ func (d *webMockDriver) NextActionFromV8(context.Context) (json.RawMessage, erro
return nil, nil
}
+func (d *webMockDriver) SetLastAction(context.Context, json.RawMessage) error { return nil }
+
// TestRunner_TraceRecordsTheValueTheVerdictUsed fails if the trace and the
// verdict disagree about an extractor. A witness is only an explanation of a
// violation if it holds the state the violated property was evaluated against.
@@ -116,3 +119,48 @@ func TestRunner_TraceRecordsTheValueTheVerdictUsed(t *testing.T) {
t.Error("no witness reached the trace; nothing was compared")
}
}
+
+// splitTableSpec registers two extractors whose goja bodies both answer "goja".
+// The page below reports only the first, so index 1 keeps goja's dump-derived
+// reading while index 0 holds the page's.
+const splitTableSpec = `
+import { actions, extract } from "@sanderling/spec";
+extract("first", () => "goja");
+extract("second", () => "goja");
+globalThis.properties = {};
+globalThis.actions = actions(() => []);
+`
+
+// TestRunner_PartialExtractorTableIsFatal pins the failure the runner used to
+// let through. JSON.stringify drops an undefined-valued key, so a page whose
+// extractors are mostly undefined off their own screen reported a table with
+// holes in it, and the run completed with half the extractors reading from V8
+// and half from goja. A delta property spanning that split convicts an app that
+// did nothing wrong, which is worse than a crash: it is a green report of a bug
+// that is not there, or a red one for a bug nobody can reproduce.
+func TestRunner_PartialExtractorTableIsFatal(t *testing.T) {
+ state := newHarnessWithSpec(t, splitTableSpec)
+ web := &webMockDriver{
+ Driver: state.mock,
+ overrides: map[int]json.RawMessage{0: json.RawMessage(`"v8"`)},
+ }
+
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ _, err := Run(ctx, Options{
+ Duration: time.Hour,
+ IdleTimeout: 20 * time.Millisecond,
+ MaxSteps: 2,
+ Driver: web,
+ Verifier: state.verifier,
+ TraceWriter: state.writer,
+ })
+ if err == nil {
+ t.Fatal("the run completed on a page that reported 1 of 2 extractors; " +
+ "the second extractor silently kept goja's value")
+ }
+ const want = "the page reported values for 1 of the spec's 2 extractors"
+ if !strings.Contains(err.Error(), want) {
+ t.Errorf("Run failed with %q, want it to name the split: %q", err, want)
+ }
+}
diff --git a/internal/runner/web_last_action_test.go b/internal/runner/web_last_action_test.go
new file mode 100644
index 0000000..22ea458
--- /dev/null
+++ b/internal/runner/web_last_action_test.go
@@ -0,0 +1,78 @@
+package runner
+
+import (
+ "context"
+ "encoding/json"
+ "testing"
+ "time"
+
+ mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock"
+)
+
+// On web the spec's extractors run in the page, so state.lastAction has to be
+// installed there by the runner. It used to be hardcoded null in
+// pkg/spec/src/web-runtime.ts, which made every property gated on the last
+// action (folio's submitMovesBalanceByTypedAmount, for one) vacuously true on
+// web: no failure, no warning, just a green run that proved nothing.
+
+const lastActionSpec = `
+import { actions } from "@sanderling/spec";
+globalThis.actions = actions(() => []);
+globalThis.properties = {};
+`
+
+// tappingWebDriver is a web target whose V8 picker always taps one named
+// control, so the runner has a real applied action to report on the next step.
+type tappingWebDriver struct {
+ *mockdriver.Driver
+ installed []string
+}
+
+func (d *tappingWebDriver) InstallBundle(context.Context, []byte) error { return nil }
+
+func (d *tappingWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) {
+ return nil, nil
+}
+
+func (d *tappingWebDriver) NextActionFromV8(context.Context) (json.RawMessage, error) {
+ return json.RawMessage(`{"kind":"Tap","x":12,"y":34,"selector":"id:TxnSubmit"}`), nil
+}
+
+func (d *tappingWebDriver) SetLastAction(_ context.Context, encoded json.RawMessage) error {
+ d.installed = append(d.installed, string(encoded))
+ return nil
+}
+
+func TestRunner_WebInstallsLastActionInThePage(t *testing.T) {
+ state := newHarnessWithSpec(t, lastActionSpec)
+ web := &tappingWebDriver{Driver: state.mock}
+
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ if _, err := Run(ctx, Options{
+ Duration: time.Hour,
+ IdleTimeout: 20 * time.Millisecond,
+ MaxSteps: 3,
+ Driver: web,
+ Verifier: state.verifier,
+ TraceWriter: state.writer,
+ }); err != nil {
+ t.Fatalf("Run: %v", err)
+ }
+
+ if len(web.installed) < 2 {
+ t.Fatalf("the page was handed lastAction %d time(s); the web path never installed it",
+ len(web.installed))
+ }
+ // Step 1 has no previous action, exactly as the goja host reports it.
+ if web.installed[0] != "null" {
+ t.Errorf("step 1 installed %s, want null", web.installed[0])
+ }
+ // Every later step carries what the runner actually applied. The shape is
+ // the goja host's (internal/verifier/marshal.go lastActionFields), pinned
+ // against it by TestLastAction_WebJSONMatchesTheGojaObject.
+ const want = `{"kind":"Tap","on":"id:TxnSubmit"}`
+ if web.installed[1] != want {
+ t.Errorf("step 2 installed %s, want %s", web.installed[1], want)
+ }
+}
diff --git a/internal/testrun/testrun.go b/internal/testrun/testrun.go
index 57cc027..8cc53a7 100644
--- a/internal/testrun/testrun.go
+++ b/internal/testrun/testrun.go
@@ -20,6 +20,26 @@ import (
const sidecarStartupTimeout = 30 * time.Second
+// launchTimeout bounds the pre-run app launch. It happens before the runner
+// starts, so --duration does not cover it, and Execute's context is the bare
+// signal-aware root with no deadline of its own: a driver wedged here would
+// hang the run forever having printed nothing and written no trace. Generous
+// enough to sit above every driver's own launch bound (the iOS clear-state path
+// reinstalls the app first) so a driver-level error is what a user usually
+// sees, and this stays the backstop. A variable so the timeout test can shrink
+// it.
+var launchTimeout = 3 * time.Minute
+
+// launchApp starts the app under test under a bounded context.
+func launchApp(ctx context.Context, activeDriver driver.DeviceDriver, options Options) error {
+ launchCtx, cancel := context.WithTimeout(ctx, launchTimeout)
+ defer cancel()
+ if err := activeDriver.Launch(launchCtx, options.BundleID, options.ClearData, nil); err != nil {
+ return fmt.Errorf("launch app: %w", err)
+ }
+ return nil
+}
+
// Options are the parameters for a single test pipeline run.
type Options struct {
Spec string
@@ -38,6 +58,10 @@ type Options struct {
// Arm labels the experiment cell this run belongs to and is recorded in
// meta.json so a directory of runs can be attributed to a cell.
Arm string
+ // ExitOnViolation stops the run at the first violation and reports the
+ // recorded violations as a ViolationsError, so a caller (CI) can tell
+ // "the run found the bug" from "the run finished clean".
+ ExitOnViolation bool
// Generator selects the action picker: "llm" or the default seeded picker.
Generator string
// LabelSource selects how candidates are named to the model picker, and is
@@ -145,8 +169,8 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error {
}
defer cleanup()
- if err := activeDriver.Launch(ctx, options.BundleID, options.ClearData, nil); err != nil {
- return fmt.Errorf("launch app: %w", err)
+ if err := launchApp(ctx, activeDriver, options); err != nil {
+ return err
}
if web, ok := activeDriver.(driver.WebDriver); ok && len(webBundle.JavaScript) > 0 {
@@ -193,16 +217,17 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error {
fmt.Fprintf(stdout, "running for %s (seed=%d)\n", options.Duration, seed)
}
summary, err := runner.Run(ctx, runner.Options{
- Duration: options.Duration,
- MaxSteps: options.MaxSteps,
- IdleTimeout: 1 * time.Second,
- BundleID: options.BundleID,
- Driver: activeDriver,
- Verifier: verifierInstance,
- TraceWriter: traceWriter,
- Logger: newProgressLogger(stdout),
- Generator: options.Generator,
- LabelSource: options.LabelSource,
+ Duration: options.Duration,
+ MaxSteps: options.MaxSteps,
+ IdleTimeout: 1 * time.Second,
+ BundleID: options.BundleID,
+ Driver: activeDriver,
+ Verifier: verifierInstance,
+ TraceWriter: traceWriter,
+ Logger: newProgressLogger(stdout),
+ Generator: options.Generator,
+ LabelSource: options.LabelSource,
+ StopOnViolation: options.ExitOnViolation,
})
terminateCtx, terminateCancel := context.WithTimeout(context.Background(), 5*time.Second)
@@ -215,9 +240,32 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error {
fmt.Fprintf(stdout, "\nelapsed: %s\n", summary.EndTime.Sub(summary.StartTime).Round(time.Millisecond))
runner.RenderSummary(stdout, summary, options.Platform)
+ return runOutcome(options, summary)
+}
+
+// runOutcome turns a finished run into the pipeline's result. Without
+// --exit-on-violation a run that found violations is still a successful run
+// (the summary reports them), which is the behaviour every existing caller
+// depends on.
+func runOutcome(options Options, summary runner.Summary) error {
+ if options.ExitOnViolation && len(summary.Violations) > 0 {
+ return ViolationsError{Count: len(summary.Violations)}
+ }
return nil
}
+// ViolationsError reports a run that recorded violations under
+// --exit-on-violation. It is deliberately distinct from every other error the
+// pipeline returns: those mean the harness broke, this one means the run did
+// its job and found something.
+type ViolationsError struct {
+ Count int
+}
+
+func (e ViolationsError) Error() string {
+ return fmt.Sprintf("%d violation record(s)", e.Count)
+}
+
// bundleInputs holds the pre-driver assembly: alias map, seed, esbuild defines,
// and the resolved spec-API/goja-runtime paths the bundler consumes.
type bundleInputs struct {
diff --git a/internal/testrun/testrun_test.go b/internal/testrun/testrun_test.go
index 6be7777..3674cef 100644
--- a/internal/testrun/testrun_test.go
+++ b/internal/testrun/testrun_test.go
@@ -1,12 +1,16 @@
package testrun
import (
+ "context"
+ "errors"
"os"
"path/filepath"
"strings"
"testing"
"time"
+ "github.com/priyanshujain/sanderling/internal/driver"
+ "github.com/priyanshujain/sanderling/internal/runner"
"github.com/priyanshujain/sanderling/internal/verifier"
)
@@ -254,3 +258,72 @@ func TestBuildRunMeta_OmitsModelWhenSpecDeclaresNoLLMGenerator(t *testing.T) {
t.Errorf("model recorded without a spec-declared llm generator: %q", meta.Model)
}
}
+
+// TestRunOutcome_ReportsViolationsOnlyUnderTheFlag pins the CI contract: the
+// typed error is what makes `sanderling test` exit 2, and it must appear only
+// when the caller asked for it. A run that finds violations without the flag
+// stays a successful run, which is what every existing invocation expects.
+func TestRunOutcome_ReportsViolationsOnlyUnderTheFlag(t *testing.T) {
+ violated := runner.Summary{
+ Steps: 7,
+ Violations: []runner.ViolationRecord{{StepIndex: 3, Properties: []string{"balanceMoves"}}},
+ }
+ clean := runner.Summary{Steps: 7}
+
+ if err := runOutcome(Options{}, violated); err != nil {
+ t.Errorf("without --exit-on-violation a violated run must succeed, got %v", err)
+ }
+ if err := runOutcome(Options{ExitOnViolation: true}, clean); err != nil {
+ t.Errorf("a clean run must succeed under --exit-on-violation, got %v", err)
+ }
+
+ err := runOutcome(Options{ExitOnViolation: true}, violated)
+ var violations ViolationsError
+ if !errors.As(err, &violations) {
+ t.Fatalf("expected a ViolationsError, got %v", err)
+ }
+ if violations.Count != 1 {
+ t.Errorf("count: got %d, want 1", violations.Count)
+ }
+}
+
+// wedgedLaunchDriver never returns from Launch, standing in for a driver whose
+// device-side session is stuck.
+type wedgedLaunchDriver struct {
+ driver.DeviceDriver
+ release chan struct{}
+}
+
+func (w *wedgedLaunchDriver) Launch(ctx context.Context, _ string, _ bool, _ map[string]string) error {
+ select {
+ case <-ctx.Done():
+ return ctx.Err()
+ case <-w.release:
+ return nil
+ }
+}
+
+// TestLaunchAppBoundsWedgedDriver proves the pre-run launch carries a deadline.
+// It runs before the runner starts, so --duration does not cover it and
+// Execute's root context has no deadline: unbounded, a wedged driver hangs the
+// run forever with no trace directory and no error.
+func TestLaunchAppBoundsWedgedDriver(t *testing.T) {
+ previous := launchTimeout
+ launchTimeout = 100 * time.Millisecond
+ defer func() { launchTimeout = previous }()
+
+ wedged := &wedgedLaunchDriver{release: make(chan struct{})}
+ defer close(wedged.release)
+
+ done := make(chan error, 1)
+ go func() { done <- launchApp(context.Background(), wedged, Options{BundleID: "com.example.app"}) }()
+
+ select {
+ case err := <-done:
+ if !errors.Is(err, context.DeadlineExceeded) {
+ t.Fatalf("err = %v, want a deadline-exceeded error", err)
+ }
+ case <-time.After(10 * time.Second):
+ t.Fatal("launchApp never returned: the pre-run launch is unbounded, so a wedged driver hangs the run forever")
+ }
+}
diff --git a/internal/verifier/marshal.go b/internal/verifier/marshal.go
index cb7852a..1ebcbf5 100644
--- a/internal/verifier/marshal.go
+++ b/internal/verifier/marshal.go
@@ -1,6 +1,7 @@
package verifier
import (
+ "bytes"
"encoding/json"
"fmt"
"slices"
@@ -358,49 +359,118 @@ func selectorObjectToString(runtime *goja.Runtime, arg goja.Value) string {
return strings.Join(parts, " ")
}
+// actionField is one property of the lastAction object, in the order the
+// object is built. Value is a string, an int, or a nested []actionField for
+// the from/to points.
+type actionField struct {
+ key string
+ value any
+}
+
+// lastActionFields is the ONE description of the lastAction shape. The goja
+// host turns it into a JS object (lastActionObject); the web host receives the
+// same fields as JSON (EncodeLastAction) and installs them as state.lastAction
+// in the page. Both hosts therefore expose identical field names, casing,
+// presence and order, so a property reading state.lastAction cannot mean one
+// thing on native and another on web.
+func lastActionFields(action *Action) []actionField {
+ point := func(x, y int) []actionField {
+ return []actionField{{key: "x", value: x}, {key: "y", value: y}}
+ }
+ fields := []actionField{{key: "kind", value: string(action.Kind)}}
+ if action.On != "" {
+ fields = append(fields, actionField{key: "on", value: action.On})
+ }
+ if action.Text != "" {
+ fields = append(fields, actionField{key: "text", value: action.Text})
+ }
+ switch action.Kind {
+ case ActionKindSwipe:
+ fields = append(fields,
+ actionField{key: "from", value: point(action.FromX, action.FromY)},
+ actionField{key: "to", value: point(action.ToX, action.ToY)})
+ if action.DurationMillis > 0 {
+ fields = append(fields,
+ actionField{key: "durationMillis", value: action.DurationMillis})
+ }
+ case ActionKindScroll:
+ fields = append(fields,
+ actionField{key: "direction", value: action.Direction},
+ actionField{key: "from", value: point(action.FromX, action.FromY)},
+ actionField{key: "to", value: point(action.ToX, action.ToY)})
+ case ActionKindPressKey:
+ fields = append(fields, actionField{key: "key", value: action.Key})
+ case ActionKindWait:
+ fields = append(fields,
+ actionField{key: "durationMillis", value: action.DurationMillis})
+ }
+ return fields
+}
+
func lastActionObject(runtime *goja.Runtime, action *Action) goja.Value {
if action == nil {
return goja.Null()
}
+ return objectFromFields(runtime, lastActionFields(action))
+}
+
+func objectFromFields(runtime *goja.Runtime, fields []actionField) *goja.Object {
object := runtime.NewObject()
- _ = object.Set("kind", string(action.Kind))
- if action.On != "" {
- _ = object.Set("on", action.On)
- }
- if action.Text != "" {
- _ = object.Set("text", action.Text)
- }
- switch action.Kind {
- case ActionKindSwipe:
- from := runtime.NewObject()
- _ = from.Set("x", action.FromX)
- _ = from.Set("y", action.FromY)
- to := runtime.NewObject()
- _ = to.Set("x", action.ToX)
- _ = to.Set("y", action.ToY)
- _ = object.Set("from", from)
- _ = object.Set("to", to)
- if action.DurationMillis > 0 {
- _ = object.Set("durationMillis", action.DurationMillis)
+ for _, field := range fields {
+ if nested, ok := field.value.([]actionField); ok {
+ _ = object.Set(field.key, objectFromFields(runtime, nested))
+ continue
}
- case ActionKindScroll:
- _ = object.Set("direction", action.Direction)
- from := runtime.NewObject()
- _ = from.Set("x", action.FromX)
- _ = from.Set("y", action.FromY)
- to := runtime.NewObject()
- _ = to.Set("x", action.ToX)
- _ = to.Set("y", action.ToY)
- _ = object.Set("from", from)
- _ = object.Set("to", to)
- case ActionKindPressKey:
- _ = object.Set("key", action.Key)
- case ActionKindWait:
- _ = object.Set("durationMillis", action.DurationMillis)
+ _ = object.Set(field.key, field.value)
}
return object
}
+// EncodeLastAction renders the previous step's action for the web host, which
+// has no Go-side state object to read: the runner pushes this JSON into the
+// page before each extractor evaluation. A nil action encodes as JSON null,
+// the same value the goja host reports on the first step of a run and after a
+// step whose action was never applied.
+func EncodeLastAction(action *Action) json.RawMessage {
+ if action == nil {
+ return json.RawMessage("null")
+ }
+ return encodeFields(lastActionFields(action))
+}
+
+func encodeFields(fields []actionField) json.RawMessage {
+ var buffer bytes.Buffer
+ buffer.WriteByte('{')
+ for index, field := range fields {
+ if index > 0 {
+ buffer.WriteByte(',')
+ }
+ buffer.Write(encodeJSValue(field.key))
+ buffer.WriteByte(':')
+ if nested, ok := field.value.([]actionField); ok {
+ buffer.Write(encodeFields(nested))
+ continue
+ }
+ buffer.Write(encodeJSValue(field.value))
+ }
+ buffer.WriteByte('}')
+ return buffer.Bytes()
+}
+
+// encodeJSValue encodes one value the way JS JSON.stringify would, so the JSON
+// the web host parses is byte-identical to what the goja object stringifies to.
+// Go escapes <, > and & by default, which JSON.stringify does not, and that
+// alone would make the two hosts encode the same selector differently.
+func encodeJSValue(value any) []byte {
+ var buffer bytes.Buffer
+ encoder := json.NewEncoder(&buffer)
+ encoder.SetEscapeHTML(false)
+ if err := encoder.Encode(value); err != nil {
+ return []byte("null")
+ }
+ return bytes.TrimRight(buffer.Bytes(), "\n")
+}
+
func runtimeMillis(stepTime, runStart time.Time) int64 {
if stepTime.IsZero() || runStart.IsZero() {
return 0
diff --git a/internal/verifier/marshal_test.go b/internal/verifier/marshal_test.go
index fa9085c..8e8654d 100644
--- a/internal/verifier/marshal_test.go
+++ b/internal/verifier/marshal_test.go
@@ -154,3 +154,49 @@ func TestLastActionObject_ExposesKindSpecificFields(t *testing.T) {
}
})
}
+
+// TestLastAction_WebJSONMatchesTheGojaObject pins the two hosts to ONE shape.
+// The goja host builds state.lastAction as a JS object; the web host receives
+// EncodeLastAction's JSON and installs the parsed value as state.lastAction in
+// the page. A field this side renames, drops or cases differently would leave a
+// spec reading state.lastAction working on native and silently mismatching on
+// web, which is the failure this whole path exists to prevent. Comparing
+// goja's own JSON.stringify against the encoder is the strongest available
+// statement that the two are the same object.
+func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) {
+ verifier := newVerifier(t)
+ mustLoad(t, verifier, `
+ globalThis.last = __sanderling__.extract(state => JSON.stringify(state.lastAction));
+ `)
+
+ for _, testCase := range []struct {
+ name string
+ action *Action
+ }{
+ {"nil", nil},
+ {"Tap", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", X: 12, Y: 34}},
+ {"TapWithoutSelector", &Action{Kind: ActionKindTap, X: 12, Y: 34}},
+ {"DoubleTap", &Action{Kind: ActionKindDoubleTap, On: `desc:say "hi"
`}},
+ {"InputText", &Action{Kind: ActionKindInputText, On: "id:field", Text: "50"}},
+ {"Swipe", &Action{Kind: ActionKindSwipe, FromX: 1, FromY: 2, ToX: 3, ToY: 4, DurationMillis: 250}},
+ {"SwipeNoDuration", &Action{Kind: ActionKindSwipe, FromX: 1, FromY: 2, ToX: 3, ToY: 4}},
+ {"Scroll", &Action{Kind: ActionKindScroll, Direction: "down", FromX: 5, FromY: 6, ToX: 5, ToY: 1}},
+ {"PressKey", &Action{Kind: ActionKindPressKey, Key: "enter"}},
+ {"Wait", &Action{Kind: ActionKindWait, DurationMillis: 500}},
+ } {
+ t.Run(testCase.name, func(t *testing.T) {
+ if err := verifier.PushSnapshot(SnapshotInput{
+ Snapshots: Snapshots{},
+ LastAction: testCase.action,
+ }); err != nil {
+ t.Fatal(err)
+ }
+ handle := verifier.runtime.GlobalObject().Get("last").ToObject(verifier.runtime)
+ goja := handle.Get("current").String()
+ web := string(EncodeLastAction(testCase.action))
+ if goja != web {
+ t.Errorf("the two hosts disagree on state.lastAction\n goja: %s\n web: %s", goja, web)
+ }
+ })
+ }
+}
diff --git a/internal/verifier/verifier_test.go b/internal/verifier/verifier_test.go
index 4adce87..5b7a6f9 100644
--- a/internal/verifier/verifier_test.go
+++ b/internal/verifier/verifier_test.go
@@ -1382,3 +1382,54 @@ func TestWithPlatform_IOSReachesPicker(t *testing.T) {
t.Errorf("key = %q, want back (native press-key pool)", action.Key)
}
}
+
+// TestOverrideExtractorValues_EmptyPayloadIsUndefined pins the cross-host
+// meaning of a page reading with no value. JSON has no undefined, so the web
+// runtime wraps every reading in a {value} envelope and the chrome driver hands
+// an absent value through as an empty payload. It has to land here as undefined,
+// because that is what PushSnapshot records for a getter that returned undefined
+// on native: decoding it as null instead would make `x.current === undefined`
+// answer one thing on the native host and another on web, for one spec.
+func TestOverrideExtractorValues_EmptyPayloadIsUndefined(t *testing.T) {
+ verifier := newVerifier(t)
+ mustLoad(t, verifier, helloSpec)
+ if err := verifier.PushSnapshot(SnapshotInput{Snapshots: Snapshots{"ledger.balance": json.RawMessage(`42`)}}); err != nil {
+ t.Fatal(err)
+ }
+ if _, err := verifier.OverrideExtractorValues(map[int]json.RawMessage{
+ 0: nil,
+ 1: json.RawMessage(`null`),
+ }); err != nil {
+ t.Fatal(err)
+ }
+
+ for _, want := range []struct {
+ expression string
+ result bool
+ }{
+ {"screen.current === undefined", true},
+ {"screen.current === null", false},
+ {"balance.current === null", true},
+ {"balance.current === undefined", false},
+ } {
+ value, err := verifier.runtime.RunString(want.expression)
+ if err != nil {
+ t.Fatalf("evaluate %s: %v", want.expression, err)
+ }
+ if value.ToBoolean() != want.result {
+ t.Errorf("a spec reading %s gets %v, want %v", want.expression, value, want.result)
+ }
+ }
+}
+
+// TestExtractorCount_ReportsEveryRegisteredExtractor keeps the web path's
+// completeness check honest: it compares the page's reading count against this
+// number, so a count that ignored an extractor would let a partial table
+// through.
+func TestExtractorCount_ReportsEveryRegisteredExtractor(t *testing.T) {
+ verifier := newVerifier(t)
+ mustLoad(t, verifier, helloSpec)
+ if got := verifier.ExtractorCount(); got != 2 {
+ t.Errorf("ExtractorCount() = %d, want 2 (helloSpec registers screen and balance)", got)
+ }
+}
diff --git a/internal/verifier/worker.go b/internal/verifier/worker.go
index 793a279..b559ea7 100644
--- a/internal/verifier/worker.go
+++ b/internal/verifier/worker.go
@@ -411,6 +411,15 @@ func (v *Verifier) ChangedExtractors() map[string]ExtractorChange {
return changes
}
+// ExtractorCount reports how many extractors the spec registered. The web path
+// compares it against the number of readings the page sent: a page reporting
+// fewer leaves the rest holding goja's dump-derived value while the others hold
+// the page's, and a property comparing previous to current across that split
+// fires on a healthy app.
+func (v *Verifier) ExtractorCount() int {
+ return len(v.extractors)
+}
+
// OverrideExtractorValues replaces each extractor's `current` slot with a
// caller-supplied value, keyed by registration index. Used by the web tick
// path so extractor bodies that ran in V8 (against the real DOM) drive the
diff --git a/pkg/spec/src/web-runtime.ts b/pkg/spec/src/web-runtime.ts
index 5e5dcc6..d8d5d64 100644
--- a/pkg/spec/src/web-runtime.ts
+++ b/pkg/spec/src/web-runtime.ts
@@ -282,13 +282,32 @@ function selectorFromString(selector: string): { css?: string; xpath?: string }
return { css: cssPart(kind, value) };
}
+// deepQueryAll resolves a CSS selector against a root AND every shadow root
+// beneath it. querySelectorAll stops dead at a shadow boundary, and a canvas app
+// (Compose for Web mounts its canvas and its whole accessibility tree inside a
+// shadow root on the mount element) keeps its entire UI on the far side of one:
+// without this a spec sees four nodes and can neither enumerate a target nor
+// resolve a testTag. Light-DOM matches come first, then shadow content in walk
+// order. XPath has no equivalent, so `text:` selectors stop at the boundary.
+function deepQueryAll(selector: string, root: ParentNode): Element[] {
+ const found: Element[] = [];
+ const visit = (scope: ParentNode): void => {
+ for (const element of Array.from(scope.querySelectorAll(selector))) found.push(element);
+ for (const element of Array.from(scope.querySelectorAll("*"))) {
+ if (element.shadowRoot) visit(element.shadowRoot);
+ }
+ };
+ visit(root);
+ return found;
+}
+
function queryElement(
root: ParentNode,
selector: unknown,
): Element | null {
if (typeof selector === "string") {
const { css, xpath } = selectorFromString(selector);
- if (css) return root.querySelector(css);
+ if (css) return deepQueryAll(css, root)[0] ?? null;
if (xpath) {
const result = document.evaluate(
xpath,
@@ -313,7 +332,7 @@ function queryElement(
}
if (selector && typeof selector === "object") {
const { css, xpath } = selectorFromObject(selector as Record);
- if (css) return root.querySelector(css);
+ if (css) return deepQueryAll(css, root)[0] ?? null;
if (xpath) {
const result = document.evaluate(
xpath,
@@ -331,13 +350,27 @@ function queryElement(
function queryAllElements(root: ParentNode, selector: unknown): Element[] {
if (typeof selector === "string") {
const { css, xpath } = selectorFromString(selector);
- if (css) return Array.from(root.querySelectorAll(css));
+ if (css) return deepQueryAll(css, root);
if (xpath) return evaluateXPathAll(xpath, root as Node);
return [];
}
+ // A selector path: every match of the first segment is searched for the rest,
+ // concatenated in walk order, mirroring FindAllBySelectorPath in
+ // internal/hierarchy. Falling through to the object branch (as this did)
+ // returned NOTHING for a path on web while native returned matches, so a spec
+ // reading state.ax.findAll([{screen}, {row}]) saw an empty list on web and
+ // every property over it passed by having nothing to check.
+ if (Array.isArray(selector)) {
+ const head = selector[0];
+ if (head === undefined) return [];
+ const heads = queryAllElements(root, head);
+ if (selector.length === 1) return heads;
+ const rest = selector.slice(1);
+ return heads.flatMap((element) => queryAllElements(element, rest));
+ }
if (selector && typeof selector === "object" && !Array.isArray(selector)) {
const { css, xpath } = selectorFromObject(selector as Record);
- if (css) return Array.from(root.querySelectorAll(css));
+ if (css) return deepQueryAll(css, root);
if (xpath) return evaluateXPathAll(xpath, root as Node);
}
return [];
@@ -391,7 +424,47 @@ function fieldHint(element: Element): string {
return element.getAttribute("name") ?? "";
}
-function elementHandle(element: Element): Record {
+// SELECTOR_TAG is the key an ax element carries the selector it was found by,
+// the same key the goja host writes (internal/verifier/bindings.go tagSelector).
+// The shared serializer (runtime-entry.ts pointOf) reads it off an author
+// target, so a spec's Tap({ on: state.ax.find(...) }) reaches the runner naming
+// the control it acted on instead of a bare pair of coordinates. Without it
+// `lastAction.on` is empty on web for exactly the actions a spec authored.
+const SELECTOR_TAG = "__sanderlingSelector";
+
+// selectorTag renders a selector argument in the canonical "k:v" grammar the
+// hierarchy package parses, chains joined by " > ". It mirrors
+// selectorStringFromJS in internal/verifier/marshal.go, so an element found by
+// the same selector is labelled with the SAME string on both hosts.
+function selectorTag(selector: unknown): string {
+ if (typeof selector === "string") return selector;
+ if (Array.isArray(selector)) {
+ return selector
+ .map(selectorTag)
+ .filter((segment) => segment !== "")
+ .join(" > ");
+ }
+ if (selector && typeof selector === "object") {
+ const source = selector as Record;
+ return Object.keys(source)
+ .filter((key) => key !== SELECTOR_TAG && source[key] !== undefined && source[key] !== null)
+ .map((key) => `${key}:${String(source[key])}`)
+ .join(" ");
+ }
+ return "";
+}
+
+// isEnabled answers the `enabled` fact. `.disabled` is a property only real form
+// controls have, so it reads undefined on the role-based controls the tappable
+// set now covers, and every one of them looked enabled however plainly it was
+// marked otherwise. internal/driver/chrome/driver.go answers the same two ways
+// for the dump the goja host reads.
+function isEnabled(element: Element): boolean {
+ if ((element as HTMLButtonElement).disabled) return false;
+ return element.getAttribute("aria-disabled") !== "true";
+}
+
+function elementHandle(element: Element, selector: unknown): Record {
const rect = element.getBoundingClientRect();
const x = Math.round(rect.left + rect.width / 2);
const y = Math.round(rect.top + rect.height / 2);
@@ -416,7 +489,7 @@ function elementHandle(element: Element): Record {
desc: ariaLabel,
class: (element as HTMLElement).className ?? "",
clickable: true,
- enabled: !(element as HTMLButtonElement).disabled,
+ enabled: isEnabled(element),
editable: isEditableElement(element as HTMLElement),
focused: document.activeElement === element,
x,
@@ -429,12 +502,15 @@ function elementHandle(element: Element): Record {
},
attrs,
dataset: datasetCopy,
- find(selector: unknown): unknown {
- const child = queryElement(element, selector);
- return child ? elementHandle(child) : undefined;
+ [SELECTOR_TAG]: selectorTag(selector),
+ find(childSelector: unknown): unknown {
+ const child = queryElement(element, childSelector);
+ return child ? elementHandle(child, childSelector) : undefined;
},
- findAll(selector: unknown): unknown[] {
- return queryAllElements(element, selector).map(elementHandle);
+ findAll(childSelector: unknown): unknown[] {
+ return queryAllElements(element, childSelector).map((child) =>
+ elementHandle(child, childSelector),
+ );
},
};
}
@@ -443,10 +519,12 @@ function buildAx(): unknown {
return {
find(selector: unknown): unknown {
const element = queryElement(document, selector);
- return element ? elementHandle(element) : undefined;
+ return element ? elementHandle(element, selector) : undefined;
},
findAll(selector: unknown): unknown[] {
- return queryAllElements(document, selector).map(elementHandle);
+ return queryAllElements(document, selector).map((element) =>
+ elementHandle(element, selector),
+ );
},
};
}
@@ -483,13 +561,22 @@ if (typeof globalThis.addEventListener === "function") {
});
}
+// lastAction is what the previous step actually did, pushed in by the Go runner
+// (internal/runner, via __sanderlingSetLastAction__) before each extractor
+// evaluation, in the shape internal/verifier/marshal.go builds for goja. The
+// page cannot derive it: only the runner knows whether the action it picked was
+// really applied, and under --generator llm the action is not picked here at
+// all. Hardcoding null here, as this file used to, makes every spec property
+// that reads state.lastAction vacuously true on web.
+let lastAction: unknown = null;
+
function buildState(): unknown {
return {
snapshots: {},
ax: buildAx(),
document,
window,
- lastAction: null,
+ lastAction,
time: 0,
logs: [],
exceptions: capturedExceptions.slice(),
@@ -535,6 +622,11 @@ const runtime = {
// the host invoking the extractor/next-action callbacks.
defineLockedGlobal("__sanderling__", runtime);
+// The host calls this once per step, before __sanderlingExtractors__.
+defineLockedGlobal("__sanderlingSetLastAction__", (value: unknown) => {
+ lastAction = value ?? null;
+});
+
// writable:false stops a page script from shadowing the runtime via plain
// assignment (the realistic in-page threat). configurable:true is required so
// unit tests sharing one process can reinstall a fake via defineProperty; a
@@ -549,9 +641,18 @@ function defineLockedGlobal(name: string, value: unknown): void {
});
}
-function evaluateExtractors(): Record {
+// Each reading is wrapped in a {value} envelope because JSON has no undefined.
+// Written straight into the map, an extractor whose getter returned undefined
+// (folio's on(route, tag) off its own screen, which is most extractors on most
+// steps) had its whole INDEX dropped by JSON.stringify, and the host kept goja's
+// dump-derived reading for it while the rest held the page's. Inside the
+// envelope the same drop means "this getter returned undefined", which is what
+// the goja host records for the same getter; a JSON null would instead claim it
+// returned null, and `x.current === undefined` would answer differently on the
+// two hosts.
+function evaluateExtractors(): Record {
const state = buildState();
- const result: Record = {};
+ const result: Record = {};
for (let i = 0; i < extractors.length; i++) {
const entry = extractors[i];
if (!entry) continue;
@@ -568,7 +669,7 @@ function evaluateExtractors(): Record {
extracting = false;
}
entry.currentValue = value;
- result[i] = sanitize(value);
+ result[i] = { value: sanitize(value) };
}
return result;
}
@@ -608,7 +709,22 @@ function sanitizeAt(value: unknown, depth: number, seen: WeakSet): unkno
// only how the DOM answers "is this clickable" / "is this editable", the two
// facts with no direct DOM equivalent of the accessibility attributes native
// platforms expose.
-const TAPPABLE_SELECTOR = 'a, button, input, select, textarea, [role="button"], [onclick]';
+//
+// TAPPABLE_ROLES are the ARIA roles whose whole contract is that a user
+// activates the element. Covering only role="button" left every other one
+// invisible to the enumeration, however plain the control looked: the replay UI
+// builds its step rows as , and the spec dogfooding it had to
+// hand-write an action to reach them because no default verb could see a single
+// row. internal/driver/chrome/driver.go resolves the same set for the hierarchy
+// dump the goja host reads, and the two are compared element by element by
+// TestHierarchy_DerivesTheSameFactsAsTheWebRuntime.
+const TAPPABLE_ROLES = [
+ "button", "link", "checkbox", "radio", "switch", "tab", "option",
+ "menuitem", "menuitemcheckbox", "menuitemradio", "treeitem",
+];
+const TAPPABLE_SELECTOR = `a, button, input, select, textarea, ${
+ TAPPABLE_ROLES.map((role) => `[role="${role}"]`).join(", ")
+}, [onclick]`;
const EDITABLE_SELECTOR = "input, textarea, [contenteditable]";
const NON_TEXT_INPUT_TYPES = [
@@ -651,12 +767,78 @@ function pointOf(element: Element): Candidate {
const HEAD_SELECTOR = "head, head *";
// targetElements is the walk the target list is built from: the document in
-// pre-order, minus the head subtree.
+// pre-order, minus the head subtree, with each shadow host's content spliced in
+// directly after the host. That is buildTree's order in
+// internal/driver/chrome/driver.go, and the two producers are compared element
+// by element in enumeration order.
function targetElements(): HTMLElement[] {
const inHead = new Set(Array.from(document.querySelectorAll(HEAD_SELECTOR)));
- return Array.from(document.querySelectorAll("*")).filter(
- (element) => !inHead.has(element),
+ const walked: HTMLElement[] = [];
+ expandShadowContent(
+ Array.from(document.querySelectorAll("*")).filter(
+ (element) => !inHead.has(element),
+ ),
+ walked,
);
+ return walked;
+}
+
+// expandShadowContent copies a tree-ordered element list into `into`, following
+// each host into its shadow root (and into nested hosts) as it goes.
+function expandShadowContent(elements: HTMLElement[], into: HTMLElement[]): void {
+ for (const element of elements) {
+ into.push(element);
+ const shadow = element.shadowRoot;
+ if (!shadow) continue;
+ expandShadowContent(Array.from(shadow.querySelectorAll("*")), into);
+ }
+}
+
+// IDENTITY_KEYS is the ladder a target's selector is built from, mirroring
+// selectorForElement in internal/verifier/worker.go: the id first (where
+// Compose for Web lands a testTag), then data-testid, then the description.
+// Every key here is one the goja host's selector grammar already understands,
+// so the runner can re-resolve the target it names. `desc` reads the same three
+// attributes, in the same order, that the hierarchy dump folds into
+// content-desc (internal/driver/chrome/driver.go); reading fewer of them would
+// let a selector this side calls unique resolve to a different element on the
+// Go side, which re-routes the action to whatever the dump matched first.
+const IDENTITY_KEYS: ReadonlyArray string]> = [
+ ["id", (element) => element.id],
+ ["data-testid", (element) => element.dataset.testid ?? ""],
+ [
+ "desc",
+ (element) =>
+ element.getAttribute("aria-label") ||
+ element.getAttribute("alt") ||
+ element.getAttribute("title") ||
+ "",
+ ],
+];
+
+// selectorsFor names each enumerated element, or leaves it unnamed. A value is
+// only used when it occurs ONCE across the enumeration, so an action carrying
+// the selector can never be re-resolved onto a sibling that shares the value
+// (folio's Home screen has many AccountCards under one testTag). Unnamed
+// elements keep the coordinates-only behaviour the web host always had.
+function selectorsFor(elements: readonly HTMLElement[]): Array {
+ const counts = IDENTITY_KEYS.map(() => new Map());
+ for (const element of elements) {
+ IDENTITY_KEYS.forEach(([, read], index) => {
+ const value = read(element);
+ if (!value) return;
+ const seen = counts[index]!;
+ seen.set(value, (seen.get(value) ?? 0) + 1);
+ });
+ }
+ return elements.map((element) => {
+ for (let index = 0; index < IDENTITY_KEYS.length; index++) {
+ const [key, read] = IDENTITY_KEYS[index]!;
+ const value = read(element);
+ if (value && counts[index]!.get(value) === 1) return `${key}:${value}`;
+ }
+ return undefined;
+ });
}
// collectTargets walks the document ONCE and reports every element with the facts
@@ -664,16 +846,17 @@ function targetElements(): HTMLElement[] {
// resolved by selector first so the DOM's answer to "clickable" and "editable"
// stays expressed in CSS, as it always was.
function collectTargets(): TargetElement[] {
- const clickable = new Set(Array.from(document.querySelectorAll(TAPPABLE_SELECTOR)));
+ const clickable = new Set(deepQueryAll(TAPPABLE_SELECTOR, document));
const editable = new Set(
- Array.from(document.querySelectorAll(EDITABLE_SELECTOR)).filter(
- isEditableElement,
- ),
+ (deepQueryAll(EDITABLE_SELECTOR, document) as HTMLElement[]).filter(isEditableElement),
);
- return targetElements().map((element) => ({
+ const elements = targetElements();
+ const selectors = selectorsFor(elements);
+ return elements.map((element, index) => ({
...pointOf(element),
+ selector: selectors[index],
clickable: clickable.has(element),
- enabled: !(element as HTMLButtonElement).disabled,
+ enabled: isEnabled(element),
editable: editable.has(element),
scrollable: isScrollable(element),
}));
@@ -734,6 +917,7 @@ export const __testing__ = {
selectorFromObject,
SELECTOR_KEYS,
unknownSelectorKeyMessage,
+ selectorTag,
xpathStringLiteral,
};
diff --git a/pkg/spec/test/folio-account-card-parse.test.ts b/pkg/spec/test/folio-account-card-parse.test.ts
new file mode 100644
index 0000000..6b046b3
--- /dev/null
+++ b/pkg/spec/test/folio-account-card-parse.test.ts
@@ -0,0 +1,139 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import {
+ cardAccountName,
+ cardBalanceText,
+ parseDollarCents,
+} from "../../../examples/folio/sanderling/predicates.ts";
+
+// Android and iOS expose AccountName/AccountBalance as their own nodes. Web
+// merges the card into one node whose text is initials + name +
+// "N transaction(s)" + balance, with no separator between the parts. Both
+// shapes have to land on the same balance, and the name has to stay usable as
+// an identity key.
+
+const balanceOf = (cardText: string) =>
+ parseDollarCents(cardBalanceText({ childText: undefined, cardText }));
+
+// Renders a card the way HomeScreen.kt does, so a test states the account and
+// lets the fixture do the concatenating.
+const card = (initials: string, name: string, count: number, balance: string) =>
+ `${initials}${name}${count === 1 ? "1 transaction" : `${count} transactions`}${balance}`;
+
+test("structured child wins over the card text", () => {
+ assert.equal(
+ cardBalanceText({ childText: "$118.00", cardText: "SASavings1 transaction$118.00" }),
+ "$118.00",
+ );
+ assert.equal(
+ cardAccountName({ childText: "Savings", cardText: "SASavings1 transaction$118.00" }),
+ "Savings",
+ );
+});
+
+test("merged card text: balance is the amount at the end, not scraped digits", () => {
+ // The naive reading, text.replace(/[^0-9]/g, ""), absorbs the 12 of
+ // "12 transactions" and returns 12258900.
+ assert.equal(balanceOf("INInvestments12 transactions$2,589.00"), 258900);
+});
+
+test("merged card text: singular transaction label", () => {
+ assert.equal(balanceOf("SASavings1 transaction$118.00"), 11800);
+});
+
+test("merged card text: negative balance keeps its sign", () => {
+ assert.equal(cardBalanceText({ childText: undefined, cardText: "TRTravel3 transactions-$45.50" }), "-$45.50");
+ assert.equal(balanceOf("TRTravel3 transactions-$45.50"), -4550);
+});
+
+test("structured child: negative balance keeps its sign", () => {
+ assert.equal(parseDollarCents(cardBalanceText({ childText: "-$45.50", cardText: undefined })), -4550);
+});
+
+test("zero-balance card is 0, not unknown", () => {
+ assert.equal(balanceOf("EFEmergency Fund0 transactions$0.00"), 0);
+ assert.equal(balanceOf("Aa0 transactions$0.00"), 0);
+});
+
+// The trap a lazy balance regex falls into: a name ending in digits runs
+// straight into the transaction count, so only anchoring the amount at the end
+// of the string gets it right.
+test("name ending in digits does not leak into the balance", () => {
+ assert.equal(balanceOf(card("T2", "Travel 2024", 12, "$75.00")), 7500);
+ assert.equal(balanceOf(card("T2", "Travel 2024", 0, "$0.00")), 0);
+ assert.equal(balanceOf(card("20", "2024", 3, "-$1,234.56")), -123456);
+});
+
+// The account key only has to be stable and per-account. newAccountBalanceIsZero
+// reads it as a set member: a key that drifted as an account's transaction
+// count grew would make an existing account look brand new, and the property
+// would fire on it for holding the balance it just earned.
+test("account key is stable as the transaction count and balance move", () => {
+ const cards: [string, string][] = [
+ ["CH", "Checking"],
+ ["T2", "Travel 2024"],
+ ["A2", "Account 2"],
+ ["Aa", "a"],
+ ["?", ""],
+ ["5T", "5 transactions"],
+ ["X9", "x9"],
+ ];
+ for (const [initials, name] of cards) {
+ const keys = new Set();
+ for (let count = 0; count <= 130; count++) {
+ keys.add(cardAccountName({
+ childText: undefined,
+ cardText: card(initials, name, count, `$${count * 7}.50`),
+ }));
+ }
+ assert.equal(keys.size, 1, `key for ${JSON.stringify(name)} drifted: ${[...keys].join(", ")}`);
+ }
+});
+
+test("account keys are distinct across the accounts a run creates", () => {
+ const names: [string, string][] = [
+ ["CH", "Checking"],
+ ["SA", "Savings"],
+ ["TR", "Travel"],
+ ["EF", "Emergency Fund"],
+ ["IN", "Investments"],
+ ["T2", "Travel 2024"],
+ ["A2", "Account 2"],
+ ["A1", "Account 12"],
+ ["Aa", "a"],
+ ];
+ const keys = names.map(([initials, name]) =>
+ cardAccountName({ childText: undefined, cardText: card(initials, name, 4, "$9.00") }));
+ assert.equal(new Set(keys).size, names.length);
+});
+
+// elementHandle in the web runtime truncates node text at 200 characters, so a
+// long account name (the input corpus types 4096 "a"s) pushes the balance off
+// the end of the string. That balance is unknown, and unknown must not read as
+// zero: newAccountBalanceIsZero passes an unknown balance rather than
+// convicting a card it could not read.
+test("card text truncated past the balance reads as unknown, not zero", () => {
+ const cardText = "AA" + "a".repeat(198);
+ assert.equal(cardBalanceText({ childText: undefined, cardText }), undefined);
+ assert.equal(balanceOf(cardText), null);
+});
+
+test("empty and missing text are unknown, not zero", () => {
+ assert.equal(parseDollarCents(undefined), null);
+ assert.equal(parseDollarCents(""), null);
+ assert.equal(parseDollarCents("no digits here"), null);
+ assert.equal(parseDollarCents("$12"), null);
+ assert.equal(cardBalanceText({ childText: undefined, cardText: undefined }), undefined);
+ assert.equal(cardAccountName({ childText: undefined, cardText: undefined }), "");
+});
+
+// The two accessibility shapes have to read the same per-card balance, which is
+// what the accounts extractor compares. The Home total is no longer a sum of
+// these: it is the app's own TOTAL BALANCE node (see folio-total-balance.test.ts).
+test("merged card text and a structured child give the same balance", () => {
+ const merged = ["INInvestments12 transactions$2,589.00", "Aa0 transactions$0.00"].map(cardText =>
+ cardBalanceText({ childText: undefined, cardText }));
+ assert.deepEqual(merged, ["$2,589.00", "$0.00"]);
+ assert.deepEqual(merged.map(parseDollarCents), [258900, 0]);
+});
diff --git a/pkg/spec/test/folio-home-card-readings.test.ts b/pkg/spec/test/folio-home-card-readings.test.ts
new file mode 100644
index 0000000..c76babe
--- /dev/null
+++ b/pkg/spec/test/folio-home-card-readings.test.ts
@@ -0,0 +1,159 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import {
+ committedTransactionsExceedSubmits,
+ countSubmitsInWindow,
+ homeAccountsOf,
+ homeTxnCountsOf,
+ readHomeCards,
+} from "../../../examples/folio/sanderling/predicates.ts";
+import type {
+ CardReading,
+ HomeCardReading,
+ TxnCount,
+} from "../../../examples/folio/sanderling/predicates.ts";
+
+const card = (name: string, balance: number | null, count: TxnCount | undefined) => ({
+ name,
+ balance,
+ count,
+});
+
+test("a laid-out card list reads as an account list and a count map", () => {
+ const cards = [card("Checking", 0, "0"), card("Travel", 2411200, "1")];
+ assert.deepEqual(homeAccountsOf(cards), [
+ { name: "Checking", balance: 0 },
+ { name: "Travel", balance: 2411200 },
+ ]);
+ assert.deepEqual(homeTxnCountsOf(cards), { Checking: "0", Travel: "1" });
+});
+
+// Android draws Home's own node a frame or two before its list, so `findAll`
+// over the cards comes back empty while the screen already claims to be Home.
+// "No cards on screen" is not "no accounts".
+test("Home with nothing laid out yet is unknown, not empty", () => {
+ assert.equal(homeAccountsOf([]), null);
+ assert.equal(homeTxnCountsOf([]), null);
+});
+
+test("a card with no readable name or count is left out of the map", () => {
+ assert.deepEqual(
+ homeTxnCountsOf([card("", 0, "3"), card("Checking", 0, undefined), card("Savings", 0, "2")]),
+ { Savings: "2" },
+ );
+});
+
+test("every card unreadable leaves nothing to compare, which is unknown", () => {
+ assert.equal(homeTxnCountsOf([card("", 0, "3"), card("Checking", 0, undefined)]), null);
+});
+
+test("off Home the carrier is reported unchanged", () => {
+ const carried = { Checking: "3" };
+ assert.deepEqual(readHomeCards({ route: "ledger", reading: null, previousCarrier: carried }), {
+ value: carried,
+ carrier: carried,
+ fresh: false,
+ });
+});
+
+test("a readable list replaces the carrier and closes the window", () => {
+ assert.deepEqual(
+ readHomeCards({ route: "home", reading: { Checking: "5" }, previousCarrier: { Checking: "3" } }),
+ { value: { Checking: "5" }, carrier: { Checking: "5" }, fresh: true },
+ );
+});
+
+// The poisoned carrier, the same defect readHomeTotalBalance was fixed for and
+// this reading was not. An empty reading written into the carrier is handed
+// straight back on every later off-Home step, so one un-laid-out Home turns the
+// comparison into {} against {} for the rest of the run. Measured on android:
+// counts_prev was {} at EVERY evaluation point of all 17 runs, which is a
+// counting invariant that cannot fire at all.
+test("an un-laid-out Home reports unknown but leaves the carrier intact", () => {
+ const carried = { Checking: "3", Savings: "1" };
+ assert.deepEqual(readHomeCards({ route: "home", reading: null, previousCarrier: carried }), {
+ value: null,
+ carrier: carried,
+ fresh: false,
+ });
+});
+
+// The trace the fix has to survive, stepped through the carrier and the window
+// the spec holds. A double-submit commits two rows against one action, the
+// Home it lands on has not drawn its list yet, and the counting invariant must
+// still be able to see the pair once a real Home comes back.
+function run(steps: { route: string | null; cards: CardReading[]; lastAction: unknown }[]) {
+ let carrier: Record | null = null;
+ let submits = 0;
+ const out: { counts: Record | null; submits: number }[] = [];
+ for (const step of steps) {
+ const reading: HomeCardReading> = readHomeCards({
+ route: step.route,
+ reading: homeTxnCountsOf(step.cards),
+ previousCarrier: carrier,
+ });
+ carrier = reading.carrier;
+ const window = countSubmitsInWindow({
+ previousCount: submits,
+ lastAction: step.lastAction as { kind?: string; on?: string } | null,
+ fresh: reading.fresh,
+ });
+ submits = window.next;
+ out.push({ counts: reading.value, submits: window.reported });
+ }
+ return out;
+}
+
+const idle = { kind: "Tap", on: "testTag:AccountCard" };
+const doubleSubmit = { kind: "DoubleTap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" };
+
+test("an un-laid-out Home no longer kills the counting invariant", () => {
+ const trace = run([
+ { route: "home", cards: [card("Checking", 0, "3")], lastAction: null },
+ { route: "ledger", cards: [], lastAction: idle },
+ { route: "home", cards: [], lastAction: doubleSubmit },
+ { route: "ledger", cards: [], lastAction: idle },
+ { route: "home", cards: [card("Checking", 0, "5")], lastAction: idle },
+ ]);
+
+ // The un-laid-out Home is unknown for its own step, and the two steps after
+ // it get the last list anyone actually read rather than an empty one.
+ assert.deepEqual(trace[2]?.counts, null);
+ assert.deepEqual(trace[3]?.counts, { Checking: "3" });
+ assert.deepEqual(trace[4]?.counts, { Checking: "5" });
+
+ // The window it did not close still holds the double-submit, so one action
+ // against two committed rows is visible at the step that can compare them.
+ assert.equal(trace[4]?.submits, 1);
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: trace[3]?.counts ?? null,
+ countsAfter: trace[4]?.counts ?? null,
+ submitsInWindow: trace[4]?.submits ?? 0,
+ }),
+ true,
+ );
+});
+
+// The other half of the pairing: the counts window has to close on the counts
+// reading, not on the total's. A Home frame can render its footer total while
+// its list is still empty, and a window that reset there would compare a pair of
+// readings spanning submits it had already forgotten.
+test("an un-laid-out Home does not close the counting window", () => {
+ const trace = run([
+ { route: "home", cards: [card("Checking", 0, "3")], lastAction: null },
+ { route: "ledger", cards: [], lastAction: doubleSubmit },
+ { route: "home", cards: [], lastAction: doubleSubmit },
+ { route: "home", cards: [card("Checking", 0, "7")], lastAction: idle },
+ ]);
+ assert.equal(trace[3]?.submits, 2);
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "3" },
+ countsAfter: trace[3]?.counts ?? null,
+ submitsInWindow: trace[3]?.submits ?? 0,
+ }),
+ true,
+ );
+});
diff --git a/pkg/spec/test/folio-new-account.test.ts b/pkg/spec/test/folio-new-account.test.ts
new file mode 100644
index 0000000..def23bb
--- /dev/null
+++ b/pkg/spec/test/folio-new-account.test.ts
@@ -0,0 +1,235 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import { createdAccountHasNonZeroBalance } from "../../../examples/folio/sanderling/predicates.ts";
+
+const created = { kind: "Tap", on: "testTag:AddAccountScreen > testTag:AddAccountSubmit" };
+const idle = { kind: "Tap", on: "testTag:HomeScreen > testTag:AccountCard" };
+
+const account = (name: string, balance: number | null) => ({ name, balance });
+
+// The property still has teeth: the account the fuzzer just asked for, holding
+// money on the step its creation landed on Home, is a real violation.
+test("an account created holding money is a violation", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000)],
+ }),
+ true,
+ );
+});
+
+test("a double-tapped create is judged the same way", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: { kind: "DoubleTap", on: "id:AddAccountSubmit" },
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000)],
+ }),
+ true,
+ );
+});
+
+test("an account created empty is what the app is supposed to do", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 0)],
+ }),
+ false,
+ );
+});
+
+test("a balance that could not be read is not evidence", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", null)],
+ }),
+ false,
+ );
+});
+
+// The false positives this replaces. Home lists the accounts that fit the
+// viewport, so an account arrives in a later reading for reasons that have
+// nothing to do with being created: the list scrolled, a clipped card finished
+// laying out, or (before the route fix) the earlier reading came off a
+// half-rendered Home mid-transition. Measured on android: an existing Travel
+// holding $24,112.00 and an existing Savings holding $429,585.00, convicted for
+// coming into view.
+test("an account scrolling into view is not an account being created", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: idle,
+ typedName: "Travel",
+ before: [account("Emergency Fund", 461012300), account("Checking", 0)],
+ after: [
+ account("Emergency Fund", 461012300),
+ account("Checking", 0),
+ account("Travel", 2411200),
+ account("Savings", 0),
+ ],
+ }),
+ false,
+ );
+});
+
+test("nor is one that appears with no action at all behind it", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: null,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 2411200)],
+ }),
+ false,
+ );
+});
+
+// Even on the creation step, the only card judged is the one that answers to
+// the name the fuzzer typed. A card that came into view alongside it is still
+// just a card that came into view.
+test("a funded account arriving beside the created one is not judged", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Savings",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Savings", 0), account("Travel", 2411200)],
+ }),
+ false,
+ );
+});
+
+test("off Home there is no reading to judge", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "ledger",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000)],
+ }),
+ false,
+ );
+});
+
+test("a transition frame names no route, so nothing is judged there either", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: null,
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000)],
+ }),
+ false,
+ );
+});
+
+test("an unknown reading on either side is not evidence", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: null,
+ after: [account("Travel", 5000)],
+ }),
+ false,
+ );
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: null,
+ }),
+ false,
+ );
+});
+
+// defaultActions types edge-case text into the name field, and an empty name is
+// not a name we can find a card by.
+test("an empty typed name attributes nothing", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: " ",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000)],
+ }),
+ false,
+ );
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: undefined,
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000)],
+ }),
+ false,
+ );
+});
+
+// Web merges the card into one node whose text opens with the avatar initials,
+// so the identity key carries them: "TRTravel" is the card for "Travel".
+test("the merged web key still matches the name that was typed", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("CHChecking", 0)],
+ after: [account("CHChecking", 0), account("TRTravel", 5000)],
+ }),
+ true,
+ );
+});
+
+// Two cards answering to one typed name leave the appearance unattributable:
+// the fuzzer creates duplicates from a five-name list, and the tree has been
+// seen exposing the same card twice on a transition frame.
+test("two cards matching the typed name are not attributable to the creation", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0)],
+ after: [account("Checking", 0), account("Travel", 5000), account("MyTravel", 900)],
+ }),
+ false,
+ );
+});
+
+test("a card that was already there is not a card that was just created", () => {
+ assert.equal(
+ createdAccountHasNonZeroBalance({
+ route: "home",
+ lastAction: created,
+ typedName: "Travel",
+ before: [account("Checking", 0), account("Travel", 2411200)],
+ after: [account("Checking", 0), account("Travel", 2411200)],
+ }),
+ false,
+ );
+});
diff --git a/pkg/spec/test/folio-per-frame-reading.test.ts b/pkg/spec/test/folio-per-frame-reading.test.ts
new file mode 100644
index 0000000..bf41da6
--- /dev/null
+++ b/pkg/spec/test/folio-per-frame-reading.test.ts
@@ -0,0 +1,64 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import { oncePerFrame, routeOfFrame } from "../../../examples/folio/sanderling/predicates.ts";
+
+// oncePerFrame is what stops the folio spec re-walking the accessibility tree
+// once per extractor, and the whole of its safety is that the state object is a
+// fresh one every step (goja's stateObject, web's buildState). These tests pin
+// both halves: the same frame is read once, a different frame is a different
+// answer. A cache that outlived its frame would freeze every reading the spec
+// takes and the properties over them would go quietly vacuous.
+const SCREENS = { login: "LoginScreen", home: "HomeScreen" } as const;
+
+interface Frame {
+ present: readonly string[];
+ finds: number;
+}
+
+const frameShowing = (...present: readonly string[]): Frame => ({ present, finds: 0 });
+
+const routeOf = oncePerFrame((frame: Frame) =>
+ routeOfFrame(SCREENS, tag => {
+ frame.finds++;
+ return frame.present.includes(tag);
+ }),
+);
+
+test("one frame is walked once, however many readings ask", () => {
+ const home = frameShowing("HomeScreen");
+ assert.equal(routeOf(home), "home");
+ assert.equal(routeOf(home), "home");
+ assert.equal(routeOf(home), "home");
+ assert.equal(home.finds, 2);
+});
+
+test("a new frame is a new answer", () => {
+ const home = frameShowing("HomeScreen");
+ const login = frameShowing("LoginScreen");
+ assert.equal(routeOf(home), "home");
+ assert.equal(routeOf(login), "login");
+ assert.equal(login.finds, 2);
+});
+
+test("a transition frame is not answered off the frame before it", () => {
+ assert.equal(routeOf(frameShowing("HomeScreen")), "home");
+ assert.equal(routeOf(frameShowing("HomeScreen", "LoginScreen")), null);
+});
+
+test("returning to an earlier frame re-reads it", () => {
+ const home = frameShowing("HomeScreen");
+ routeOf(home);
+ routeOf(frameShowing("LoginScreen"));
+ assert.equal(routeOf(home), "home");
+ assert.equal(home.finds, 4);
+});
+
+// What makes memoizing the card list worth more than memoizing the route: the
+// three readings taken off it share one parse instead of three.
+test("a frame's reading is handed back by identity", () => {
+ const cardsOf = oncePerFrame((frame: Frame) => frame.present.map(tag => ({ tag })));
+ const home = frameShowing("HomeScreen");
+ assert.equal(cardsOf(home), cardsOf(home));
+ assert.notEqual(cardsOf(home), cardsOf(frameShowing("HomeScreen")));
+});
diff --git a/pkg/spec/test/folio-submit-balance-predicate.test.ts b/pkg/spec/test/folio-submit-balance-predicate.test.ts
index 90c0885..57a94ce 100644
--- a/pkg/spec/test/folio-submit-balance-predicate.test.ts
+++ b/pkg/spec/test/folio-submit-balance-predicate.test.ts
@@ -13,6 +13,7 @@ test("single submit: delta matches typed amount", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 1000,
currTotalBalance: 1500,
@@ -26,6 +27,7 @@ test("double submit: delta is twice the typed amount, fires", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 1000,
currTotalBalance: 2000,
@@ -39,6 +41,7 @@ test("DoubleTap kind also caught when delta exceeds typed amount", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "DoubleTap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 0,
currTotalBalance: 1000,
@@ -52,6 +55,7 @@ test("wrong action kind: vacuous true even with mismatch", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "InputText", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 1000,
currTotalBalance: 1000,
@@ -65,6 +69,7 @@ test("wrong target: vacuous true even with mismatch", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: "testTag:LoginScreen > testTag:LoginSubmit" },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 1000,
currTotalBalance: 1000,
@@ -78,6 +83,7 @@ test("null lastAction: vacuous true", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: null,
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 1000,
currTotalBalance: 2000,
@@ -91,6 +97,7 @@ test("zero typedAmount: vacuous true", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 0,
prevTotalBalance: 1000,
currTotalBalance: 1500,
@@ -104,6 +111,7 @@ test("selector as object: coerced safely and TxnSubmit detected", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: { testTag: "TxnSubmit" } },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 0,
currTotalBalance: 1000,
@@ -117,6 +125,7 @@ test("selector as object without TxnSubmit: vacuous true", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: { testTag: "LoginSubmit" } },
+ submitsInWindow: 1,
typedAmount: 500,
prevTotalBalance: 0,
currTotalBalance: 1000,
@@ -130,6 +139,7 @@ test("raw whole-dollar input: single submit clears", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: parseTypedAmount("50"),
prevTotalBalance: 5000,
currTotalBalance: 10000,
@@ -143,6 +153,7 @@ test("raw whole-dollar input: double submit fires", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: parseTypedAmount("50"),
prevTotalBalance: 5000,
currTotalBalance: 15000,
@@ -156,6 +167,7 @@ test("decimal input from empty prior balance clears", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: parseTypedAmount("5.50"),
prevTotalBalance: 0,
currTotalBalance: 550,
@@ -169,6 +181,7 @@ test("DoubleTap kind with raw whole-dollar input fires", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "DoubleTap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: parseTypedAmount("100"),
prevTotalBalance: 0,
currTotalBalance: 20000,
@@ -182,6 +195,7 @@ test("route gate: ledger landing with stale carrier is skipped", () => {
submitChangesBalanceByTypedAmount({
route: "ledger",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 5000,
prevTotalBalance: 0,
currTotalBalance: 0,
@@ -195,6 +209,7 @@ test("route gate: add-transaction landing with double-submit delta is skipped",
submitChangesBalanceByTypedAmount({
route: "add-transaction",
lastAction: { kind: "DoubleTap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 5000,
prevTotalBalance: 0,
currTotalBalance: 10000,
@@ -208,6 +223,7 @@ test("route gate: null route is skipped", () => {
submitChangesBalanceByTypedAmount({
route: null,
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 5000,
prevTotalBalance: 0,
currTotalBalance: 0,
@@ -221,6 +237,7 @@ test("route gate: home landing with matching delta passes", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 5000,
prevTotalBalance: 0,
currTotalBalance: 5000,
@@ -234,6 +251,7 @@ test("route gate: home landing with double-insert delta fires", () => {
submitChangesBalanceByTypedAmount({
route: "home",
lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
typedAmount: 5000,
prevTotalBalance: 0,
currTotalBalance: 10000,
@@ -241,3 +259,246 @@ test("route gate: home landing with double-insert delta fires", () => {
false,
);
});
+
+// Precision. Cents are integers in float64 here, so the equality only means
+// something while every number involved is exactly representable. The app takes
+// any amount that fits a Kotlin Long, and an iOS run reached a balance around
+// 1e18 cents, where representable values sit 128 apart: the delta of a
+// perfectly healthy single submit no longer reads back as the typed amount.
+const HUGE_BALANCE = 999999999999999900;
+
+test("above 2^53 the arithmetic itself is wrong, which is why the guard exists", () => {
+ assert.notEqual(Math.abs(HUGE_BALANCE + 1600 - HUGE_BALANCE), 1600);
+});
+
+test("above 2^53 a healthy single submit is not reported", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 1600,
+ prevTotalBalance: HUGE_BALANCE,
+ currTotalBalance: HUGE_BALANCE + 1600,
+ }),
+ true,
+ );
+});
+
+test("above 2^53 a double-submit delta is not reported either", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 1600,
+ prevTotalBalance: HUGE_BALANCE,
+ currTotalBalance: HUGE_BALANCE + 3200,
+ }),
+ true,
+ );
+});
+
+test("an unreadable previous balance above 2^53 is not evidence", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 1600,
+ prevTotalBalance: HUGE_BALANCE,
+ currTotalBalance: 5000,
+ }),
+ true,
+ );
+});
+
+// A typed amount past the safe range cannot be compared either. parseTypedAmount
+// returns 0 for those now, but the predicate takes the number from its caller
+// and must not convict on one it cannot hold.
+test("typed amount above 2^53 is not evidence", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 1e23,
+ prevTotalBalance: 0,
+ currTotalBalance: 0,
+ }),
+ true,
+ );
+});
+
+// The boundary, from both sides. MAX_SAFE_INTEGER still gets judged; one cent
+// more is where counting stops being exact.
+test("boundary: a double submit landing exactly on MAX_SAFE_INTEGER still fires", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 4503599627370495,
+ prevTotalBalance: 0,
+ currTotalBalance: 9007199254740990,
+ }),
+ false,
+ );
+});
+
+test("boundary: a single submit landing exactly on MAX_SAFE_INTEGER passes", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 9007199254740991,
+ prevTotalBalance: 0,
+ currTotalBalance: 9007199254740991,
+ }),
+ true,
+ );
+});
+
+test("boundary: one cent past MAX_SAFE_INTEGER stops being evidence", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 4503599627370496,
+ prevTotalBalance: 0,
+ currTotalBalance: 9007199254740992,
+ }),
+ true,
+ );
+});
+
+// The guard covers the balances and the typed amount, not their difference: two
+// safe balances subtract exactly whenever the result could have matched a safe
+// typed amount, so a mismatch here is real and must still be reported.
+test("a large but exact difference between safe balances still fires", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 500,
+ prevTotalBalance: -9007199254740991,
+ currTotalBalance: 9007199254740991,
+ }),
+ false,
+ );
+});
+
+// The 21-digit corpus amount end to end: the app refuses it, so nothing moves,
+// and the property must stay quiet rather than demand a 1e23-cent move.
+test("21-digit typed amount with an unmoved balance is not a violation", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: parseTypedAmount("999999999999999999999"),
+ prevTotalBalance: 220900,
+ currTotalBalance: 220900,
+ }),
+ true,
+ );
+});
+
+// Freshness. prevTotalBalance is the last total we READ, so the window between
+// it and now can hold more than one submit's transactions. A delta measured
+// over such a window is not evidence about the amount typed into any one of
+// them, and the android run that produced a 13000 delta against a typed 19600
+// is what that looks like: the window held a double-submit's two 19600 debits
+// and an unrelated 26200 credit.
+test("freshness: two submits in the window is vacuous, not a conviction", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "DoubleTap", on: submitOn },
+ submitsInWindow: 2,
+ typedAmount: 19600,
+ prevTotalBalance: 0,
+ currTotalBalance: -13000,
+ }),
+ true,
+ );
+});
+
+test("freshness: two submits cannot convict even on a clean 2x delta", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 2,
+ typedAmount: 500,
+ prevTotalBalance: 1000,
+ currTotalBalance: 2000,
+ }),
+ true,
+ );
+});
+
+// The boundary of the rule, from both sides. One submit is the only window the
+// property judges: zero means the total moved without a submit landing in it
+// (nothing to attribute the move to), and two or more means the move is shared.
+test("freshness boundary: exactly one submit is the window that convicts", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "DoubleTap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 19600,
+ prevTotalBalance: 0,
+ currTotalBalance: -39200,
+ }),
+ false,
+ );
+});
+
+test("freshness boundary: one submit with a healthy 1x delta still passes", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 1,
+ typedAmount: 19600,
+ prevTotalBalance: 0,
+ currTotalBalance: -19600,
+ }),
+ true,
+ );
+});
+
+test("freshness boundary: three submits is vacuous", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 3,
+ typedAmount: 500,
+ prevTotalBalance: 0,
+ currTotalBalance: 2500,
+ }),
+ true,
+ );
+});
+
+// A zero count would mean the step's own action was not counted as a submit,
+// which contradicts the action gate above it. Guard it anyway: a window with no
+// submit in it explains no balance move.
+test("freshness boundary: a window with no submit in it is vacuous", () => {
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: { kind: "Tap", on: submitOn },
+ submitsInWindow: 0,
+ typedAmount: 500,
+ prevTotalBalance: 1000,
+ currTotalBalance: 2000,
+ }),
+ true,
+ );
+});
diff --git a/pkg/spec/test/folio-submit-window.test.ts b/pkg/spec/test/folio-submit-window.test.ts
new file mode 100644
index 0000000..da422ed
--- /dev/null
+++ b/pkg/spec/test/folio-submit-window.test.ts
@@ -0,0 +1,151 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import {
+ countSubmitsInWindow,
+ isTxnSubmitTap,
+ readHomeTotalBalance,
+} from "../../../examples/folio/sanderling/predicates.ts";
+
+const submitOn = "testTag:AddTransactionScreen > testTag:TxnSubmit";
+
+test("a tap on TxnSubmit is a commit", () => {
+ assert.equal(isTxnSubmitTap({ kind: "Tap", on: submitOn }), true);
+});
+
+test("a double-tap on TxnSubmit is ONE commit action, not two", () => {
+ const window = countSubmitsInWindow({
+ previousCount: 0,
+ lastAction: { kind: "DoubleTap", on: submitOn },
+ fresh: true,
+ });
+ assert.equal(window.reported, 1);
+});
+
+test("a selector object naming TxnSubmit is a commit", () => {
+ assert.equal(isTxnSubmitTap({ kind: "Tap", on: { testTag: "TxnSubmit" } }), true);
+});
+
+test("typing into the amount field is not a commit", () => {
+ assert.equal(isTxnSubmitTap({ kind: "InputText", on: submitOn }), false);
+});
+
+test("tapping some other button is not a commit", () => {
+ assert.equal(isTxnSubmitTap({ kind: "Tap", on: "testTag:AddAccountSubmit" }), false);
+});
+
+test("no action at all is not a commit", () => {
+ assert.equal(isTxnSubmitTap(null), false);
+});
+
+test("a fresh Home reading closes the window and the next one starts empty", () => {
+ assert.deepEqual(
+ countSubmitsInWindow({ previousCount: 0, lastAction: { kind: "Tap", on: submitOn }, fresh: true }),
+ { reported: 1, next: 0 },
+ );
+});
+
+test("landing off Home keeps the submit in the window for the next step", () => {
+ assert.deepEqual(
+ countSubmitsInWindow({ previousCount: 0, lastAction: { kind: "Tap", on: submitOn }, fresh: false }),
+ { reported: 1, next: 1 },
+ );
+});
+
+test("a non-submit step neither adds to nor forgets the window", () => {
+ assert.deepEqual(
+ countSubmitsInWindow({ previousCount: 1, lastAction: { kind: "Tap", on: "testTag:AccountCard" }, fresh: false }),
+ { reported: 1, next: 1 },
+ );
+});
+
+test("a second submit with no Home reading between them counts two", () => {
+ assert.deepEqual(
+ countSubmitsInWindow({ previousCount: 1, lastAction: { kind: "DoubleTap", on: submitOn }, fresh: true }),
+ { reported: 2, next: 0 },
+ );
+});
+
+// The two traces the freshness rule exists to tell apart, driven step by step
+// through the same pair of carriers the spec holds.
+function run(steps: { route: string | null; totalText?: string; lastAction: unknown }[]) {
+ let carrier: number | null = null;
+ let submits = 0;
+ const out: { total: number | null; submits: number }[] = [];
+ for (const step of steps) {
+ const reading = readHomeTotalBalance({
+ route: step.route,
+ totalText: step.totalText,
+ previousCarrier: carrier,
+ });
+ carrier = reading.carrier;
+ const window = countSubmitsInWindow({
+ previousCount: submits,
+ lastAction: step.lastAction as { kind?: string; on?: string } | null,
+ fresh: reading.fresh,
+ });
+ submits = window.next;
+ out.push({ total: reading.value, submits: window.reported });
+ }
+ return out;
+}
+
+const idle = { kind: "Tap", on: "testTag:AccountCard" };
+const submit = { kind: "Tap", on: submitOn };
+const doubleSubmit = { kind: "DoubleTap", on: submitOn };
+
+// A double-submit pops the back stack twice (each Submit calls
+// navigator.back), so unlike a healthy single submit it lands back on Home,
+// which is why the property can see it at all.
+test("clean double submit: one action in the window, delta is 2x", () => {
+ const trace = run([
+ { route: "home", totalText: "$0.00", lastAction: null },
+ { route: "ledger", lastAction: idle },
+ { route: "ledger", lastAction: idle },
+ { route: "home", totalText: "$100.00", lastAction: doubleSubmit },
+ ]);
+ assert.equal(trace[3]?.submits, 1);
+ assert.equal(trace[0]?.total, 0);
+ assert.equal(trace[3]?.total, 10000);
+});
+
+// The contaminated window from the android run: an unrelated submit committed
+// while we were off Home, then the double-submit landed. The delta spans three
+// transactions, so it is not evidence about either typed amount.
+test("two submits between Home visits: the window is not evidence", () => {
+ const trace = run([
+ { route: "home", totalText: "$0.00", lastAction: null },
+ { route: "ledger", lastAction: idle },
+ { route: "ledger", lastAction: submit },
+ { route: "ledger", lastAction: idle },
+ { route: "home", totalText: "-$130.00", lastAction: doubleSubmit },
+ ]);
+ assert.equal(trace[4]?.submits, 2);
+});
+
+// Freshness is restored by seeing Home, not by time passing.
+test("a Home visit between two submits restores a one-action window", () => {
+ const trace = run([
+ { route: "home", totalText: "$0.00", lastAction: null },
+ { route: "ledger", lastAction: submit },
+ { route: "home", totalText: "$262.00", lastAction: idle },
+ { route: "ledger", lastAction: idle },
+ { route: "home", totalText: "$66.00", lastAction: doubleSubmit },
+ ]);
+ assert.equal(trace[1]?.submits, 1);
+ assert.equal(trace[2]?.submits, 1);
+ assert.equal(trace[4]?.submits, 1);
+});
+
+// An unreadable Home is not a Home reading: it must not close the window, or
+// the count would go back to zero against a total nobody read.
+test("an unreadable Home does not close the window", () => {
+ const trace = run([
+ { route: "home", totalText: "$0.00", lastAction: null },
+ { route: "ledger", lastAction: submit },
+ { route: "home", totalText: undefined, lastAction: idle },
+ { route: "home", totalText: "$66.00", lastAction: doubleSubmit },
+ ]);
+ assert.equal(trace[2]?.total, null);
+ assert.equal(trace[3]?.submits, 2);
+});
diff --git a/pkg/spec/test/folio-total-balance.test.ts b/pkg/spec/test/folio-total-balance.test.ts
index 759b608..67c8efa 100644
--- a/pkg/spec/test/folio-total-balance.test.ts
+++ b/pkg/spec/test/folio-total-balance.test.ts
@@ -1,97 +1,88 @@
import assert from "node:assert/strict";
import { test } from "node:test";
-import { computeHomeTotalBalance } from "../../../examples/folio/sanderling/predicates.ts";
+import { readHomeTotalBalance } from "../../../examples/folio/sanderling/predicates.ts";
-test("on Home with two cards ($10, $20): returns $30", () => {
- assert.equal(
- computeHomeTotalBalance({
- cardBalanceTexts: ["$10.00", "$20.00"],
- previousCarrier: 0,
- }),
- 3000,
+test("on Home the app's own total is the reading, the carrier and fresh", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "home", totalText: "$30.00", previousCarrier: 0 }),
+ { value: 3000, carrier: 3000, fresh: true },
);
});
-test("off Home (no cards) after a Home visit of $30: returns carrier $30", () => {
- assert.equal(
- computeHomeTotalBalance({
- cardBalanceTexts: [],
- previousCarrier: 3000,
- }),
- 3000,
+test("off Home there is nothing to read, so the carrier is reported unchanged", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "ledger", totalText: undefined, previousCarrier: 3000 }),
+ { value: 3000, carrier: 3000, fresh: false },
);
});
-test("off Home (no cards) with carrier still 0: returns 0", () => {
- assert.equal(
- computeHomeTotalBalance({
- cardBalanceTexts: [],
- previousCarrier: 0,
- }),
- 0,
+test("off Home before any Home visit reports the null carrier, still not fresh", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "ledger", totalText: undefined, previousCarrier: null }),
+ { value: null, carrier: null, fresh: false },
);
});
-test("sequence: Home $30, off-Home, Home $50 tracks new Home totals", () => {
- let carrier = 0;
- carrier = computeHomeTotalBalance({
- cardBalanceTexts: ["$10.00", "$20.00"],
- previousCarrier: carrier,
+test("a negative total parses with its sign", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "home", totalText: "-$1,234.56", previousCarrier: 0 }),
+ { value: -123456, carrier: -123456, fresh: true },
+ );
+});
+
+test("a fresh Home total overrides whatever the carrier held", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "home", totalText: "$7.50", previousCarrier: 9999 }),
+ { value: 750, carrier: 750, fresh: true },
+ );
+});
+
+// The poisoned carrier. An unreadable Home total is UNKNOWN for that step, so
+// null is reported and the property goes vacuous, but the carrier must keep the
+// last total we actually read. Writing null into the carrier is what used to end
+// the run: off-Home steps hand the carrier straight back, so a single
+// unreadable Home left every later step null.
+test("an unreadable Home total reports null but leaves the carrier intact", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "home", totalText: undefined, previousCarrier: 3000 }),
+ { value: null, carrier: 3000, fresh: false },
+ );
+});
+
+test("a garbled Home total is unknown, not zero", () => {
+ assert.deepEqual(
+ readHomeTotalBalance({ route: "home", totalText: "$", previousCarrier: 3000 }),
+ { value: null, carrier: 3000, fresh: false },
+ );
+});
+
+test("an unreadable Home no longer poisons the steps after it", () => {
+ let carrier: number | null = null;
+ const seen: (number | null)[] = [];
+ const step = (route: string | null, totalText: string | undefined) => {
+ const reading = readHomeTotalBalance({ route, totalText, previousCarrier: carrier });
+ carrier = reading.carrier;
+ seen.push(reading.value);
+ };
+
+ step("home", "$30.00");
+ step("home", undefined);
+ step("ledger", undefined);
+ step("ledger", undefined);
+ step("home", "$50.00");
+
+ assert.deepEqual(seen, [3000, null, 3000, 3000, 5000]);
+});
+
+// The clipped fifth account card that started this: it is not a card reading
+// any more, and the footer total the app renders is unaffected by which cards
+// the viewport happens to fit.
+test("Home total is one node, so an off-screen account cannot change it", () => {
+ const withFiveCards = readHomeTotalBalance({
+ route: "home",
+ totalText: "$2,589.00",
+ previousCarrier: 0,
});
- assert.equal(carrier, 3000);
- carrier = computeHomeTotalBalance({
- cardBalanceTexts: [],
- previousCarrier: carrier,
- });
- assert.equal(carrier, 3000);
- carrier = computeHomeTotalBalance({
- cardBalanceTexts: ["$20.00", "$30.00"],
- previousCarrier: carrier,
- });
- assert.equal(carrier, 5000);
-});
-
-test("Ledger step (no Home cards) holds the carrier, ignores Ledger balance", () => {
- let carrier = 0;
- carrier = computeHomeTotalBalance({
- cardBalanceTexts: ["$10.00", "$20.00"],
- previousCarrier: carrier,
- });
- assert.equal(carrier, 3000);
- carrier = computeHomeTotalBalance({
- cardBalanceTexts: [],
- previousCarrier: carrier,
- });
- assert.equal(carrier, 3000);
-});
-
-test("negative card balance parses with sign and sums correctly", () => {
- assert.equal(
- computeHomeTotalBalance({
- cardBalanceTexts: ["-$5.00", "$10.00"],
- previousCarrier: 0,
- }),
- 500,
- );
-});
-
-test("single card on Home overrides any previous carrier", () => {
- assert.equal(
- computeHomeTotalBalance({
- cardBalanceTexts: ["$7.50"],
- previousCarrier: 9999,
- }),
- 750,
- );
-});
-
-test("undefined card balance text is treated as 0", () => {
- assert.equal(
- computeHomeTotalBalance({
- cardBalanceTexts: [undefined, "$10.00"],
- previousCarrier: 0,
- }),
- 1000,
- );
+ assert.deepEqual(withFiveCards, { value: 258900, carrier: 258900, fresh: true });
});
diff --git a/pkg/spec/test/folio-transition-frame.test.ts b/pkg/spec/test/folio-transition-frame.test.ts
new file mode 100644
index 0000000..f785642
--- /dev/null
+++ b/pkg/spec/test/folio-transition-frame.test.ts
@@ -0,0 +1,142 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import {
+ countSubmitsInWindow,
+ readHomeCards,
+ readHomeTotalBalance,
+ routeOfFrame,
+ submitChangesBalanceByTypedAmount,
+} from "../../../examples/folio/sanderling/predicates.ts";
+
+// The spec's own screen table. A frame is the set of markers its accessibility
+// tree carries, which is all routeOfFrame is allowed to look at.
+const SCREENS = {
+ login: "LoginScreen",
+ "add-account": "AddAccountScreen",
+ "add-transaction": "AddTransactionScreen",
+ ledger: "LedgerScreen",
+ home: "HomeScreen",
+};
+
+const frame =
+ (...tags: string[]) =>
+ (tag: string) =>
+ tags.includes(tag);
+
+test("a frame showing one screen names its route", () => {
+ assert.equal(routeOfFrame(SCREENS, frame("LoginScreen")), "login");
+ assert.equal(routeOfFrame(SCREENS, frame("AddAccountScreen")), "add-account");
+ assert.equal(routeOfFrame(SCREENS, frame("AddTransactionScreen")), "add-transaction");
+ assert.equal(routeOfFrame(SCREENS, frame("LedgerScreen")), "ledger");
+ assert.equal(routeOfFrame(SCREENS, frame("HomeScreen")), "home");
+});
+
+test("a frame showing no screen at all is unknown", () => {
+ assert.equal(routeOfFrame(SCREENS, frame()), null);
+ assert.equal(routeOfFrame(SCREENS, frame("SomethingElse")), null);
+});
+
+// The android transition frame: 425 of 1879 steps across 17 measured runs carry
+// two screens, in every combination the navigation graph allows. Ranking the
+// markers and taking the first answers add-transaction for the first of these
+// while a second, unscoped look answers "on Home" -- and two answers for one
+// frame is the defect. There is one answer now, and on a transition frame it is
+// "I do not know".
+test("a transition frame showing two screens names neither", () => {
+ assert.equal(routeOfFrame(SCREENS, frame("AddTransactionScreen", "HomeScreen")), null);
+ assert.equal(routeOfFrame(SCREENS, frame("HomeScreen", "LedgerScreen")), null);
+ assert.equal(routeOfFrame(SCREENS, frame("AddAccountScreen", "HomeScreen")), null);
+ assert.equal(routeOfFrame(SCREENS, frame("HomeScreen", "LoginScreen")), null);
+ assert.equal(routeOfFrame(SCREENS, frame("AddTransactionScreen", "LedgerScreen")), null);
+});
+
+test("the three-screen frames android also emits name nothing", () => {
+ assert.equal(
+ routeOfFrame(SCREENS, frame("AddTransactionScreen", "HomeScreen", "LedgerScreen")),
+ null,
+ );
+});
+
+// Everything read off Home takes the route as its only input, so a frame that
+// is not Home cannot be read as Home by anything.
+test("a transition frame's half-drawn Home total is not a reading", () => {
+ const route = routeOfFrame(SCREENS, frame("AddTransactionScreen", "HomeScreen"));
+ assert.deepEqual(
+ readHomeTotalBalance({ route, totalText: "$86,911.00", previousCarrier: 8681600 }),
+ { value: 8681600, carrier: 8681600, fresh: false },
+ );
+});
+
+test("nor is its half-drawn card list", () => {
+ const route = routeOfFrame(SCREENS, frame("AddAccountScreen", "HomeScreen"));
+ const carried = { Travel: "8", Checking: "0" };
+ const partial = { Travel: "8" };
+ assert.deepEqual(readHomeCards>({ route, reading: partial, previousCarrier: carried }), {
+ value: carried,
+ carrier: carried,
+ fresh: false,
+ });
+});
+
+// The false conviction itself, android seed 3, steps 98-102 of the recorded
+// trace. Five submits deep into the window the app sits on AddTransaction with
+// "339" typed; a double-tap on Back starts the trip Home; the frame that comes
+// back carries BOTH screens with Home's total already drawn behind the outgoing
+// one. The old spec read that total as a fresh Home reading, reset the window to
+// zero, and aimed the next tap at a TxnSubmit button that had stopped existing.
+// The tap landed on Home, committed nothing, and the property demanded 33900 of
+// movement for it. Nine of the eleven android convictions were this, all at
+// delta 0.0x, all a single Tap where the real bug is a DoubleTap.
+test("the measured android transition chain no longer convicts at delta 0", () => {
+ let carrier: number | null = 8681600;
+ let submits = 5;
+ const step = (
+ tags: string[],
+ totalText: string | undefined,
+ lastAction: { kind: string; on: string } | null,
+ ) => {
+ const route = routeOfFrame(SCREENS, frame(...tags));
+ const reading = readHomeTotalBalance({ route, totalText, previousCarrier: carrier });
+ const window = countSubmitsInWindow({ previousCount: submits, lastAction, fresh: reading.fresh });
+ carrier = reading.carrier;
+ submits = window.next;
+ return { route, total: reading.value, submits: window.reported };
+ };
+
+ const back = { kind: "DoubleTap", on: "id:BackButton" };
+ const phantomSubmit = { kind: "Tap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" };
+
+ const transition = step(["AddTransactionScreen", "HomeScreen"], "$86,911.00", back);
+ assert.equal(transition.route, null);
+ assert.equal(transition.total, 8681600);
+ assert.equal(transition.submits, 5);
+
+ const landing = step(["HomeScreen"], "$86,911.00", phantomSubmit);
+ assert.equal(landing.submits, 6);
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: landing.route,
+ lastAction: phantomSubmit,
+ submitsInWindow: landing.submits,
+ typedAmount: 33900,
+ prevTotalBalance: transition.total,
+ currTotalBalance: landing.total,
+ }),
+ true,
+ );
+
+ // What the reset bought the old spec: the same landing, judged against a
+ // window of one and a total the transition frame had already banked.
+ assert.equal(
+ submitChangesBalanceByTypedAmount({
+ route: "home",
+ lastAction: phantomSubmit,
+ submitsInWindow: 1,
+ typedAmount: 33900,
+ prevTotalBalance: 8691100,
+ currTotalBalance: 8691100,
+ }),
+ false,
+ );
+});
diff --git a/pkg/spec/test/folio-txn-count-invariant.test.ts b/pkg/spec/test/folio-txn-count-invariant.test.ts
new file mode 100644
index 0000000..aea2f60
--- /dev/null
+++ b/pkg/spec/test/folio-txn-count-invariant.test.ts
@@ -0,0 +1,452 @@
+import assert from "node:assert/strict";
+import { test } from "node:test";
+
+import {
+ cardTxnCount,
+ committedTransactionsExceedSubmits,
+ homeTxnCountsOf,
+} from "../../../examples/folio/sanderling/predicates.ts";
+
+// Android and iOS give the count its own node; web merges the card into one
+// string, where the count sits between the name and the balance. The reading
+// carries which of the two it came from: a number is a count nothing else could
+// have leaked into, a string is a digit run that may have.
+test("a dedicated count node reads as a number, not a digit run", () => {
+ assert.equal(
+ cardTxnCount({ childText: "12 transactions", cardText: "INInvestments12 transactions$2,589.00" }),
+ 12,
+ );
+});
+
+test("merged card text: the count is taken from in front of the balance", () => {
+ assert.equal(
+ cardTxnCount({ childText: undefined, cardText: "INInvestments12 transactions$2,589.00" }),
+ "12",
+ );
+});
+
+test("merged card text: the singular label parses too", () => {
+ assert.equal(
+ cardTxnCount({ childText: undefined, cardText: "SASavings1 transaction$118.00" }),
+ "1",
+ );
+});
+
+// The balance has to come off first, or a name ending in digits would be read
+// as the count.
+test("a card with no readable count is unknown, not zero", () => {
+ assert.equal(cardTxnCount({ childText: undefined, cardText: undefined }), undefined);
+ assert.equal(cardTxnCount({ childText: undefined, cardText: "no digits here" }), undefined);
+ assert.equal(cardTxnCount({ childText: "", cardText: "AA" + "a".repeat(198) }), undefined);
+});
+
+// Measured on a real web run: the account named "-1" holding 2 transactions
+// merges to "-1-12 transactions-$119.00", and the maximal digit run reads 12.
+test("merged text runs a digit-ending name into the count", () => {
+ assert.equal(
+ cardTxnCount({ childText: undefined, cardText: "-1-12 transactions-$119.00" }),
+ "12",
+ );
+});
+
+const before = { Checking: "3", Savings: "1" };
+
+test("healthy window: three submits, three transactions", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: { Checking: "5", Savings: "2" },
+ submitsInWindow: 3,
+ }),
+ false,
+ );
+});
+
+test("rejected submits commit nothing, which is under the bound", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: { Checking: "3", Savings: "1" },
+ submitsInWindow: 4,
+ }),
+ false,
+ );
+});
+
+// The bug, stated directly: one tap, two rows.
+test("double submit: one action commits two transactions", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: { Checking: "5", Savings: "1" },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+// The point of counting actions against transactions rather than gating on a
+// one-submit window: a wide window is still a sound comparison.
+test("wide window: five submits committing six transactions still fires", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: { Checking: "8", Savings: "4" },
+ submitsInWindow: 5,
+ }),
+ true,
+ );
+});
+
+test("boundary: committed equal to the submit count is not a violation", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: { Checking: "4", Savings: "1" },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+test("boundary: one transaction past the submit count is", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: { Checking: "4", Savings: "2" },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+// Only accounts in both readings count. A card that scrolled out of the
+// viewport, or one whose count was unreadable, drops out of the sum, so the
+// result is a lower bound on what committed. Losing a card can only cost a
+// detection; it must never manufacture one.
+test("an account missing from the later reading is not counted", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "3", Savings: "1" },
+ countsAfter: { Checking: "3" },
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+});
+
+test("an account appearing only in the later reading is not counted", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "3" },
+ countsAfter: { Checking: "3", Travel: "9" },
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+});
+
+test("a card that scrolled away and back is not double counted", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "3" },
+ countsAfter: { Checking: "4", Savings: "40" },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+test("an unknown reading on either side is not evidence", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: null,
+ countsAfter: { Checking: "99" },
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "0" },
+ countsAfter: null,
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+});
+
+// The real trace this came from: at the violating step of seeds 3 and 5 the
+// window held exactly one submit action and the account's count moved by two.
+test("the measured web witness: submits 1, count delta 2", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { INInvestments: "12", Aa: "0" },
+ countsAfter: { INInvestments: "14", Aa: "0" },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+// The length rule, which is what keeps the merged-text prefix honest. The
+// account named "-1" reads 19 at nine transactions and 110 at ten: same account,
+// a delta of 91 out of a true delta of 1. Different run lengths are dropped.
+test("a count crossing a digit boundary is dropped, not convicted on", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { "-1-": "19" },
+ countsAfter: { "-1-": "110" },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+test("same run length keeps the prefixed delta exact", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { "-1-": "110" },
+ countsAfter: { "-1-": "112" },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { "-1-": "110" },
+ countsAfter: { "-1-": "111" },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+test("an unreadably long run is not evidence", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "1".repeat(21) },
+ countsAfter: { Checking: "9".repeat(21) },
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+});
+
+// The other side of that rule, and the reason it is scoped to merged text: a
+// count read off its own node has no account name in front of it, so its digits
+// ARE the count and a decade crossing is just a number getting longer. Both
+// windows below are real android seed-9 readings that the unscoped length rule
+// threw away, in runs that then finished clean.
+test("a dedicated node's count crossing a decade is usable evidence", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: 7 },
+ countsAfter: { Checking: 12 },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Savings: 4 },
+ countsAfter: { Savings: 10 },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+// Recovering the window is only worth anything if it still acquits the healthy
+// case, so the same crossing under a submit that earned it must not fire.
+test("a dedicated node's healthy decade crossing does not convict", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: 9 },
+ countsAfter: { Checking: 10 },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: 9 },
+ countsAfter: { Checking: 11 },
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+// The same numbers off merged text, where the digits may not be the count at
+// all: still dropped.
+test("the merged-text equivalent of that crossing is still dropped", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "9" },
+ countsAfter: { Checking: "10" },
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "7" },
+ countsAfter: { Checking: "12" },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+// The boundary itself. A pair whose two readings came from different sources is
+// vouched for by neither rule: the string may carry a name prefix the number
+// does not, so subtracting them is not a transaction count.
+test("a pair straddling the two sources is not comparable", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: 7 },
+ countsAfter: { Checking: "12" },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: { Checking: "7" },
+ countsAfter: { Checking: 12 },
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+// End to end from the two accessibility shapes, which is where the distinction
+// is actually made: the same account, the same true counts, read once off a
+// dedicated node and once off merged card text.
+const dedicated = (name: string, count: number) => ({
+ name,
+ balance: 0,
+ count: cardTxnCount({ childText: `${count} transactions`, cardText: undefined }),
+});
+
+const merged = (initials: string, name: string, count: number) => ({
+ name,
+ balance: 0,
+ count: cardTxnCount({
+ childText: undefined,
+ cardText: `${initials}${name}${count} transactions$0.00`,
+ }),
+});
+
+test("a dedicated-node card list convicts across a decade", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: homeTxnCountsOf([dedicated("Checking", 9)]),
+ countsAfter: homeTxnCountsOf([dedicated("Checking", 11)]),
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+test("the merged-text card list drops the same pair", () => {
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: homeTxnCountsOf([merged("CH", "Checking", 9)]),
+ countsAfter: homeTxnCountsOf([merged("CH", "Checking", 11)]),
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+});
+
+// Home lists whatever fits the viewport, and Folio lets two accounts share a
+// name, so one name can arrive on two cards. Keying counts by name collapsed
+// them onto the last card, and the two readings a window compares then came off
+// DIFFERENT cards: the probe below is a healthy app, one submit, and a scroll.
+test("two cards sharing a name do not become one count", () => {
+ const before = homeTxnCountsOf([{ name: "Travel", balance: 0, count: 0 }]);
+ const after = homeTxnCountsOf([
+ { name: "Travel", balance: 0, count: 0 },
+ { name: "Travel", balance: 500, count: 8 },
+ ]);
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: after,
+ submitsInWindow: 1,
+ }),
+ false,
+ );
+ assert.deepEqual(after, null);
+});
+
+test("a name on two cards is dropped from both readings", () => {
+ const before = homeTxnCountsOf([
+ { name: "Travel", balance: 0, count: "3" },
+ { name: "Travel", balance: 0, count: "9" },
+ { name: "Checking", balance: 0, count: "2" },
+ ]);
+ const after = homeTxnCountsOf([
+ { name: "Travel", balance: 0, count: "3" },
+ { name: "Travel", balance: 0, count: "11" },
+ { name: "Checking", balance: 0, count: "2" },
+ ]);
+ assert.deepEqual(before, { Checking: "2" });
+ assert.deepEqual(after, { Checking: "2" });
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: after,
+ submitsInWindow: 0,
+ }),
+ false,
+ );
+});
+
+// The other card need not be readable to spoil the identity: an unreadable
+// count still means the name on the map may not be the card that was read.
+test("a duplicate name is dropped even when the twin has no count", () => {
+ assert.deepEqual(
+ homeTxnCountsOf([
+ { name: "Travel", balance: 0, count: 4 },
+ { name: "Travel", balance: 0, count: undefined },
+ { name: "Savings", balance: 0, count: 1 },
+ ]),
+ { Savings: 1 },
+ );
+});
+
+// Dropping the ambiguous name must not disable the property for the rest.
+test("a unique name is still counted beside a dropped duplicate", () => {
+ const before = homeTxnCountsOf([
+ { name: "Travel", balance: 0, count: 3 },
+ { name: "Travel", balance: 0, count: 1 },
+ { name: "Checking", balance: 0, count: 4 },
+ ]);
+ const after = homeTxnCountsOf([
+ { name: "Travel", balance: 0, count: 3 },
+ { name: "Travel", balance: 0, count: 1 },
+ { name: "Checking", balance: 0, count: 6 },
+ ]);
+ assert.deepEqual(before, { Checking: 4 });
+ assert.equal(
+ committedTransactionsExceedSubmits({
+ countsBefore: before,
+ countsAfter: after,
+ submitsInWindow: 1,
+ }),
+ true,
+ );
+});
+
+test("distinct names are all counted", () => {
+ assert.deepEqual(
+ homeTxnCountsOf([
+ { name: "Checking", balance: 0, count: "3" },
+ { name: "Savings", balance: 0, count: "1" },
+ ]),
+ { Checking: "3", Savings: "1" },
+ );
+});
diff --git a/pkg/spec/test/folio-typed-amount.test.ts b/pkg/spec/test/folio-typed-amount.test.ts
index d0a387a..2a75298 100644
--- a/pkg/spec/test/folio-typed-amount.test.ts
+++ b/pkg/spec/test/folio-typed-amount.test.ts
@@ -43,14 +43,39 @@ test("zero returns 0", () => {
assert.equal(parseTypedAmount("0"), 0);
});
-test("leading plus sign tolerated as positive", () => {
- assert.equal(parseTypedAmount("+50"), 5000);
+// The app's parseCents matches ^\d+(\.\d{1,2})?$ against the trimmed input, so
+// a sign is rejected and no transaction is created. Reading "-50" as 5000 cents
+// would make the balance property demand a move the app never made.
+test("leading plus sign rejected, like the app", () => {
+ assert.equal(parseTypedAmount("+50"), 0);
});
-test("leading minus sign tolerated as positive", () => {
- assert.equal(parseTypedAmount("-50"), 5000);
+test("leading minus sign rejected, like the app", () => {
+ assert.equal(parseTypedAmount("-50"), 0);
});
test("comma-separated thousands accepted", () => {
assert.equal(parseTypedAmount("1,234.56"), 123456);
});
+
+// The input corpus types this 21-digit run into every field. parseCents calls
+// toLongOrNull on the whole part, which is null past Long.MAX, so the app
+// refuses the submit; float64 would have read it as 1e23 and asked the property
+// to find a balance move of 1e23 cents that never happened.
+test("21-digit corpus amount returns 0: the app rejects it", () => {
+ assert.equal(parseTypedAmount("999999999999999999999"), 0);
+});
+
+test("amount too large for exact cents returns 0", () => {
+ assert.equal(parseTypedAmount("100000000000000"), 0);
+});
+
+// 9007199254740991 cents is Number.MAX_SAFE_INTEGER: the last amount whose
+// cents survive the multiply intact.
+test("largest exactly representable amount is kept", () => {
+ assert.equal(parseTypedAmount("90071992547409.91"), 9007199254740991);
+});
+
+test("one cent past the safe range returns 0", () => {
+ assert.equal(parseTypedAmount("90071992547409.92"), 0);
+});
diff --git a/pkg/spec/test/web-dom-harness.ts b/pkg/spec/test/web-dom-harness.ts
index e0d3bdb..cc0e671 100644
--- a/pkg/spec/test/web-dom-harness.ts
+++ b/pkg/spec/test/web-dom-harness.ts
@@ -14,6 +14,15 @@ export interface FakeElementSpec {
y: number;
width: number;
height: number;
+ // id/testid/label/alt/title are how the host names a target: it builds the
+ // selector an action carries from them, so a fake needs them to exercise that
+ // naming. alt and title are the fallbacks the hierarchy dump folds into
+ // content-desc, which the host has to fall back to in the same order.
+ id?: string;
+ testid?: string;
+ label?: string;
+ alt?: string;
+ title?: string;
// clickable/editable place the element in the selector sets the host queries;
// the fake answers those queries directly rather than matching CSS.
clickable?: boolean;
@@ -28,6 +37,9 @@ export interface FakeElement extends FakeElementSpec {
tagName: string;
type: string;
isContentEditable: boolean;
+ id: string;
+ dataset: Record;
+ getAttribute(name: string): string | null;
scrollHeight: number;
clientHeight: number;
scrollWidth: number;
@@ -49,6 +61,10 @@ export function fakeElement(spec: FakeElementSpec): FakeElement {
tagName: spec.tag.toUpperCase(),
type: spec.tag === "input" ? "text" : "",
isContentEditable: editable && spec.tag !== "input" && spec.tag !== "textarea",
+ id: spec.id ?? "",
+ dataset: { testid: spec.testid },
+ getAttribute: (name: string) =>
+ ({ "aria-label": spec.label, alt: spec.alt, title: spec.title })[name] ?? null,
scrollHeight: spec.overflows ? spec.height * 2 : spec.height,
clientHeight: spec.height,
scrollWidth: spec.width,
diff --git a/pkg/spec/test/web-runtime.test.ts b/pkg/spec/test/web-runtime.test.ts
index d3e486f..11d4074 100644
--- a/pkg/spec/test/web-runtime.test.ts
+++ b/pkg/spec/test/web-runtime.test.ts
@@ -128,6 +128,53 @@ test("queryTargets reports a disabled control rather than dropping it", () => {
});
});
+// A target with no selector is a target no property can name. The action the
+// picker builds from it carries coordinates only, so `lastAction.on` is empty
+// and any property matching on WHICH control was acted upon cannot fire.
+test("queryTargets names a uniquely identified target", () => {
+ const submit = fakeElement({
+ tag: "button", x: 0, y: 0, width: 40, height: 20, clickable: true, id: "TxnSubmit",
+ });
+ const byTestid = fakeElement({
+ tag: "button", x: 0, y: 40, width: 40, height: 20, clickable: true, testid: "cancel",
+ });
+ const byLabel = fakeElement({
+ tag: "button", x: 0, y: 80, width: 40, height: 20, clickable: true, label: "Close",
+ });
+ // alt and title are the fallbacks the hierarchy dump folds into content-desc,
+ // so the host has to fall back to them in the same order or a name it calls
+ // unique resolves to a different element on the Go side.
+ const byAlt = fakeElement({ tag: "img", x: 0, y: 120, width: 40, height: 20, alt: "Logo" });
+ const byTitle = fakeElement({ tag: "div", x: 0, y: 160, width: 40, height: 20, title: "Help" });
+ const anonymous = fakeElement({ tag: "div", x: 0, y: 200, width: 10, height: 10 });
+ withFakeDocument([submit, byTestid, byLabel, byAlt, byTitle, anonymous], () => {
+ const targets = host.queryTargets();
+ assert.equal(targets[0]!.selector, "id:TxnSubmit");
+ assert.equal(targets[1]!.selector, "data-testid:cancel");
+ assert.equal(targets[2]!.selector, "desc:Close");
+ assert.equal(targets[3]!.selector, "desc:Logo");
+ assert.equal(targets[4]!.selector, "desc:Help");
+ assert.equal(targets[5]!.selector, undefined);
+ });
+});
+
+// A repeated id (folio's Home screen renders one AccountCard testTag per
+// account) names no single element, so the runner would re-resolve the action
+// onto whichever sibling it found first. Better unnamed than mis-aimed.
+test("queryTargets leaves duplicated identities unnamed", () => {
+ const first = fakeElement({
+ tag: "div", x: 0, y: 0, width: 40, height: 20, clickable: true, id: "AccountCard",
+ });
+ const second = fakeElement({
+ tag: "div", x: 0, y: 40, width: 40, height: 20, clickable: true, id: "AccountCard",
+ });
+ withFakeDocument([first, second], () => {
+ const targets = host.queryTargets();
+ assert.equal(targets[0]!.selector, undefined);
+ assert.equal(targets[1]!.selector, undefined);
+ });
+});
+
test("queryTargets caches within a tick until reset", () => {
const button = fakeElement({ tag: "button", x: 0, y: 0, width: 10, height: 10, clickable: true });
withFakeDocument([button], () => {
@@ -159,6 +206,13 @@ function withState(run: () => void) {
}
}
+// Every reading leaves the runtime inside a {value} envelope, so an extractor
+// whose getter returned undefined keeps its index instead of being dropped by
+// JSON.stringify.
+function readingOf(values: Record, index: number): unknown {
+ return values[index]!.value;
+}
+
test("named() sets the extractor's display name", () => {
const handle = __testing__.runtime.extract(() => "home").named("route");
const entry = __testing__.extractors.find((e) => e.handle === handle);
@@ -204,6 +258,63 @@ test("an uncaught cross-extractor read aborts evaluateExtractors", () => {
);
});
+// JSON.stringify drops an undefined-valued key, so a reading written straight
+// into the table took the extractor's whole INDEX with it when the getter
+// returned undefined - folio's on(route, tag) off its own screen, which is most
+// of its extractors on most steps. The host then kept goja's dump-derived value
+// for those and the page's for the rest, and a property comparing previous to
+// current across that split convicts an app that did nothing wrong.
+test("an extractor that returned undefined keeps its index through JSON", () => {
+ __testing__.extractors.length = 0;
+ __testing__.runtime.extract(() => undefined);
+ __testing__.runtime.extract(() => null);
+ __testing__.runtime.extract(() => 5);
+ let table: Record = {};
+ withState(() => {
+ table = __testing__.evaluateExtractors();
+ });
+
+ const overTheWire = JSON.parse(JSON.stringify(table)) as Record;
+ assert.deepEqual(Object.keys(overTheWire), ["0", "1", "2"]);
+ // undefined and null have to stay distinguishable across the wire: the goja
+ // host records undefined for a getter that returned undefined, so reporting
+ // null instead would make `x.current === undefined` answer one thing on
+ // native and another on web.
+ assert.equal("value" in overTheWire["0"]!, false);
+ assert.equal(overTheWire["1"]!.value, null);
+ assert.equal(overTheWire["2"]!.value, 5);
+});
+
+// state.lastAction is the one piece of state the page cannot observe for
+// itself: only the runner knows which action it actually applied. While the web
+// runtime hardcoded null there, a spec property gated on the last action (e.g.
+// folio's submitMovesBalanceByTypedAmount, which only looks at taps on
+// TxnSubmit) was vacuously true on web forever, and the run went green having
+// checked nothing.
+function lastActionSeenByASpec(pushed: unknown): unknown {
+ const setLastAction = (globalThis as Record)
+ .__sanderlingSetLastAction__ as (value: unknown) => void;
+ __testing__.extractors.length = 0;
+ __testing__.runtime.extract((state) => (state as { lastAction: unknown }).lastAction);
+ let out: Record = {};
+ withState(() => {
+ setLastAction(pushed);
+ out = __testing__.evaluateExtractors();
+ });
+ return readingOf(out, 0);
+}
+
+test("state.lastAction carries the action the host pushed", () => {
+ const action = { kind: "Tap", on: "id:TxnSubmit" };
+ assert.deepEqual(lastActionSeenByASpec(action), action);
+});
+
+test("state.lastAction is null when the host pushed nothing", () => {
+ // The first step of a run, and any step whose action was never applied: the
+ // goja host reports null there, so the web host must too.
+ assert.equal(lastActionSeenByASpec(null), null);
+});
+
// sanitize runs over every extractor's return value before it leaves the
// runtime. A user extractor that returns a page object reachable from
// document/window can be self-referential, carry functions, or nest deeply;
@@ -212,11 +323,11 @@ test("an uncaught cross-extractor read aborts evaluateExtractors", () => {
function sanitizeViaExtract(value: unknown): unknown {
__testing__.extractors.length = 0;
__testing__.runtime.extract(() => value);
- let out: Record = {};
+ let out: Record = {};
withState(() => {
out = __testing__.evaluateExtractors();
});
- return out[0];
+ return readingOf(out, 0);
}
test("sanitize breaks a self-referential cycle instead of overflowing", () => {
@@ -521,3 +632,105 @@ test("a non-editable element carries no hintText", () => {
assert.equal(attrsOf(element).hintText, undefined);
assert.equal(handleOf(element).editable, false);
});
+
+// An ax element is labelled with the selector it was found by, in the same
+// canonical grammar selectorStringFromJS emits in internal/verifier/marshal.go.
+// The label is what a spec's own Tap({ on: state.ax.find(...) }) carries to the
+// runner: with no label the action is coordinates only, `lastAction.on` is
+// empty, and a property matching on WHICH control was tapped cannot fire.
+const { selectorTag } = __testing__;
+
+test("selectorTag renders the selector shapes the goja host renders", () => {
+ assert.equal(selectorTag("testTag:TxnSubmit"), "testTag:TxnSubmit");
+ assert.equal(selectorTag({ testTag: "TxnSubmit" }), "testTag:TxnSubmit");
+ assert.equal(
+ selectorTag([{ testTag: "AddTransactionScreen" }, { testTag: "TxnSubmit" }]),
+ "testTag:AddTransactionScreen > testTag:TxnSubmit",
+ );
+ assert.equal(selectorTag({ testTag: "Row", "aria-label": "first" }), "testTag:Row aria-label:first");
+ assert.equal(selectorTag(undefined), "");
+});
+
+// A selector path scopes the second segment to each match of the first. It
+// returned nothing at all on web while returning matches on native, so folio's
+// accounts/totalBalance extractors (findAll([{HomeScreen}, {AccountCard}]))
+// were empty on every web step and the properties over them checked nothing.
+test("ax.findAll resolves a selector path segment by segment", () => {
+ const rect = { left: 0, top: 0, right: 10, bottom: 10, width: 10, height: 10 };
+ const node = (id: string, answers: Record = {}) => ({
+ id,
+ tagName: "DIV",
+ className: "",
+ textContent: id,
+ dataset: {},
+ getAttribute: () => null,
+ getBoundingClientRect: () => rect,
+ querySelectorAll: (selector: string) => answers[selector] ?? [],
+ });
+ const cardCss = `:is([data-testid="AccountCard"], [id="AccountCard"])`;
+ const screenCss = `:is([data-testid="HomeScreen"], [id="HomeScreen"])`;
+ const cards = [node("first"), node("second")];
+ const home = node("HomeScreen", { [cardCss]: cards });
+
+ const g = globalThis as Record;
+ const originalDocument = g.document;
+ const originalWindow = g.window;
+ g.document = { querySelectorAll: (selector: string) => (selector === screenCss ? [home] : []) };
+ g.window = {};
+ try {
+ __testing__.extractors.length = 0;
+ __testing__.runtime.extract((state) => {
+ const ax = (state as { ax: { findAll(s: unknown): Record[] } }).ax;
+ return ax
+ .findAll([{ testTag: "HomeScreen" }, { testTag: "AccountCard" }])
+ .map((card) => card.text);
+ });
+ const values = __testing__.evaluateExtractors();
+ // Scoped to the head match: the cards come from the HomeScreen node, not
+ // from a document-wide sweep for AccountCard.
+ assert.deepEqual(readingOf(values, 0), ["first", "second"]);
+ } finally {
+ g.document = originalDocument;
+ g.window = originalWindow;
+ }
+});
+
+test("ax.find and ax.findAll label the element with its selector", () => {
+ const rect = { left: 0, top: 0, right: 10, bottom: 10, width: 10, height: 10 };
+ const submit = {
+ id: "TxnSubmit",
+ tagName: "DIV",
+ className: "",
+ textContent: "Submit",
+ dataset: {},
+ getAttribute: () => null,
+ getBoundingClientRect: () => rect,
+ };
+ const matches = `:is([data-testid="TxnSubmit"], [id="TxnSubmit"])`;
+ const g = globalThis as Record;
+ const originalDocument = g.document;
+ const originalWindow = g.window;
+ g.document = { querySelectorAll: (selector: string) => (selector === matches ? [submit] : []) };
+ g.window = {};
+ try {
+ __testing__.extractors.length = 0;
+ __testing__.runtime.extract((state) => {
+ const ax = (state as { ax: { find(s: unknown): Record | undefined } }).ax;
+ return ax.find({ testTag: "TxnSubmit" });
+ });
+ __testing__.runtime.extract((state) => {
+ const ax = (state as { ax: { findAll(s: unknown): Record[] } }).ax;
+ return ax.findAll({ testTag: "TxnSubmit" });
+ });
+ const values = __testing__.evaluateExtractors();
+ const found = readingOf(values, 0) as Record;
+ assert.equal(found.__sanderlingSelector, "testTag:TxnSubmit");
+ // findAll passes each element through map(); passing the callback by
+ // reference would hand the array INDEX to the runtime as the selector.
+ const all = readingOf(values, 1) as Record[];
+ assert.equal(all[0]!.__sanderlingSelector, "testTag:TxnSubmit");
+ } finally {
+ g.document = originalDocument;
+ g.window = originalWindow;
+ }
+});
diff --git a/replay-ui/public/fonts/.gitkeep b/replay-ui/public/fonts/.gitkeep
deleted file mode 100644
index e69de29..0000000
diff --git a/replay-ui/sanderling/spec.ts b/replay-ui/sanderling/spec.ts
new file mode 100644
index 0000000..0044eaa
--- /dev/null
+++ b/replay-ui/sanderling/spec.ts
@@ -0,0 +1,180 @@
+// Sanderling fuzzing sanderling's own replay UI.
+//
+// Six of the seven properties here are cross-panel agreements: two panels that
+// derive the same fact by different paths have to say the same thing. That
+// holds for any trace, so this spec never needs recalibrating when the fixture
+// run changes. The seventh is the stock noUncaughtExceptions.
+//
+// The hooks it drives (data-testid, data-step, ...) were added to the UI for
+// this spec. Needing them is the lesson: a UI with no stable handles is a UI
+// nothing can assert on, and that is as true for a person writing a test as it
+// is for a fuzzer.
+
+import { type AccessibilityElement, type Key, PressKey, Tap, actions, always, extract, from, next, weighted } from "@sanderling/spec";
+import { defaultActions, noUncaughtExceptions } from "@sanderling/spec/defaults";
+
+function numberOf(value: string | undefined): number | null {
+ if (value === undefined || value === "") return null;
+ const parsed = Number(value);
+ return Number.isFinite(parsed) ? parsed : null;
+}
+
+function dataOf(element: AccessibilityElement | undefined, key: string): string | undefined {
+ const attrs = (element as unknown as { attrs?: Record } | undefined)?.attrs;
+ return attrs ? attrs[key] : undefined;
+}
+
+// The toolbar's own claim about which step is on screen, and how many there are.
+const toolbar = extract("toolbar", (s) => {
+ const indicator = s.ax.find({ "data-testid": "step-indicator" });
+ if (!indicator) return null;
+ return {
+ step: numberOf(dataOf(indicator, "step")),
+ stepCount: numberOf(dataOf(indicator, "stepCount")),
+ };
+});
+
+// The step list's claim: which rows exist, which one is selected, which ones
+// carry a violation marker.
+const stepRows = extract("stepRows", (s) =>
+ s.ax.findAll({ "data-testid": "step-row" }).map((row) => ({
+ step: numberOf(dataOf(row, "step")),
+ active: dataOf(row, "active") === "true",
+ violating: dataOf(row, "violations") === "true",
+ })),
+);
+
+// What the "state before" panel is showing, read off the image URL it renders.
+// Scoped to that panel by name rather than by DOM position: an earlier draft
+// took the first screenshot on the page, and the fuzzer put the before panel on
+// another tab, which left the after panel's image first and fired the property
+// against a UI that was behaving correctly.
+const beforeScreenshotStep = extract("beforeScreenshotStep", (s) => {
+ const image = s.ax.find([{ "data-testid": "state-before" }, { "data-testid": "screenshot" }]);
+ return image ? numberOf(dataOf(image, "step")) : null;
+});
+
+// Which tab is selected in each tab strip, as one comparable string.
+const activeTabs = extract("activeTabs", (s) =>
+ s.ax
+ .findAll({ "data-testid": "tab" })
+ .filter((tab) => dataOf(tab, "active") === "true")
+ .map((tab) => dataOf(tab, "tabId") ?? "")
+ .join(","),
+);
+
+// Two counts of one fact: the tab badge counts the step's violation records,
+// the panel counts the property rows it renders as violated.
+const violationBadges = extract("violationBadges", (s) =>
+ s.ax.findAll({ "data-testid": "violations-badge" }).map((badge) => numberOf(badge.text)),
+);
+const violationPanelCounts = extract("violationPanelCounts", (s) =>
+ s.ax
+ .findAll({ "data-testid": "violations-panel" })
+ .map((panel) => numberOf(dataOf(panel, "violationCount"))),
+);
+
+// The step in the URL is user input: it survives a reload, a typo, and every
+// tap that navigates. It must always land inside the run.
+const selectedStepIsInRange = always(() => {
+ const current = toolbar.current;
+ if (!current || current.step === null || current.stepCount === null) return true;
+ return current.step >= 1 && current.step <= current.stepCount;
+});
+
+// Exactly one row is highlighted whenever the list is on screen. Zero means the
+// toolbar is showing a step the list has no row for, which is what an
+// off-by-one or a failed clamp looks like from the list's side.
+const exactlyOneStepIsSelected = always(() => {
+ const rows = stepRows.current;
+ if (rows.length === 0) return true;
+ return rows.filter((row) => row.active).length === 1;
+});
+
+// The step count the toolbar prints and the number of rows the list renders are
+// two readings of the same run.
+const stepCountMatchesTheList = always(() => {
+ const current = toolbar.current;
+ const rows = stepRows.current;
+ if (!current || current.stepCount === null || rows.length === 0) return true;
+ return current.stepCount === rows.length;
+});
+
+// The screenshot panel builds its URL from the loaded run's step; the toolbar
+// prints the step from the URL. They must name the same step.
+const screenshotShowsTheSelectedStep = always(() => {
+ const current = toolbar.current;
+ const shown = beforeScreenshotStep.current;
+ if (!current || current.step === null || shown === null) return true;
+ return shown === current.step;
+});
+
+// Switching a tab is a view change, not a navigation: it must never move the
+// run to another step.
+const switchingTabsKeepsTheStep = always(
+ next(() => {
+ const previousTabs = activeTabs.previous;
+ const previousToolbar = toolbar.previous;
+ const currentToolbar = toolbar.current;
+ if (previousTabs === undefined || previousTabs === activeTabs.current) return true;
+ if (!previousToolbar || !currentToolbar) return true;
+ return previousToolbar.step === currentToolbar.step;
+ }),
+);
+
+// A badge that counts violations the panel cannot show a row for is a violation
+// the operator can see a number for and never read.
+const badgeCountMatchesThePanel = always(() => {
+ const badges = violationBadges.current;
+ const panels = violationPanelCounts.current;
+ if (badges.length === 0 || panels.length === 0) return true;
+ const badge = badges[0];
+ if (badge === null) return true;
+ return panels.every((count) => count === null || count === badge);
+});
+
+export const properties = {
+ noUncaughtExceptions,
+ selectedStepIsInRange,
+ exactlyOneStepIsSelected,
+ stepCountMatchesTheList,
+ screenshotShowsTheSelectedStep,
+ switchingTabsKeepsTheStep,
+ badgeCountMatchesThePanel,
+};
+
+// Step rows get their own weight for the same reason tabs do below: the default
+// enumeration reaches them now that role="option" is in the tappable set, but
+// one row out of the page's clickable elements is a thin chance, and both
+// step-facing properties are vacuous on a run that never selects one.
+const rowElements = extract("rowElements", (s) => s.ax.findAll({ "data-testid": "step-row" }));
+
+const selectAStep = actions(() => {
+ const rows = rowElements.current;
+ return rows.length === 0 ? [] : [Tap({ on: from(rows).generate() })];
+});
+
+// Left/right are the UI's own prev/next step bindings, and the one path into
+// the step navigation that does not go through a tap.
+const arrowKeys = from(["left", "right"]);
+const navigateByKeyboard = actions(() => [PressKey({ key: arrowKeys.generate() })]);
+
+// Tabs get their own weight rather than being left to the undirected tap mix:
+// with ~15 clickable elements on the page, an undirected run went 40 steps
+// without switching a single tab, which left both tab-facing properties
+// vacuously true. Weighting them is what makes those properties mean something.
+const tabElements = extract("tabElements", (s) => s.ax.findAll({ "data-testid": "tab" }));
+
+const switchATab = actions(() => {
+ const tabs = tabElements.current;
+ return tabs.length === 0 ? [] : [Tap({ on: from(tabs).generate() })];
+});
+
+// defaultActions carries the rest: the jump-to-violation button, the theme
+// toggle, the link back to the run list, and the scrolling.
+export const actionsRoot = weighted(
+ [30, selectAStep],
+ [20, navigateByKeyboard],
+ [25, switchATab],
+ [25, defaultActions],
+);
diff --git a/replay-ui/src/components/Tabs.tsx b/replay-ui/src/components/Tabs.tsx
index ec34bbf..3e1190c 100644
--- a/replay-ui/src/components/Tabs.tsx
+++ b/replay-ui/src/components/Tabs.tsx
@@ -65,6 +65,7 @@ export default function Tabs({ tabs, defaultTabId, ariaLabel }: TabsProps) {
{
diff --git a/replay-ui/src/panels/ViolationsPanel.tsx b/replay-ui/src/panels/ViolationsPanel.tsx
index 85bc5bb..41c0875 100644
--- a/replay-ui/src/panels/ViolationsPanel.tsx
+++ b/replay-ui/src/panels/ViolationsPanel.tsx
@@ -130,13 +130,18 @@ export default function ViolationsPanel({
}
return (
-
+ row.status === "violated").length}
+ >
{violationsOnly ? null : (
properties
@@ -149,7 +154,12 @@ export default function ViolationsPanel({
const residual = residuals?.[name];
const witness = status === "violated" ? witnesses?.[name] : undefined;
return (
-
+
0 ? (
-
+
{stepViolations.length}
) : undefined,
@@ -239,7 +239,7 @@ export default function RunDetail() {
label: "Violations",
badge:
stepViolations.length > 0 ? (
-
+
{stepViolations.length}
) : undefined,
@@ -267,7 +267,11 @@ export default function RunDetail() {
{basename(run.spec_path)} seed={run.seed}
-
+
step {stepIndex} / {stepCount ?? 0}
@@ -319,14 +323,14 @@ export default function RunDetail() {
-
+
-
+
state after
diff --git a/replay-ui/src/styles/tokens.css b/replay-ui/src/styles/tokens.css
index 036c58f..00e1b51 100644
--- a/replay-ui/src/styles/tokens.css
+++ b/replay-ui/src/styles/tokens.css
@@ -8,7 +8,7 @@
--text-muted: #6b6b6b;
--accent-violation: #d32f2f;
--accent-positive: #2e7d32;
- --font-mono: "JetBrains Mono", ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
+ --font-mono: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
}
:root[data-theme="dark"],
diff --git a/replay-ui/src/styles/typography.css b/replay-ui/src/styles/typography.css
index 2060768..9b065b3 100644
--- a/replay-ui/src/styles/typography.css
+++ b/replay-ui/src/styles/typography.css
@@ -1,27 +1,3 @@
-@font-face {
- font-family: "JetBrains Mono";
- font-style: normal;
- font-weight: 400;
- font-display: swap;
- src: url("/fonts/JetBrainsMono-Regular.woff2") format("woff2");
-}
-
-@font-face {
- font-family: "JetBrains Mono";
- font-style: normal;
- font-weight: 500;
- font-display: swap;
- src: url("/fonts/JetBrainsMono-Medium.woff2") format("woff2");
-}
-
-@font-face {
- font-family: "JetBrains Mono";
- font-style: normal;
- font-weight: 700;
- font-display: swap;
- src: url("/fonts/JetBrainsMono-Bold.woff2") format("woff2");
-}
-
body {
font-family: var(--font-mono);
font-size: 13px;
diff --git a/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt b/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt
index 0470d27..daedba2 100644
--- a/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt
+++ b/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt
@@ -63,14 +63,82 @@ internal const val STABILITY_POLL_CAP_MILLIS = 2000L
// missed.
internal const val MIN_STABLE_STREAK_MILLIS = 750L
+// TRANSITION_* bound the wait a snapshot pays when the tree it read shows two
+// routes at once, which on Compose means a NavHost cross-fade is in flight.
+// They are shorter than the constants above because this wait has one job,
+// outlasting the fade, where the poll above must also catch an async effect
+// that fires late.
+//
+// The streak matches the iOS companion's 300ms for the same reason: what this
+// wait is for is the cross-fade, and the cross-fade is caught by the
+// route-screen check rather than by streak length, so the streak only has to
+// be long enough that the reads spanning it are not all inside one frame.
+// MIN_STABLE_STREAK's 750ms is sized for a poll that must also catch an async
+// effect firing late, and at ~160ms per read-and-interval it costs three or
+// four more reads than this does.
+internal const val TRANSITION_STABLE_STREAK_MILLIS = 300L
+
+// The interval is the iOS companion's rather than STABILITY_POLL_INTERVAL's
+// 250ms: the wide interval exists to stop a per-step poll hammering
+// UiAutomation, and this one runs only on the frames that show two routes,
+// about a quarter of steps on folio. The read itself paces the loop.
+internal const val TRANSITION_POLL_INTERVAL_MILLIS = 100L
+
+// The cap is the iOS companion's 1500ms, and it has to be at least this: the
+// NavHost cross-fade is a 700ms tween (Compose navigation's default enter and
+// exit), it starts when the action lands rather than when the snapshot begins,
+// and the streak above has to fit after it. Measured on the emulator, a 600ms
+// cap left about a third of fades unfinished.
+//
+// A layout that holds two routes at rest pays the full cap on every snapshot
+// RPC it is read with, and the runner issues more than one: fetchSyncedState
+// (internal/runner) re-fetches a tree it considers transitional up to 4 times.
+// It stops early once two consecutive fetches come back byte-identical, so such
+// a layout costs 2 caps on a still tree and up to 4 on one that jitters
+// underneath. Per step, not once per step.
+internal const val TRANSITION_POLL_CAP_MILLIS = 1500L
+
+// awaitSettledTree reads the hierarchy and, while the tree it gets back holds
+// more than one route, keeps reading until the cross-fade lands or the cap
+// expires. It returns the last tree read, so the caller gets the settled one
+// rather than paying for another read.
+//
+// Two nodes carrying the SAME route id are one destination, not a transition;
+// countRouteScreens counts distinct tags, so a screen that nests a repeat of
+// its own id does not pay this wait at all.
+internal fun awaitSettledTree(read: () -> String): String {
+ var json = read()
+ if (countRouteScreens(json) <= 1) return json
+ pollUntilStable(
+ TRANSITION_POLL_CAP_MILLIS,
+ TRANSITION_STABLE_STREAK_MILLIS,
+ TRANSITION_POLL_INTERVAL_MILLIS,
+ ) {
+ json = read()
+ stabilitySnapshot(json)
+ }
+ return json
+}
+
// pollUntilStable returns when the snapshot has been non-null and equal to
-// itself for an uninterrupted stretch of at least MIN_STABLE_STREAK_MILLIS,
-// capped at timeoutMillis. snapshot must omit transient attributes (e.g.
-// measure-pass bounds) so layout-only flicker doesn't extend the wait, and
-// must return null when the snapshot looks transitional (e.g. mid NavHost
-// cross-fade) so the streak resets and the loop keeps polling instead of
-// declaring a partial state stable.
-internal fun pollUntilStable(timeoutMillis: Long, snapshot: () -> String?) {
+// itself for an uninterrupted stretch of at least streakMillis, capped at
+// timeoutMillis. snapshot must omit transient attributes (e.g. measure-pass
+// bounds) so layout-only flicker doesn't extend the wait, and must return null
+// when the snapshot looks transitional (e.g. mid NavHost cross-fade) so the
+// streak resets and the loop keeps polling instead of declaring a partial
+// state stable.
+//
+// streakMillis is quiet the poll OBSERVED: the clock starts when the read that
+// first matched its predecessor returns, so the reads spanning it are not
+// charged to it. A hierarchy fetch on Android costs more than the poll
+// interval, and charging it would let a 500ms read clear the default 750ms
+// streak having watched 250ms of quiet.
+internal fun pollUntilStable(
+ timeoutMillis: Long,
+ streakMillis: Long = MIN_STABLE_STREAK_MILLIS,
+ intervalMillis: Long = STABILITY_POLL_INTERVAL_MILLIS,
+ snapshot: () -> String?,
+) {
if (timeoutMillis <= 0) return
val deadline = System.currentTimeMillis() + timeoutMillis
var prior = try {
@@ -80,7 +148,7 @@ internal fun pollUntilStable(timeoutMillis: Long, snapshot: () -> String?) {
}
var streakStart = 0L
while (System.currentTimeMillis() < deadline) {
- Thread.sleep(STABILITY_POLL_INTERVAL_MILLIS)
+ Thread.sleep(intervalMillis)
val current = try {
snapshot()
} catch (_: Exception) {
@@ -89,7 +157,7 @@ internal fun pollUntilStable(timeoutMillis: Long, snapshot: () -> String?) {
val now = System.currentTimeMillis()
if (prior != null && current != null && prior == current) {
if (streakStart == 0L) streakStart = now
- if (now - streakStart >= MIN_STABLE_STREAK_MILLIS) return
+ if (now - streakStart >= streakMillis) return
} else {
streakStart = 0L
}
@@ -119,36 +187,43 @@ private val ROUTE_TAG_KEYS = setOf(
private val jsonMapper = com.fasterxml.jackson.module.kotlin.jacksonObjectMapper()
+// countRouteScreens counts DISTINCT route-level destination tags, not the nodes
+// carrying them. A screen that nests a node repeating its own route id puts two
+// tagged nodes in the tree while only one destination is on screen; counting
+// nodes would read that as a cross-fade that never ends, and the caller would
+// pay its full poll budget on every step of that screen and never settle.
internal fun countRouteScreens(treeJson: String): Int {
if (treeJson.isBlank()) return 0
return try {
val root = jsonMapper.readTree(treeJson)
- countRouteScreens(root)
+ val tags = mutableSetOf
()
+ collectRouteScreens(root, tags)
+ tags.size
} catch (_: Exception) {
0
}
}
-private fun countRouteScreens(
+private fun collectRouteScreens(
node: com.fasterxml.jackson.databind.JsonNode,
-): Int {
- var count = 0
+ into: MutableSet,
+) {
val attributes = node.get("attributes")
if (attributes != null && attributes.isObject) {
for (key in ROUTE_TAG_KEYS) {
val value = attributes.get(key) ?: continue
if (value.isNull) continue
- if (value.asText().endsWith("Screen")) {
- count++
+ val text = value.asText()
+ if (text.endsWith("Screen")) {
+ into.add(text)
break
}
}
}
val children = node.get("children")
if (children != null && children.isArray) {
- for (child in children) count += countRouteScreens(child)
+ for (child in children) collectRouteScreens(child, into)
}
- return count
}
// structuralHash hashes a Maestro TreeNode-shaped JSON string by walking it
@@ -873,13 +948,29 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend {
override fun recentLogs(sinceUnixMillis: Long, minLevel: String) =
readLogcat(serial, sinceUnixMillis, minLevel)
+ // snapshot waits out a NavHost cross-fade before it reads, so the runner is
+ // never handed a tree holding two routes at once. It belongs here rather
+ // than in waitForIdle: the runner gives waitForIdle a one-second deadline
+ // and abandons the RPC when it expires, which is not enough room for a
+ // 700ms fade that began before the settle did, and a wait that outlives the
+ // deadline just races the runner's own fetch on the device-side server.
+ // The snapshot RPC carries the step's deadline instead, so the wait can run
+ // to a bound that actually covers the animation.
+ //
+ // The predicate costs nothing on a settled frame: the read it needs is the
+ // read the snapshot was going to do anyway. That is what makes this
+ // affordable, where the structural poll that used to run in waitForIdle was
+ // not: it fetched the hierarchy ~4 more times on every mutating step.
+ override fun snapshot(): SnapshotSample =
+ SnapshotSample(awaitSettledTree { hierarchy() }, screenshot())
+
override fun waitForIdle(durationMillis: Long) {
// waitForAppToSettle blocks on the View-system animation and maestro's
- // own structural settle, which is enough on its own. A follow-up
- // structural-hash poll used to run here, but each hierarchy fetch is
- // ~500ms on a physical device, so it cost ~2.8s per mutating step for
- // marginal benefit; the runner already re-fetches while a frame still
- // looks transitional.
+ // own structural settle. It cannot see a Compose cross-fade: the fade
+ // keeps both routes alive with the tree byte-identical, so a settle
+ // that watches for change returns in the middle of one. That is what
+ // snapshot above waits out; a structural poll here used to try, cost
+ // ~2.8s per mutating step, and was removed.
driver.waitForAppToSettle(null, null, durationMillis.toInt())
}
diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt
new file mode 100644
index 0000000..3c7c776
--- /dev/null
+++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt
@@ -0,0 +1,108 @@
+package dev.sanderling.sidecar
+
+import org.junit.Test
+import kotlin.test.assertEquals
+import kotlin.test.assertTrue
+
+// The Android backend's snapshot waits out a NavHost cross-fade before it
+// reads. Without that wait the runner is handed a tree holding two routes at
+// once, refuses to act on it, and the step applies nothing; how many steps land
+// that way varies run to run, so the same seed walks a different trajectory
+// each time. These cover the wait (awaitSettledTree) and what it costs.
+class RouteTransitionTest {
+ private fun screen(id: String, child: String = "") =
+ """{"attributes":{"resource-id":"$id"},"children":[$child]}"""
+
+ private fun tree(vararg children: String) =
+ """{"attributes":{"resource-id":"root"},"children":[${children.joinToString(",")}]}"""
+
+ private val crossFade = tree(screen("LedgerScreen"), screen("AddTransactionScreen"))
+ private val landed = tree(screen("AddTransactionScreen"))
+
+ @Test fun waitsForTheCrossFadeToLandAndReturnsTheLandedTree() {
+ // The first read caught both routes alive. The wait must keep reading
+ // until only the destination is left, and hand back that tree: the
+ // caller records what it returns, so returning the fade would put the
+ // frame in the trace whether or not the wait happened.
+ var reads = 0
+ val settled = awaitSettledTree {
+ reads++
+ if (reads <= 3) crossFade else landed
+ }
+ assertTrue(reads > 3, "must keep reading until the fade lands, reads=$reads")
+ assertEquals(1, countRouteScreens(settled), "must return a tree with one route")
+ }
+
+ @Test fun settledFrameCostsExactlyOneRead() {
+ // A hierarchy read is the expensive part of a step, and this one is the
+ // read the snapshot was going to do anyway. A frame that is already on
+ // one route must not pay for a second: that cost, on every step, is why
+ // the unconditional structural poll was removed from waitForIdle.
+ var reads = 0
+ val settled = awaitSettledTree {
+ reads++
+ landed
+ }
+ assertEquals(1, reads, "a one-route frame must read once and return")
+ assertEquals(landed, settled)
+ }
+
+ @Test fun repeatedRouteIdIsOneScreenNotATransition() {
+ // A screen nesting a node that repeats its own route id puts two tagged
+ // nodes in the tree while one destination is on screen. Counting nodes
+ // would read that as a fade that never ends: every step of that screen
+ // would burn the whole poll budget and still hand over a frame the
+ // runner refuses to act on.
+ val nested = tree(screen("HomeScreen", screen("HomeScreen")))
+ assertEquals(1, countRouteScreens(nested), "the same id twice is one route")
+ var reads = 0
+ awaitSettledTree {
+ reads++
+ nested
+ }
+ assertEquals(1, reads, "a repeated route id must not be treated as a transition")
+ }
+
+ @Test fun aLayoutThatKeepsTwoRoutesIsBoundedByTheCap() {
+ // Two routes alive at rest is a real layout, not a fade, and no amount
+ // of waiting will resolve it. The wait is bounded by wall clock, so
+ // such a screen costs the cap once per step and nothing more.
+ val start = System.currentTimeMillis()
+ val settled = awaitSettledTree { crossFade }
+ val elapsed = System.currentTimeMillis() - start
+ assertTrue(
+ elapsed >= TRANSITION_POLL_CAP_MILLIS,
+ "must actually wait out a two-route tree, elapsed=${elapsed}ms",
+ )
+ assertTrue(
+ elapsed < TRANSITION_POLL_CAP_MILLIS + 1000L,
+ "must stop at the cap, elapsed=${elapsed}ms",
+ )
+ assertEquals(crossFade, settled, "the caller still gets a tree to record")
+ }
+
+ @Test fun capCoversTheNavHostFadePlusTheStreak() {
+ // Compose navigation's default enter and exit are a 700ms tween, and
+ // the fade starts when the action lands, not when the snapshot begins.
+ // A cap that does not clear the fade and the streak after it hands the
+ // caller a transitional frame, which is the whole defect.
+ val fadeMillis = 700L
+ val start = System.currentTimeMillis()
+ val settled = awaitSettledTree {
+ if (System.currentTimeMillis() - start < fadeMillis) crossFade else landed
+ }
+ val elapsed = System.currentTimeMillis() - start
+
+ assertEquals(landed, settled, "must hand back the landed tree, not the fade")
+ assertTrue(
+ elapsed >= fadeMillis,
+ "cannot have settled before the fade ended, elapsed=${elapsed}ms",
+ )
+ assertTrue(
+ elapsed < TRANSITION_POLL_CAP_MILLIS,
+ "the ${TRANSITION_POLL_CAP_MILLIS}ms cap has to leave room for a ${fadeMillis}ms " +
+ "fade and the ${TRANSITION_STABLE_STREAK_MILLIS}ms streak after it, but the " +
+ "wait ran to the cap instead, elapsed=${elapsed}ms",
+ )
+ }
+}
diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt
index de9937e..c8da826 100644
--- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt
+++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt
@@ -16,28 +16,68 @@ class StabilityPollTest {
assertTrue(elapsed < 3000L, "should not run to cap when stable, elapsed=${elapsed}ms")
}
+ @Test fun slowSnapshotReadsDoNotEatTheStreak() {
+ // Every other test here uses an instant lambda and so passes whether or
+ // not a read is charged to the streak; this one is the difference.
+ // StubDriverBackend's waitForIdle polls a real `uiautomator dump`,
+ // which costs hundreds of milliseconds, so the slow read is the case it
+ // runs in.
+ val readMillis = 400L
+ val sampleStarts = mutableListOf()
+ val sampleEnds = mutableListOf()
+ val start = System.currentTimeMillis()
+ pollUntilStable(5000L) {
+ sampleStarts += System.currentTimeMillis() - start
+ Thread.sleep(readMillis)
+ sampleEnds += System.currentTimeMillis() - start
+ "stable"
+ }
+
+ // What the poll actually watched: the last read began this long after
+ // the first one returned, and every sample in between matched.
+ val observedQuiet = sampleStarts.last() - sampleEnds.first()
+ assertTrue(
+ observedQuiet >= MIN_STABLE_STREAK_MILLIS,
+ "the poll returned having observed only ${observedQuiet}ms of quiet, not " +
+ "${MIN_STABLE_STREAK_MILLIS}ms; starts=$sampleStarts ends=$sampleEnds",
+ )
+ assertTrue(
+ sampleStarts.size >= 3,
+ "a ${readMillis}ms read cannot clear the streak in one pair, starts=$sampleStarts",
+ )
+ }
+
@Test fun streakResetsOnAnyChange() {
// A late transition that fires after the prior streak has already
// begun must reset the clock: the post-transition stable window has
// to start over and meet MIN_STABLE_STREAK_MILLIS from scratch.
var calls = 0
- val start = System.currentTimeMillis()
- // First 4 samples are "calm", then 1 transient change, then "stable"
- // forever - the calm prefix is meaningless because of the transition.
+ var transientAt = 0L
+ // A short "calm" prefix, one transient change, then "stable" forever.
+ // The prefix is meaningless because of the transition: only the streak
+ // that starts after it can end the wait.
pollUntilStable(3000L) {
calls++
when {
- calls <= 4 -> "calm"
- calls == 5 -> "transient"
+ calls <= 2 -> "calm"
+ calls == 3 -> {
+ transientAt = System.currentTimeMillis()
+ "transient"
+ }
else -> "stable"
}
}
- val elapsed = System.currentTimeMillis() - start
+ val sinceTransition = System.currentTimeMillis() - transientAt
assertTrue(
- elapsed >= MIN_STABLE_STREAK_MILLIS,
- "post-transition streak must reach ${MIN_STABLE_STREAK_MILLIS}ms, elapsed=${elapsed}ms",
+ calls >= 8,
+ "after the transition the poll needs a fresh matching pair and then a full " +
+ "${MIN_STABLE_STREAK_MILLIS}ms of quiet, which is 8 samples, got $calls",
+ )
+ assertTrue(
+ sinceTransition >= MIN_STABLE_STREAK_MILLIS,
+ "the calm prefix must not count: a full ${MIN_STABLE_STREAK_MILLIS}ms streak has to " +
+ "start over after the transition, returned ${sinceTransition}ms after it",
)
- assertTrue(calls >= 10, "expected enough samples to span calm + transient + stable streak, got $calls")
}
@Test fun transitionalNullsForceLoopToKeepWaiting() {
diff --git a/test/browser/browser_test.go b/test/browser/browser_test.go
index e603d6c..3cae9cd 100644
--- a/test/browser/browser_test.go
+++ b/test/browser/browser_test.go
@@ -53,6 +53,30 @@ func TestBrowserCounterInvariantHolds(t *testing.T) {
}
}
+// TestBrowserShadowDOMIsReachable drives a page whose entire UI (canvas, the
+// button over it, and the counter) lives inside a shadow root, the shape
+// Compose for Web produces. Both the enumeration and the selector lookup have
+// to cross the boundary for the counter to move at all, so the property firing
+// is the end-to-end evidence.
+func TestBrowserShadowDOMIsReachable(t *testing.T) {
+ violations := runFixture(t, "shadow")
+ if !slices.Contains(violations, "counterNeverMoves") {
+ t.Fatalf("nothing inside the shadow root was ever tapped; violations=%v", violations)
+ }
+}
+
+// TestBrowserAriaRoleIsTappable drives a page whose only controls are
+// rows, the shape the replay UI gives its step list. The
+// tappable set covered role="button" and nothing else, so a spec had to
+// hand-write an action to reach a row; with the standard interactive roles
+// covered, the default tap enumeration finds them and the counter moves.
+func TestBrowserAriaRoleIsTappable(t *testing.T) {
+ violations := runFixture(t, "aria-roles")
+ if !slices.Contains(violations, "noRowWasEverSelected") {
+ t.Fatalf("no role=\"option\" row was ever tapped; violations=%v", violations)
+ }
+}
+
// runFixture serves the named testdata case over an in-process file server,
// drives it through headless Chrome with a fixed seed and a bounded step count,
// and returns every property name that was ever reported violated.
@@ -180,3 +204,21 @@ func testdataDir(t *testing.T) string {
func specSrcDir(t *testing.T) string {
return filepath.Join(repoRoot(t), "pkg", "spec", "src")
}
+
+// TestBrowserUndefinedExtractorStaysUndefined drives the four layers of the
+// undefined reading through one run: the page wraps each reading in a {value}
+// envelope, the driver unwraps an absent value to an empty payload, the runner
+// checks the page reported one reading per extractor, and the verifier decodes
+// the empty payload as undefined. Each layer has its own unit test; only a run
+// proves they compose. Written straight into the map instead, an undefined
+// reading lost its whole index to JSON.stringify and that extractor silently
+// kept goja's dump-derived value while its neighbours held the page's.
+func TestBrowserUndefinedExtractorStaysUndefined(t *testing.T) {
+ violations := runFixture(t, "undefined-extractor")
+ if slices.Contains(violations, "undefinedStaysUndefined") {
+ t.Error("a reading the page could not take did not reach the spec as undefined")
+ }
+ if !slices.Contains(violations, "counterNeverMoves") {
+ t.Fatalf("nothing was ever tapped, so the property above held vacuously; violations=%v", violations)
+ }
+}
diff --git a/test/browser/testdata/aria-roles/index.html b/test/browser/testdata/aria-roles/index.html
new file mode 100644
index 0000000..40d9696
--- /dev/null
+++ b/test/browser/testdata/aria-roles/index.html
@@ -0,0 +1,29 @@
+
+
+
+
+ aria roles
+
+
+
+
+ step 1
+ step 2
+ step 3
+
+ 0
+
+
+
diff --git a/test/browser/testdata/aria-roles/spec.ts b/test/browser/testdata/aria-roles/spec.ts
new file mode 100644
index 0000000..066edd6
--- /dev/null
+++ b/test/browser/testdata/aria-roles/spec.ts
@@ -0,0 +1,16 @@
+import { always, extract, taps } from "@sanderling/spec";
+
+const selections = extract((s) => {
+ const el = s.ax.find({ id: "selected" });
+ return el ? parseInt(el.text, 10) || 0 : 0;
+}).named("selections");
+
+// Every control on the page is an , the shape the replay UI
+// gives its own step rows. Nothing else can move this counter, so the spec
+// carries no action of its own: this property firing IS the evidence that the
+// default enumeration offered a tap on a role-based control.
+const noRowWasEverSelected = always(() => selections.current === 0);
+
+export const properties = { noRowWasEverSelected };
+
+export const actionsRoot = taps;
diff --git a/test/browser/testdata/shadow/index.html b/test/browser/testdata/shadow/index.html
new file mode 100644
index 0000000..6e21e34
--- /dev/null
+++ b/test/browser/testdata/shadow/index.html
@@ -0,0 +1,43 @@
+
+
+
+
+ shadow
+
+
+
+
+
+
+
diff --git a/test/browser/testdata/shadow/spec.ts b/test/browser/testdata/shadow/spec.ts
new file mode 100644
index 0000000..c70f37b
--- /dev/null
+++ b/test/browser/testdata/shadow/spec.ts
@@ -0,0 +1,17 @@
+import { always, extract, taps } from "@sanderling/spec";
+
+const bumps = extract((s) => {
+ const el = s.ax.find({ id: "count" });
+ return el ? parseInt(el.text, 10) || 0 : 0;
+}).named("bumps");
+
+// The button, the counter that records the taps, and the canvas that services
+// them all live inside a shadow root. A host that stops at the shadow boundary
+// enumerates no target to tap and resolves no selector to read, so the counter
+// can never leave zero: this property firing IS the evidence that both the
+// enumeration and the selector lookup crossed the boundary.
+const counterNeverMoves = always(() => bumps.current === 0);
+
+export const properties = { counterNeverMoves };
+
+export const actionsRoot = taps;
diff --git a/test/browser/testdata/undefined-extractor/index.html b/test/browser/testdata/undefined-extractor/index.html
new file mode 100644
index 0000000..4df2a39
--- /dev/null
+++ b/test/browser/testdata/undefined-extractor/index.html
@@ -0,0 +1,18 @@
+
+
+
+
+
+
+
+
diff --git a/test/browser/testdata/undefined-extractor/spec.ts b/test/browser/testdata/undefined-extractor/spec.ts
new file mode 100644
index 0000000..bf7fb6b
--- /dev/null
+++ b/test/browser/testdata/undefined-extractor/spec.ts
@@ -0,0 +1,26 @@
+import { always, extract, taps } from "@sanderling/spec";
+
+// Two extractors that resolve nothing, the shape every route-scoped extractor
+// has on the steps it is off its own screen (folio has nine of them). JSON has
+// no undefined, so these are what the {value} envelope exists for.
+const absent = extract((s) => s.ax.find({ id: "nothing-is-ever-here" })).named("absent");
+const absentText = extract((s) => s.ax.find({ id: "missing" })?.text).named("absentText");
+
+const bumps = extract((s) => {
+ const el = s.ax.find({ id: "count" });
+ return el ? parseInt(el.text, 10) || 0 : 0;
+}).named("bumps");
+
+// A reading the page could not take must arrive as undefined, not as null and
+// not as goja's reading of the dump. This property is what a spec sees.
+const undefinedStaysUndefined = always(
+ () => absent.current === undefined && absentText.current === undefined,
+);
+
+// The run has to actually be doing something, or the property above holds for
+// the wrong reason.
+const counterNeverMoves = always(() => bumps.current === 0);
+
+export const properties = { undefinedStaysUndefined, counterNeverMoves };
+
+export const actionsRoot = taps;