diff --git a/.env.local.example b/.env.local.example index 879acdd..f2b8f72 100644 --- a/.env.local.example +++ b/.env.local.example @@ -5,7 +5,7 @@ # / `release-android-local` / `release-npm-dry` Make targets don't need any # of these. They're snapshot/local-only. # -# In CI, these are provided via GitHub Actions secrets (see .github/workflows/release.yml). +# In CI, these are provided via GitHub Actions secrets (see .github/workflows/ci.yml). # npm automation token (bypasses 2FA). # Create at npmjs.com → Access Tokens → Generate New Token → Automation. diff --git a/.github/actions/folio-app/action.yml b/.github/actions/folio-app/action.yml new file mode 100644 index 0000000..f036868 --- /dev/null +++ b/.github/actions/folio-app/action.yml @@ -0,0 +1,81 @@ +name: folio app +description: Install the toolchain folio needs on one platform, and build the app there. + +inputs: + platform: + description: android, ios or web + required: true + +runs: + using: composite + steps: + - name: Set up the JDKs + uses: actions/setup-java@v5 + with: + distribution: temurin + # The metro gradle plugin folio builds with needs a 21 runtime; the + # sidecar toolchain pins 17. Both are installed so gradle can pick. + java-version: | + 17 + 21 + + # The iOS app builds its Kotlin framework through the folio gradle project, + # which configures :app:androidApp, so this is needed off Android too. + # `make sanderling-android` wants it as well, for the sidecar JAR. + - name: Set up Android SDK + if: inputs.platform != 'web' + uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1 + + - name: Cache Gradle + if: inputs.platform != 'ios' + uses: actions/cache@v6 + with: + path: | + ~/.gradle/caches + ~/.gradle/wrapper + key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }} + restore-keys: | + folio-gradle-${{ runner.os }}- + + # idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so + # it has to be named by its full tap path. xcodegen and just are core. + - name: Install idb-companion, xcodegen and just + if: inputs.platform == 'ios' + shell: bash + run: brew install facebook/fb/idb-companion xcodegen just + + # Both asset tarballs are built by the prepare scripts, and the runner + # bundle is an xcodebuild of companion/Sources. Keyed on the scripts and + # the versions the Makefile embeds, so a later run reuses them. This has to + # land before `make sanderling-ios`, which is what consumes them. + - name: Cache the companion and runner bundles + if: inputs.platform == 'ios' + uses: actions/cache@v6 + with: + path: | + internal/driver/ioscompanion/companionassets/assets + internal/driver/ioscompanion/runnerassets/assets + key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }} + + # Without this the emulator falls back to software rendering and every + # step costs several seconds. + - name: Enable KVM + if: inputs.platform == 'android' + shell: bash + run: | + echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \ + | sudo tee /etc/udev/rules.d/99-kvm4all.rules + sudo udevadm control --reload-rules + sudo udevadm trigger --name-match=kvm + + - name: Build the folio APK + if: inputs.platform == 'android' + shell: bash + working-directory: examples/folio + run: ./gradlew :app:androidApp:assembleDebug + + - name: Build the folio wasmJs app + if: inputs.platform == 'web' + shell: bash + working-directory: examples/folio + run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution diff --git a/.github/actions/headless-chrome/action.yml b/.github/actions/headless-chrome/action.yml new file mode 100644 index 0000000..223df70 --- /dev/null +++ b/.github/actions/headless-chrome/action.yml @@ -0,0 +1,30 @@ +name: headless chrome +description: Install Chrome and prove it starts headless before a driver depends on it. + +runs: + using: composite + steps: + # stable is setup-chrome v2's own default, spelled out so a new release of + # the action cannot move the browser these jobs drive. The alternative it + # offers is Chrome for Testing latest, which tracks ahead of the channel + # users run. + - uses: browser-actions/setup-chrome@2e1d749697dd1612b833dba4a722266286fbefcd # v2.1.2 + with: + chrome-version: stable + + # Ubuntu 24.04 (current ubuntu-latest) restricts unprivileged user + # namespaces via AppArmor, which stops headless Chrome from starting even + # with --no-sandbox: the process launches but never opens its DevTools + # socket. Re-enable them so the driver's Chrome can come up. + - name: Allow Chrome under unprivileged user namespaces + shell: bash + run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0 + + # Fail here with Chrome's own stderr if the browser can't launch, instead + # of letting the driver report an opaque DevTools timeout downstream. + - name: Verify headless Chrome starts + shell: bash + run: | + chrome --version + chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \ + --dump-dom 'data:text/html,ok' diff --git a/.github/scripts/folio-run-test.sh b/.github/scripts/folio-run-test.sh new file mode 100755 index 0000000..ef2344b --- /dev/null +++ b/.github/scripts/folio-run-test.sh @@ -0,0 +1,373 @@ +#!/usr/bin/env bash +# Drives folio-run.sh through a stubbed `sanderling` binary and checks the +# verdict it reaches from each shape of trace. What is under test is the +# classification, not the fuzzer: the stub writes the trace the run would have +# written and exits the code the run would have exited. +# +# folio-run.sh is invoked as `bash -eo pipefail -c ` + const withoutSetter = `
no sanderling runtime here
` + pages := map[string]string{"/with": withSetter, "/without": withoutSetter} + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/html") + _, _ = w.Write([]byte(pages[r.URL.Path])) + })) + defer server.Close() + + d := New() + defer d.Terminate(context.Background()) + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + + if err := d.Launch(ctx, server.URL+"/with", false, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + logs := json.RawMessage(`[{"unixMillis":1,"level":"E","tag":"console","message":"boom"}]`) + if err := d.SetLogs(ctx, logs); err != nil { + t.Fatalf("SetLogs on a page that defines the setter: %v", err) + } + var seen []map[string]any + if err := chromedp.Run(d.tabCtx, + chromedp.Evaluate(`window.__logsSeen`, &seen)); err != nil { + t.Fatalf("read installed logs: %v", err) + } + if len(seen) != 1 || seen[0]["level"] != "E" || seen[0]["message"] != "boom" { + t.Errorf("the page received %v, want the error-level entry the driver captured", seen) + } + + if err := d.Launch(ctx, server.URL+"/without", false, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + if err := d.SetLogs(ctx, logs); err == nil { + t.Error("SetLogs reported success on a page with no setter; " + + "a runtime that cannot take the step's logs is indistinguishable from one that did") + } +} + // TestEvaluateExtractors_ReportsAMissingTable is the same failure on the other // sampler. An empty override map is what a spec with no extractors returns, so // treating a missing table as {} makes "this page has no sanderling runtime" diff --git a/internal/driver/chrome/logs_test.go b/internal/driver/chrome/logs_test.go new file mode 100644 index 0000000..4cb81ea --- /dev/null +++ b/internal/driver/chrome/logs_test.go @@ -0,0 +1,60 @@ +package chrome + +import ( + "testing" + + "github.com/chromedp/cdproto/runtime" +) + +// Every console verb has to land on the logcat scale driver.LogEntry declares: +// the runner fetches at "E" and the default properties count entries whose +// level equals "E", so a level spelled any other way is an error the spec never +// sees. A verb with no mapping is info, which is honest about severity without +// fabricating an error the page never logged. +func TestConsoleLevel(t *testing.T) { + cases := map[runtime.APIType]string{ + runtime.APITypeError: "E", + runtime.APITypeAssert: "E", + runtime.APITypeWarning: "W", + runtime.APITypeDebug: "D", + runtime.APITypeLog: "I", + runtime.APITypeInfo: "I", + runtime.APITypeTable: "I", + runtime.APIType("countReset"): "I", + } + for apiType, want := range cases { + if got := consoleLevel(apiType); got != want { + t.Errorf("consoleLevel(%q) = %q, want %q", apiType, got, want) + } + } +} + +// A level the scale does not name is unknown, not verbose. Ranking it below +// every threshold is what silently emptied the web log channel: the entries +// existed, the filter dropped them, and the run reported nothing. Evidence the +// filter cannot rank has to reach the caller, who can at least see it. +func TestMeetsLevel(t *testing.T) { + cases := []struct { + level string + minLevel string + want bool + }{ + {"E", "E", true}, + {"F", "E", true}, + {"W", "E", false}, + {"I", "E", false}, + {"D", "E", false}, + {"V", "E", false}, + {"W", "W", true}, + {"I", "W", false}, + {"D", "V", true}, + {"ERROR", "E", true}, + {"WARNING", "E", true}, + {"", "E", true}, + } + for _, tc := range cases { + if got := meetsLevel(tc.level, tc.minLevel); got != tc.want { + t.Errorf("meetsLevel(%q, %q) = %v, want %v", tc.level, tc.minLevel, got, tc.want) + } + } +} diff --git a/internal/driver/driver.go b/internal/driver/driver.go index 1ec667b..8d4acd6 100644 --- a/internal/driver/driver.go +++ b/internal/driver/driver.go @@ -110,6 +110,12 @@ type FocusedWindowChecker interface { FocusedWindowApp(ctx context.Context) (string, error) } +// LogEntry is one line of device log. Level is logcat's single-letter scale on +// every platform: "V", "D", "I", "W", "E", "F", ordered as written. The runner +// fetches at "E" and the default properties count entries whose level equals +// "E", so a driver that spells a level any other way empties the channel +// without failing anything: the entries never arrive and every property reading +// state.logs holds vacuously. type LogEntry struct { UnixMillis int64 Level string diff --git a/internal/driver/ioscompanion/device.go b/internal/driver/ioscompanion/device.go index 67536a7..b8dbdff 100644 --- a/internal/driver/ioscompanion/device.go +++ b/internal/driver/ioscompanion/device.go @@ -31,17 +31,22 @@ type DeviceOptions struct { BundleID string // AppPath is the .app bundle installed via devicectl for clear-state. AppPath string + // ClearState reinstalls the app while NewDevice runs, before the runner's + // test session exists. Clear state is a property of the driver rather than + // of a launch: see Launch. + ClearState bool // Output receives the runner session log path and driver warnings. Output io.Writer // DoubleTapGapMilliseconds overrides the synthesized double-tap gap. DoubleTapGapMilliseconds float64 // Test seams. Production leaves them nil and NewDevice wires the real - // build/spawn/tunnel/dial. - spawnRunner func(ctx context.Context, address string) (*exec.Cmd, error) - startTunnel func(ctx context.Context, hardwareUDID, localAddress, devicePort string) (io.Closer, error) - dialRunner func(address string) (transport.Companion, error) - pickAddress func() (string, error) + // build/spawn/tunnel/dial/devicectl. + spawnRunner func(ctx context.Context, address string) (*exec.Cmd, error) + startTunnel func(ctx context.Context, hardwareUDID, localAddress, devicePort string) (io.Closer, error) + dialRunner func(address string) (transport.Companion, error) + pickAddress func() (string, error) + reinstallApp func(ctx context.Context) error } // deviceStartupTimeout bounds the runner's startup once its hosting test @@ -61,6 +66,9 @@ func NewDevice(ctx context.Context, options DeviceOptions) (*Driver, error) { if options.CoreDeviceID == "" { return nil, errors.New("ios device: CoreDeviceID is required") } + if options.ClearState && options.BundleID == "" { + return nil, errors.New("ios device: clear-state needs BundleID: there is nothing to uninstall without it") + } output := options.Output if output == nil { output = io.Discard @@ -95,17 +103,24 @@ func NewDevice(ctx context.Context, options DeviceOptions) (*Driver, error) { } } if options.pickAddress != nil { - d.pickDeviceAddress = options.pickAddress + d.pickRunnerAddress = options.pickAddress } else { - d.pickDeviceAddress = pickLoopbackAddress + d.pickRunnerAddress = pickLoopbackAddress } // Device seams: clear-state reinstalls via devicectl; the container reset and // paste grant are simulator-only and become no-ops. The runner types - // natively, so no paste prompt is ever hit. - d.reinstallApp = d.devicectlReinstall + // natively, so no paste prompt is ever hit. Stopping the app before the + // clear is a no-op too: devicectl addresses processes by pid rather than by + // bundle, and the uninstall that is the device's only clear takes the + // running app with it, which is what a terminate here would be for. + d.reinstallApp = options.reinstallApp + if d.reinstallApp == nil { + d.reinstallApp = d.devicectlReinstall + } d.resetContainer = d.deviceResetContainerUnsupported d.grantPaste = func(context.Context) error { return nil } + d.terminateApp = func(context.Context) error { return nil } d.restart = d.respawnDevice d.processContext, d.processCancel = context.WithCancel(ctx) @@ -116,6 +131,14 @@ func NewDevice(ctx context.Context, options DeviceOptions) (*Driver, error) { } d.deviceLock = lock + if options.ClearState { + if err := d.clearAppState(ctx); err != nil { + d.Close() + return nil, err + } + d.clearedBundleID = options.BundleID + } + if err := d.bringUpDevice(ctx); err != nil { d.Close() return nil, err @@ -136,7 +159,7 @@ func NewDevice(ctx context.Context, options DeviceOptions) (*Driver, error) { // health. The build runs inside spawnRunner under the process context, so the // startup timeout only bounds the post-spawn wait, not the build. func (d *Driver) bringUpDevice(ctx context.Context) error { - address, err := d.pickDeviceAddress() + address, err := d.pickRunnerAddress() if err != nil { return err } @@ -205,8 +228,14 @@ func (d *Driver) respawnDevice(ctx context.Context) error { // devicectlReinstall uninstalls then installs the app bundle via devicectl, // keyed on the CoreDevice id. App lifecycle stays with devicectl: the runner's // own install path is simulator-specific. +// A failed uninstall ends the reinstall: installing over an app keeps its data, +// so clear-state would be reported without happening. Uninstalling an app that +// is not installed exits 0 ("App uninstalled." on a paired iPhone running iOS +// 26.5), so there is no benign failure here to sort out from a real one. func (d *Driver) devicectlReinstall(ctx context.Context) error { - _ = exec.CommandContext(ctx, "xcrun", "devicectl", "device", "uninstall", "app", "--device", d.coreDeviceID, d.bundleID).Run() + if output, err := exec.CommandContext(ctx, "xcrun", "devicectl", "device", "uninstall", "app", "--device", d.coreDeviceID, d.bundleID).CombinedOutput(); err != nil { + return fmt.Errorf("devicectl uninstall %s: %w: %s", d.bundleID, err, strings.TrimSpace(string(output))) + } output, err := exec.CommandContext(ctx, "xcrun", "devicectl", "device", "install", "app", "--device", d.coreDeviceID, d.appPath).CombinedOutput() if err != nil { return fmt.Errorf("devicectl install: %w: %s", err, strings.TrimSpace(string(output))) @@ -214,13 +243,11 @@ func (d *Driver) devicectlReinstall(ctx context.Context) error { return nil } -// deviceResetContainerUnsupported warns once that device clear-state needs an -// app path for a devicectl reinstall: there is no simulator-style data-container -// wipe on a physical device. +// deviceResetContainerUnsupported ends the run: there is no simulator-style +// data-container wipe on a physical device, so a clear-state with no app path +// to reinstall from cannot happen. Warning and carrying on hands the run every +// previous run's data while the flag says it started clean. func (d *Driver) deviceResetContainerUnsupported(context.Context) error { - if !d.clearStateWarned { - fmt.Fprintln(d.output, "clear-state on a physical device requires --ios-app-path for a reinstall; skipping (state not cleared)") - d.clearStateWarned = true - } - return nil + return errors.New("clear-state on a physical device requires --ios-app-path for a reinstall; " + + "there is no data-container wipe on a device, so the run would start on the previous run's state") } diff --git a/internal/driver/ioscompanion/device_test.go b/internal/driver/ioscompanion/device_test.go index 5cacfc0..15bf23a 100644 --- a/internal/driver/ioscompanion/device_test.go +++ b/internal/driver/ioscompanion/device_test.go @@ -6,6 +6,8 @@ import ( "io" "net" "os/exec" + "slices" + "strings" "testing" "github.com/priyanshujain/sanderling/internal/driver/ioscompanion/transport" @@ -81,6 +83,25 @@ func TestNewDeviceWiresRunnerOnlyMode(t *testing.T) { } } +// TestNewDeviceWiresTheAddressPickerEveryBringUpUses keeps the device driver +// whole. bringUpRunner reads its picker from a field rather than calling the +// package function, and NewDevice left that field nil, so the only thing +// standing between a device run and a nil call was which restart path ran. +func TestNewDeviceWiresTheAddressPickerEveryBringUpUses(t *testing.T) { + address := startLoopbackListener(t) + options := testDeviceOptions(address, newDeviceCompanion()) + options.HardwareUDID = "00008140-PICKER" + d, err := NewDevice(context.Background(), options) + if err != nil { + t.Fatalf("NewDevice: %v", err) + } + defer d.Close() + + if err := d.bringUpRunner(context.Background()); err != nil { + t.Fatalf("bringUpRunner: %v", err) + } +} + func TestNewDeviceRequiresIdentifiers(t *testing.T) { if _, err := NewDevice(context.Background(), DeviceOptions{CoreDeviceID: "x"}); err == nil { t.Fatal("missing HardwareUDID must error") @@ -152,17 +173,100 @@ func TestDeviceEraseAndPressKeyRouteThroughEditor(t *testing.T) { } } -func TestDeviceClearStateWithoutAppPathWarnsOnce(t *testing.T) { - output := &bytes.Buffer{} - d := &Driver{output: output, deviceMode: true} - d.resetContainer = d.deviceResetContainerUnsupported - for i := 0; i < 2; i++ { - if err := d.deviceResetContainerUnsupported(context.Background()); err != nil { - t.Fatal(err) +func TestNewDeviceReinstallsOnceBeforeTheRunnerSession(t *testing.T) { + address := startLoopbackListener(t) + probe := &clearStateProbe{} + options := testDeviceOptions(address, newDeviceCompanion()) + options.HardwareUDID = "00008140-CLEAR" + options.AppPath = "/tmp/Sample.app" + options.ClearState = true + options.reinstallApp = func(context.Context) error { probe.record("reinstall"); return nil } + spawn := options.spawnRunner + options.spawnRunner = func(ctx context.Context, runnerAddress string) (*exec.Cmd, error) { + probe.record("runner session") + return spawn(ctx, runnerAddress) + } + + d, err := NewDevice(context.Background(), options) + if err != nil { + t.Fatalf("NewDevice: %v", err) + } + defer d.Close() + if err := d.Launch(context.Background(), "", true, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + + want := []string{"reinstall", "runner session"} + if got := probe.recorded(); !slices.Equal(got, want) { + t.Fatalf("calls = %v, want %v: devicectl must reinstall once, before the runner's test session attaches", got, want) + } +} + +func TestDevicectlReinstallStopsWhenTheUninstallFails(t *testing.T) { + log := scriptedXcrun(t, `"devicectl device uninstall "*) echo "ERROR: Internal logic error: Connection was invalidated"; exit 1;; +"devicectl device install "*) :;;`) + d := &Driver{coreDeviceID: "CORE-DEVICE", bundleID: "app.example", appPath: "/tmp/Sample.app"} + + err := d.devicectlReinstall(context.Background()) + + if err == nil { + t.Fatal("devicectlReinstall reported success while app.example kept the data clear-state was asked to remove") + } + for _, want := range []string{"app.example", "Connection was invalidated"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("error %q does not quote %q", err, want) } } - if got := bytes.Count(output.Bytes(), []byte("requires --ios-app-path")); got != 1 { - t.Fatalf("warning emitted %d times, want once", got) + calls := xcrunCalls(t, log) + if slices.ContainsFunc(calls, func(call string) bool { return strings.HasPrefix(call, "devicectl device install") }) { + t.Errorf("xcrun calls = %v: installing over the app carries its data into the run", calls) + } +} + +func TestDevicectlReinstallProceedsWhenNothingIsInstalled(t *testing.T) { + log := scriptedXcrun(t, `"devicectl device uninstall "*) echo "App uninstalled.";; +"devicectl device install "*) :;;`) + d := &Driver{coreDeviceID: "CORE-DEVICE", bundleID: "app.example", appPath: "/tmp/Sample.app"} + + if err := d.devicectlReinstall(context.Background()); err != nil { + t.Fatalf("devicectlReinstall: %v", err) + } + + want := []string{ + "devicectl device uninstall app --device CORE-DEVICE app.example", + "devicectl device install app --device CORE-DEVICE /tmp/Sample.app", + } + if got := xcrunCalls(t, log); !slices.Equal(got, want) { + t.Fatalf("xcrun calls = %v, want %v", got, want) + } +} + +// TestNewDeviceRefusesClearStateWithoutAnAppPath keeps the device from starting +// a run whose clear-state cannot happen. There is no data-container wipe on a +// physical device, so without an app path to reinstall from, carrying on hands +// the run every previous run's data under a flag that says otherwise. +func TestNewDeviceRefusesClearStateWithoutAnAppPath(t *testing.T) { + address := startLoopbackListener(t) + options := testDeviceOptions(address, newDeviceCompanion()) + options.HardwareUDID = "00008140-NO-APP-PATH" + options.ClearState = true + spawned := false + spawn := options.spawnRunner + options.spawnRunner = func(ctx context.Context, runnerAddress string) (*exec.Cmd, error) { + spawned = true + return spawn(ctx, runnerAddress) + } + + d, err := NewDevice(context.Background(), options) + if err == nil { + d.Close() + t.Fatal("NewDevice returned a driver whose clear-state never happened") + } + if !strings.Contains(err.Error(), "--ios-app-path") { + t.Fatalf("err = %v, want it to name the flag that makes the clear possible", err) + } + if spawned { + t.Fatal("the run started anyway; a clear-state that cannot happen must end the run, not open it") } } diff --git a/internal/driver/ioscompanion/driver.go b/internal/driver/ioscompanion/driver.go index f0cd8ba..e186434 100644 --- a/internal/driver/ioscompanion/driver.go +++ b/internal/driver/ioscompanion/driver.go @@ -53,6 +53,14 @@ var shutdownGrace = 15 * time.Second // A variable so the timeout test can shrink it. var launchTimeout = 90 * time.Second +// launchRecoveryTimeout bounds the whole recovery a blown launch bound +// triggers, the session restart and the second attempt together. It keeps the +// launch path inside the three minutes testrun allows it, so what a user sees +// when the app really cannot be launched stays the driver's error rather than +// that backstop firing over the top of it. A variable so the bound test can +// shrink it. +var launchRecoveryTimeout = 60 * time.Second + // longPressHoldMilliseconds is how long LongPress holds the finger down. const longPressHoldMilliseconds = 600 @@ -65,16 +73,25 @@ type Options struct { // AppPath is the .app bundle directory. Required for clear-state reinstall; // when empty, clear state falls back to resetting the data container. AppPath string + // ClearState resets the app to first-launch state while New runs, before + // any automation session attaches. Clear state is a property of the driver + // rather than of a launch: see Launch. + ClearState bool // Output receives companion stdout and stderr plus driver warnings. Output io.Writer // DoubleTapGapMilliseconds overrides the synthesized double-tap gap. DoubleTapGapMilliseconds float64 - // spawnChild, dialCompanion, and pickAddress are test seams. Production - // leaves them nil and New wires the real extraction, spawn, and dial. - spawnChild func(ctx context.Context, address string) (*exec.Cmd, error) - dialCompanion func(address string) (transport.Companion, error) - pickAddress func() (string, error) + // These are test seams. Production leaves them nil and New wires the real + // extraction, spawn, dial and simctl calls. + spawnChild func(ctx context.Context, address string) (*exec.Cmd, error) + dialCompanion func(address string) (transport.Companion, error) + pickAddress func() (string, error) + spawnRunner func(ctx context.Context, address string) (*exec.Cmd, error) + dialRunner func(address string) (transport.Companion, error) + reinstallApp func(ctx context.Context) error + resetContainer func(ctx context.Context) error + terminateApp func(ctx context.Context) error } // Driver implements driver.DeviceDriver against an iOS simulator companion. @@ -85,6 +102,13 @@ type Driver struct { appPath string output io.Writer + // clearedBundleID names the app New (or NewDevice) reset to first-launch + // state before attaching, which is the only point in a run where clearing + // is safe. Launch refuses a clear-state request for anything else rather + // than reinstalling under a live session or reporting a reset that only + // ever reached another bundle. + clearedBundleID string + screenWidth int screenHeight int @@ -114,6 +138,10 @@ type Driver struct { // A seam so tests skip the simctl shell-outs. reinstallApp func(ctx context.Context) error + // terminateApp stops the app before its state is cleared. A seam so tests + // skip the simctl shell-out. + terminateApp func(ctx context.Context) error + // grantPaste pre-authorizes the app's pasteboard access. A seam so tests // skip the sqlite shell-out. grantPaste func(ctx context.Context) error @@ -136,15 +164,19 @@ type Driver struct { dialRunner func(address string) (transport.Companion, error) hybrid bool + // pickRunnerAddress hands every bring-up a free loopback port, on the + // simulator and the device alike. One field, so no path can be wired + // without it. + pickRunnerAddress func() (string, error) + // Device-mode fields. On the physical-device path d.companion is the runner // dialed over a usbmux tunnel, hybrid is false, and runnerClient is nil. // coreDeviceID feeds devicectl; tunnel is the in-process usbmux forwarder // bridging the host loopback port to the runner's device-side port. - deviceMode bool - coreDeviceID string - tunnel io.Closer - startTunnel func(ctx context.Context, hardwareUDID, localAddress, devicePort string) (io.Closer, error) - pickDeviceAddress func() (string, error) + deviceMode bool + coreDeviceID string + tunnel io.Closer + startTunnel func(ctx context.Context, hardwareUDID, localAddress, devicePort string) (io.Closer, error) // processContext owns the companion child's lifetime: it is derived from // New's context (so a canceled run still reaps the child) and canceled by @@ -189,6 +221,9 @@ func New(ctx context.Context, options Options) (*Driver, error) { if options.UniqueDeviceIdentifier == "" { return nil, errors.New("ios companion: UniqueDeviceIdentifier is required") } + if options.ClearState && options.BundleID == "" { + return nil, errors.New("ios companion: clear-state needs BundleID: there is nothing to uninstall or wipe without it") + } output := options.Output if output == nil { output = io.Discard @@ -206,6 +241,11 @@ func New(ctx context.Context, options Options) (*Driver, error) { doubleTapGapMilliseconds: gap, spawnChild: options.spawnChild, dial: options.dialCompanion, + spawnRunner: options.spawnRunner, + dialRunner: options.dialRunner, + reinstallApp: options.reinstallApp, + resetContainer: options.resetContainer, + terminateApp: options.terminateApp, hybrid: hybridCompanionEnabled(), } if driverInstance.spawnChild == nil { @@ -232,9 +272,17 @@ func New(ctx context.Context, options Options) (*Driver, error) { return nil, err } driverInstance.address = address + driverInstance.pickRunnerAddress = pickAddress driverInstance.restart = driverInstance.respawnAndRedial - driverInstance.resetContainer = driverInstance.resetDataContainer - driverInstance.reinstallApp = driverInstance.simctlReinstall + if driverInstance.resetContainer == nil { + driverInstance.resetContainer = driverInstance.resetDataContainer + } + if driverInstance.reinstallApp == nil { + driverInstance.reinstallApp = driverInstance.simctlReinstall + } + if driverInstance.terminateApp == nil { + driverInstance.terminateApp = driverInstance.simctlTerminate + } driverInstance.grantPaste = driverInstance.grantPasteboardAccess driverInstance.processContext, driverInstance.processCancel = context.WithCancel(ctx) @@ -245,6 +293,14 @@ func New(ctx context.Context, options Options) (*Driver, error) { } driverInstance.deviceLock = lock + if options.ClearState { + if err := driverInstance.clearAppState(ctx); err != nil { + driverInstance.Close() + return nil, err + } + driverInstance.clearedBundleID = options.BundleID + } + if err := driverInstance.bringUp(ctx); err != nil { driverInstance.Close() return nil, err @@ -358,7 +414,7 @@ func (d *Driver) bringUpRunner(ctx context.Context) error { // A fresh port every bring-up: after a restart the dying session's // listener may still answer on the old port and would satisfy the wait // below with a dead server. - address, err := pickLoopbackAddress() + address, err := d.pickRunnerAddress() if err != nil { return err } @@ -441,6 +497,26 @@ func isConnectionError(err error) bool { return false } +// isBudgetExpiry reports whether err is a call that outlived its bound rather +// than a failure the transport can name. Each transport says so differently: +// the runner wraps the context's error when cancellation has landed and the +// connection's i/o timeout when the deadline it armed from that context fires +// first, and the legacy companion returns a gRPC status. Only an expiry earns +// a session restart; an error the runner reports has already said what a fresh +// session would say. +func isBudgetExpiry(err error) bool { + if err == nil { + return false + } + if errors.Is(err, context.DeadlineExceeded) || errors.Is(err, os.ErrDeadlineExceeded) { + return true + } + if statusValue, ok := status.FromError(err); ok { + return statusValue.Code() == codes.DeadlineExceeded + } + return false +} + func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, env map[string]string) error { if bundleID != "" { d.bundleID = bundleID @@ -452,6 +528,14 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e // loudly rather than silently dropping the request. return errors.New("ios companion: launch with environment variables is unsupported on this backend") } + if clearState && (d.clearedBundleID == "" || d.clearedBundleID != d.bundleID) { + // Clearing here would uninstall and reinstall the app underneath a live + // automation session, which is what races FrontBoard's registration and + // leaves the session launching a bundle FrontBoard has not registered. + return fmt.Errorf("ios companion: clear-state must be requested when the driver is created (Options.ClearState) "+ + "for the bundle being launched; this backend cleared %q before its automation session existed, not %q", + d.clearedBundleID, d.bundleID) + } // Terminate first so the launch is a clean cold start regardless of the // app's prior state. A not-running app is not an error here. @@ -459,12 +543,6 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e return companion.Terminate(callCtx, d.bundleID) }) - if clearState { - if err := d.clearAppState(ctx); err != nil { - return err - } - } - // Grant the app pasteboard access before it runs so unicode input (which // must go through the pasteboard, since HID cannot express it) never trips // the iOS paste-permission prompt. clearState reinstall resets the grant, @@ -477,14 +555,55 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e } } - if err := d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error { - return companion.Launch(callCtx, d.bundleID, true) - }); err != nil { + if err := d.launchWithSessionRecovery(ctx); err != nil { return fmt.Errorf("launch %s: %w", d.bundleID, err) } return nil } +// launchWithSessionRecovery runs the launch RPC and, when it blows its own +// bound, replaces the session and launches again. +// +// A launch the simulator refuses, which is what a clear-state reinstall racing +// FrontBoard's registration produces, never comes back as an error: XCTest +// records the refusal as a test failure the runner cannot observe, then holds +// the session's main thread for about four minutes walking a diagnostic chain +// (a 120s accessibility wait, a spindump, an idle wait). So there is no error +// text to key a retry on, only the expired bound, and every later call queues +// behind the same wedge. Only a session that never served the refused launch +// can serve the retry, which is why this restarts rather than calls again. +func (d *Driver) launchWithSessionRecovery(ctx context.Context) error { + launch := func(callCtx context.Context, companion transport.Companion) error { + return companion.Launch(callCtx, d.bundleID, true) + } + err := d.lifecycleCall(ctx, launch) + // A caller whose own budget ran out gets no restart: the bound that expired + // was the caller's to spend, and the second attempt would inherit it dead. + if !isBudgetExpiry(err) || ctx.Err() != nil || d.restart == nil { + return err + } + fmt.Fprintf(d.output, "launch %s blew its %v bound (%v); restarting the session and launching once more\n", + d.bundleID, launchTimeout, err) + + // The restart runs under the driver's own lifetime context for the same + // reason withRecovery's does, while the second attempt stays on the + // caller's. Both end at one deadline, so a launch that already spent + // launchTimeout cannot then wait out a session cold start on top of it. + recoveryDeadline := time.Now().Add(launchRecoveryTimeout) + restartCtx := d.processContext + if restartCtx == nil { + restartCtx = ctx + } + restartCtx, cancelRestart := context.WithDeadline(restartCtx, recoveryDeadline) + defer cancelRestart() + if restartErr := d.restart(restartCtx); restartErr != nil { + return fmt.Errorf("session restart failed: %w (original: %v)", restartErr, err) + } + relaunchCtx, cancelRelaunch := context.WithDeadline(ctx, recoveryDeadline) + defer cancelRelaunch() + return d.lifecycleCall(relaunchCtx, launch) +} + // lifecycleCall runs an app lifecycle RPC against lifecycleCompanion under a // launchTimeout-bounded context, with the usual one-restart recovery. The // companion is resolved inside the retry so a restart's replacement client @@ -511,8 +630,14 @@ func (d *Driver) lifecycleCompanion() transport.Companion { // clearAppState resets the app to a first-launch state. With an app path it // uninstalls and reinstalls; without one it falls back to wiping the app's data -// container and warns once that a full reinstall needs the app path. +// container and warns once that a full reinstall needs the app path. Called +// only from construction, before any automation session is attached to the app. func (d *Driver) clearAppState(ctx context.Context) error { + // Nothing may be writing to the state while it goes, which is the ordering + // Launch used to hold: uninstall copes with a running app, deleting the + // data container out from under one does not. Best effort, because an app + // that is not running reports a failure that means nothing here. + _ = d.terminateApp(ctx) if d.appPath != "" { if err := d.reinstallApp(ctx); err != nil { return fmt.Errorf("reinstall %s: %w", d.appPath, err) @@ -529,8 +654,14 @@ func (d *Driver) clearAppState(ctx context.Context) error { // simctlReinstall uninstalls and reinstalls the app bundle via simctl. App // lifecycle stays with simctl: the companion's install RPC misreads current // simulator targets' architectures and rejects valid bundles. +// A failed uninstall ends the reinstall: `simctl install` over an installed app +// carries its data container across, so clear-state would be reported without +// happening. Uninstalling an app that is not installed exits 0, so there is no +// benign failure here to sort out from a real one. func (d *Driver) simctlReinstall(ctx context.Context) error { - _ = exec.CommandContext(ctx, "xcrun", "simctl", "uninstall", d.udid, d.bundleID).Run() + if output, err := exec.CommandContext(ctx, "xcrun", "simctl", "uninstall", d.udid, d.bundleID).CombinedOutput(); err != nil { + return fmt.Errorf("simctl uninstall %s: %w: %s", d.bundleID, err, strings.TrimSpace(string(output))) + } output, err := exec.CommandContext(ctx, "xcrun", "simctl", "install", d.udid, d.appPath).CombinedOutput() if err != nil { return fmt.Errorf("simctl install: %w: %s", err, strings.TrimSpace(string(output))) @@ -538,6 +669,18 @@ func (d *Driver) simctlReinstall(ctx context.Context) error { return nil } +// simctlTerminate stops the app under test. Launch used to terminate through +// the automation session before clearing; the clear now runs before any session +// exists, so simctl is what is left to stop the app with. An app that is not +// running reports a failure that means nothing to the caller, which is why +// clearAppState treats this as best effort. +func (d *Driver) simctlTerminate(ctx context.Context) error { + if output, err := exec.CommandContext(ctx, "xcrun", "simctl", "terminate", d.udid, d.bundleID).CombinedOutput(); err != nil { + return fmt.Errorf("simctl terminate %s: %w: %s", d.bundleID, err, strings.TrimSpace(string(output))) + } + return nil +} + // grantPasteboardAccess authorizes the app to read the pasteboard without the // iOS permission prompt, by writing an allow row into the simulator's privacy // (TCC) database. This is the simulator counterpart to `simctl privacy grant`, @@ -965,7 +1108,10 @@ func (d *Driver) WaitForIdle(ctx context.Context, _ time.Duration) error { } // RecentLogs returns no entries: the companion log RPC is a follow-up, so v1 -// reports an empty slice rather than failing. +// reports an empty slice rather than failing. Every property reading state.logs +// therefore holds vacuously on iOS and nothing says so; closing it means +// tailing idb's streaming log RPC and mapping os_log levels onto the +// single-letter scale driver.LogEntry declares. func (d *Driver) RecentLogs(_ context.Context, _ time.Time, _ string) ([]driver.LogEntry, error) { return []driver.LogEntry{}, nil } diff --git a/internal/driver/ioscompanion/driver_test.go b/internal/driver/ioscompanion/driver_test.go index 856af80..b6d13dd 100644 --- a/internal/driver/ioscompanion/driver_test.go +++ b/internal/driver/ioscompanion/driver_test.go @@ -18,10 +18,12 @@ import ( "testing" "time" + "google.golang.org/grpc" "google.golang.org/grpc/codes" "google.golang.org/grpc/status" "github.com/priyanshujain/sanderling/internal/driver" + "github.com/priyanshujain/sanderling/internal/driver/ioscompanion/companionpb" "github.com/priyanshujain/sanderling/internal/driver/ioscompanion/transport" ) @@ -183,47 +185,280 @@ func TestLaunchContinuesWhenGrantFails(t *testing.T) { } } -func TestLaunchClearStateReinstallsWithAppPath(t *testing.T) { +// clearStateProbe records, in order, the calls a run makes to reset the app and +// to bring the runner's automation session up. A reinstall recorded after the +// session is the ordering that races FrontBoard. +type clearStateProbe struct { + mutex sync.Mutex + events []string +} + +func (p *clearStateProbe) record(event string) { + p.mutex.Lock() + defer p.mutex.Unlock() + p.events = append(p.events, event) +} + +func (p *clearStateProbe) recorded() []string { + p.mutex.Lock() + defer p.mutex.Unlock() + out := make([]string, len(p.events)) + copy(out, p.events) + return out +} + +// clearStateOptions wires every seam a hybrid bring-up needs, so New runs its +// real sequence against fakes: no simulator, no simctl, no XCTest session. +func clearStateOptions(t *testing.T, probe *clearStateProbe, udid string, clearState bool) Options { + t.Helper() + t.Setenv("SANDERLING_SIMULATOR_COMPANION", "") + address := startLoopbackListener(t) + return Options{ + UniqueDeviceIdentifier: udid, + BundleID: "com.example.app", + ClearState: clearState, + Output: &bytes.Buffer{}, + pickAddress: func() (string, error) { return address, nil }, + spawnChild: func(context.Context, string) (*exec.Cmd, error) { return &exec.Cmd{}, nil }, + dialCompanion: func(string) (transport.Companion, error) { + return &fakeCompanion{accessibilityJSON: "[]"}, nil + }, + spawnRunner: func(context.Context, string) (*exec.Cmd, error) { + probe.record("runner session") + return &exec.Cmd{}, nil + }, + dialRunner: func(string) (transport.Companion, error) { + return &fakeCompanion{accessibilityJSON: "[]"}, nil + }, + reinstallApp: func(context.Context) error { probe.record("reinstall"); return nil }, + resetContainer: func(context.Context) error { probe.record("reset container"); return nil }, + terminateApp: func(context.Context) error { probe.record("stop app"); return nil }, + } +} + +func TestClearStateReinstallsOnceBeforeTheRunnerSession(t *testing.T) { + probe := &clearStateProbe{} + options := clearStateOptions(t, probe, "CLEAR-REINSTALL-UDID", true) + options.AppPath = "/tmp/Sample.app" + + d, err := New(context.Background(), options) + if err != nil { + t.Fatalf("New: %v", err) + } + defer d.Close() + if err := d.Launch(context.Background(), "", true, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + + want := []string{"stop app", "reinstall", "runner session"} + if got := probe.recorded(); !slices.Equal(got, want) { + t.Fatalf("calls = %v, want %v: the reinstall must run once, on a stopped app, before the automation session attaches", got, want) + } +} + +func TestClearStateWithoutAppPathWipesContainerBeforeTheRunnerSession(t *testing.T) { + probe := &clearStateProbe{} + output := &bytes.Buffer{} + options := clearStateOptions(t, probe, "CLEAR-CONTAINER-UDID", true) + options.Output = output + + d, err := New(context.Background(), options) + if err != nil { + t.Fatalf("New: %v", err) + } + defer d.Close() + if err := d.Launch(context.Background(), "", true, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + + want := []string{"stop app", "reset container", "runner session"} + if got := probe.recorded(); !slices.Equal(got, want) { + t.Fatalf("calls = %v, want %v: the fallback must wipe a stopped app's container once, before the session, and never reinstall", got, want) + } + if warnings := strings.Count(output.String(), "resetting the data container only"); warnings != 1 { + t.Fatalf("warning emitted %d times, want once", warnings) + } +} + +func TestWithoutClearStateTheAppIsLeftAlone(t *testing.T) { + probe := &clearStateProbe{} + options := clearStateOptions(t, probe, "NO-CLEAR-UDID", false) + options.AppPath = "/tmp/Sample.app" + + d, err := New(context.Background(), options) + if err != nil { + t.Fatalf("New: %v", err) + } + defer d.Close() + if err := d.Launch(context.Background(), "", false, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + + want := []string{"runner session"} + if got := probe.recorded(); !slices.Equal(got, want) { + t.Fatalf("calls = %v, want %v: a run that did not ask for clear state must not touch the install", got, want) + } +} + +func TestLaunchRefusesClearStateTheDriverWasNotBuiltFor(t *testing.T) { companion := &fakeCompanion{accessibilityJSON: "[]"} d := newTestDriver(companion) d.appPath = "/tmp/Sample.app" reinstalls := 0 d.reinstallApp = func(context.Context) error { reinstalls++; return nil } - if err := d.Launch(context.Background(), "", true, nil); err != nil { - t.Fatalf("Launch: %v", err) + + err := d.Launch(context.Background(), "", true, nil) + if err == nil || !strings.Contains(err.Error(), "clear-state") { + t.Fatalf("Launch err = %v, want a refusal naming clear-state", err) } - if reinstalls != 1 { - t.Fatalf("clear-state with app path must reinstall exactly once; got %d", reinstalls) + if reinstalls != 0 { + t.Fatalf("reinstalls = %d, want 0: a live session must never have the app reinstalled under it", reinstalls) } - if indexOf(companion.calls, "launch") < indexOf(companion.calls, "terminate") { - t.Fatalf("launch must still follow terminate; got %v", companion.calls) + if indexOf(companion.recorded(), "launch") >= 0 { + t.Fatalf("a refused launch must not reach the companion; got %v", companion.recorded()) } } -func TestLaunchClearStateFallbackWarnsOnce(t *testing.T) { +// TestLaunchRefusesClearStateForABundleItDidNotClear holds the guard to the +// fact it is guarding. A driver built to clear one bundle has cleared nothing +// for another, so reporting that launch as a clear-state launch is a reset the +// caller was told happened and did not. +func TestLaunchRefusesClearStateForABundleItDidNotClear(t *testing.T) { companion := &fakeCompanion{accessibilityJSON: "[]"} - output := &bytes.Buffer{} d := newTestDriver(companion) - d.output = output - resets := 0 - d.resetContainer = func(context.Context) error { resets++; return nil } + d.clearedBundleID = "com.example.app" - for i := 0; i < 2; i++ { - if err := d.Launch(context.Background(), "", true, nil); err != nil { - t.Fatalf("Launch %d: %v", i, err) + err := d.Launch(context.Background(), "com.other.app", true, nil) + + if err == nil || !strings.Contains(err.Error(), "clear-state") { + t.Fatalf("Launch err = %v, want a refusal naming clear-state: com.other.app was never cleared", err) + } + if indexOf(companion.recorded(), "launch") >= 0 { + t.Fatalf("a launch reporting a clear that never happened must not reach the companion; got %v", companion.recorded()) + } +} + +func TestNewRejectsClearStateWithoutBundleID(t *testing.T) { + probe := &clearStateProbe{} + options := clearStateOptions(t, probe, "NO-BUNDLE-UDID", true) + options.BundleID = "" + + if _, err := New(context.Background(), options); err == nil || !strings.Contains(err.Error(), "BundleID") { + t.Fatalf("New err = %v, want a refusal naming BundleID", err) + } + if got := probe.recorded(); len(got) != 0 { + t.Fatalf("calls = %v, want none: clearing an unnamed bundle would reinstall without resetting anything", got) + } +} + +// scriptedXcrun puts an xcrun on PATH that logs each invocation's arguments and +// answers from replies, a `case "$*" in` body, so a reinstall runs its real +// command sequence and the log holds what reached the tool. +func scriptedXcrun(t *testing.T, replies string) string { + t.Helper() + directory := t.TempDir() + log := filepath.Join(directory, "xcrun.log") + script := "#!/bin/sh\necho \"$*\" >> " + log + "\ncase \"$*\" in\n" + replies + "\nesac\n" + if err := os.WriteFile(filepath.Join(directory, "xcrun"), []byte(script), 0o755); err != nil { + t.Fatalf("write xcrun: %v", err) + } + t.Setenv("PATH", directory) + return log +} + +func xcrunCalls(t *testing.T, log string) []string { + t.Helper() + contents, err := os.ReadFile(log) + if err != nil { + if os.IsNotExist(err) { + return nil + } + t.Fatalf("read %s: %v", log, err) + } + return strings.Split(strings.TrimSpace(string(contents)), "\n") +} + +func TestSimctlReinstallStopsWhenTheUninstallFails(t *testing.T) { + log := scriptedXcrun(t, `"simctl uninstall "*) echo "Simulator device failed to uninstall app.example."; echo "Uninstall prohibited."; exit 22;; +"simctl install "*) :;;`) + d := &Driver{udid: "SIM-UDID", bundleID: "app.example", appPath: "/tmp/Sample.app"} + + err := d.simctlReinstall(context.Background()) + + if err == nil { + t.Fatal("simctlReinstall reported success while app.example kept the data clear-state was asked to remove") + } + for _, want := range []string{"app.example", "Uninstall prohibited."} { + if !strings.Contains(err.Error(), want) { + t.Errorf("error %q does not quote %q", err, want) } } - if resets != 2 { - t.Fatalf("resetContainer called %d times, want 2", resets) + if calls := xcrunCalls(t, log); slices.Contains(calls, "simctl install SIM-UDID /tmp/Sample.app") { + t.Errorf("xcrun calls = %v: installing over the app carries its data into the run", calls) } - warnings := strings.Count(output.String(), "resetting the data container only") - if warnings != 1 { - t.Fatalf("warning emitted %d times, want once", warnings) +} + +func TestSimctlReinstallProceedsWhenNothingIsInstalled(t *testing.T) { + log := scriptedXcrun(t, `"simctl uninstall "*) :;; +"simctl install "*) :;;`) + d := &Driver{udid: "SIM-UDID", bundleID: "app.example", appPath: "/tmp/Sample.app"} + + if err := d.simctlReinstall(context.Background()); err != nil { + t.Fatalf("simctlReinstall: %v", err) } - for _, call := range companion.calls { - if call == "install" || call == "uninstall" { - t.Fatalf("fallback path must not install/uninstall; got %v", companion.calls) - } + + want := []string{"simctl uninstall SIM-UDID app.example", "simctl install SIM-UDID /tmp/Sample.app"} + if got := xcrunCalls(t, log); !slices.Equal(got, want) { + t.Fatalf("xcrun calls = %v, want %v", got, want) + } +} + +// TestClearStateStopsTheAppBeforeWipingItsContainer covers the ordering Launch +// used to hold. simctl uninstall copes with a running app; deleting the data +// container out from under one does not, and the CI iOS leg passes no app path +// so it is the wipe that runs. A run whose previous run was interrupted finds +// the app still up. +func TestClearStateStopsTheAppBeforeWipingItsContainer(t *testing.T) { + container := t.TempDir() + stale := filepath.Join(container, "Documents") + if err := os.Mkdir(stale, 0o755); err != nil { + t.Fatal(err) + } + log := scriptedXcrun(t, `"simctl terminate "*) :;; +"simctl get_app_container "*) echo `+container+`;;`) + d := &Driver{udid: "SIM-UDID", bundleID: "app.example", output: &bytes.Buffer{}} + d.terminateApp = d.simctlTerminate + d.resetContainer = d.resetDataContainer + + if err := d.clearAppState(context.Background()); err != nil { + t.Fatalf("clearAppState: %v", err) + } + + want := []string{ + "simctl terminate SIM-UDID app.example", + "simctl get_app_container SIM-UDID app.example data", + } + if got := xcrunCalls(t, log); !slices.Equal(got, want) { + t.Fatalf("xcrun calls = %v, want %v: the app was still writing to the container being deleted", got, want) + } + if _, err := os.Stat(stale); !os.IsNotExist(err) { + t.Fatalf("stat %s = %v, want the previous run's state gone", stale, err) + } +} + +// TestClearStateSurvivesAnAppThatIsNotRunning holds the terminate to best +// effort. simctl exits non-zero when there is nothing to stop, and a first run +// on a fresh simulator must not fail on it. +func TestClearStateSurvivesAnAppThatIsNotRunning(t *testing.T) { + container := t.TempDir() + scriptedXcrun(t, `"simctl terminate "*) echo "No matching processes belonging to bundle identifier app.example"; exit 3;; +"simctl get_app_container "*) echo `+container+`;;`) + d := &Driver{udid: "SIM-UDID", bundleID: "app.example", output: &bytes.Buffer{}} + d.terminateApp = d.simctlTerminate + d.resetContainer = d.resetDataContainer + + if err := d.clearAppState(context.Background()); err != nil { + t.Fatalf("clearAppState: %v: an app that is not running is not a failure to clear", err) } } @@ -952,6 +1187,301 @@ func TestLaunchLeavesATighterCallerDeadlineAlone(t *testing.T) { } } +// blownBudgetShape is one way a transport the driver launches through reports +// a lifecycle call outliving its budget. +type blownBudgetShape struct { + name string + err error +} + +// blownBudgetShapes drives every such transport against a server that never +// answers and returns the error each one really produces. The runner transport +// has two: wrapTransport wraps the context's own error once cancellation has +// landed, and the connection's i/o timeout when the deadline it armed from +// that context fires first. The legacy transport reports the same expiry as a +// gRPC status. Only the first satisfies errors.Is(err, context.DeadlineExceeded), +// so a fake that returns ctx.Err() raw shows the driver a recovery that two +// thirds of production can never reach. +func blownBudgetShapes(t *testing.T) []blownBudgetShape { + t.Helper() + return []blownBudgetShape{ + {"runner context deadline", runnerContextDeadlineError(t)}, + {"runner connection deadline", silentRunnerLaunchError(t, connectionDeadlineOnly{time.Now().Add(100 * time.Millisecond)})}, + {"legacy grpc deadline", silentGRPCLaunchError(t)}, + } +} + +// runnerContextDeadlineError is the shape the runner transport produces once +// the context's own cancellation has landed. +func runnerContextDeadlineError(t *testing.T) error { + t.Helper() + ctx, cancel := context.WithDeadline(context.Background(), time.Now().Add(-time.Second)) + defer cancel() + return silentRunnerLaunchError(t, ctx) +} + +// connectionDeadlineOnly carries a deadline the runner transport arms the +// connection with, while its own cancellation never lands. That is the race +// wrapTransport's second branch exists for: the connection's deadline fires +// while ctx.Err() is still nil. +type connectionDeadlineOnly struct{ deadline time.Time } + +func (c connectionDeadlineOnly) Deadline() (time.Time, bool) { return c.deadline, true } +func (c connectionDeadlineOnly) Done() <-chan struct{} { return nil } +func (c connectionDeadlineOnly) Err() error { return nil } +func (c connectionDeadlineOnly) Value(any) any { return nil } + +// silentRunnerLaunchError returns what the real runner transport produces for a +// launch nobody ever answers. +func silentRunnerLaunchError(t *testing.T, ctx context.Context) error { + t.Helper() + listener, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + closed := make(chan struct{}) + go func() { + conn, acceptErr := listener.Accept() + if acceptErr != nil { + return + } + defer conn.Close() + <-closed + }() + companion, err := transport.DialRunner(listener.Addr().String(), "SIM-UDID", "com.example.app") + if err != nil { + listener.Close() + t.Fatal(err) + } + t.Cleanup(func() { + close(closed) + _ = companion.Close() + listener.Close() + }) + + launchErr := companion.Launch(ctx, "com.example.app", true) + if launchErr == nil { + t.Fatal("the runner transport reported a launch no server ever answered") + } + return launchErr +} + +// silentCompanionServer is the legacy companion with a launch that never +// answers, so the caller's own deadline is what ends the call. +type silentCompanionServer struct { + companionpb.UnimplementedCompanionServiceServer +} + +func (silentCompanionServer) Launch(stream grpc.BidiStreamingServer[companionpb.LaunchRequest, companionpb.LaunchResponse]) error { + <-stream.Context().Done() + return stream.Context().Err() +} + +// silentGRPCLaunchError returns what the legacy transport produces for the same +// launch, which SANDERLING_SIMULATOR_COMPANION=legacy still runs on. +func silentGRPCLaunchError(t *testing.T) error { + t.Helper() + listener, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + server := grpc.NewServer() + companionpb.RegisterCompanionServiceServer(server, silentCompanionServer{}) + go server.Serve(listener) + t.Cleanup(server.Stop) + + companion, err := transport.Dial(listener.Addr().String()) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { companion.Close() }) + + ctx, cancel := context.WithTimeout(context.Background(), 250*time.Millisecond) + defer cancel() + launchErr := companion.Launch(ctx, "com.example.app", true) + if launchErr == nil { + t.Fatal("the legacy transport reported a launch the companion never answered") + } + return launchErr +} + +// wedgedUntilRestartCompanion models the session a refused launch leaves +// behind: the refusal is never reported, no later launch is answered until the +// session itself is replaced, and the expired bound reaches the driver in +// whatever shape its transport gives it. +type wedgedUntilRestartCompanion struct { + fakeCompanion + blownBudget error + mutex sync.Mutex + replaced bool + attempted int +} + +func (w *wedgedUntilRestartCompanion) replaceSession() { + w.mutex.Lock() + defer w.mutex.Unlock() + w.replaced = true +} + +func (w *wedgedUntilRestartCompanion) launchAttempts() int { + w.mutex.Lock() + defer w.mutex.Unlock() + return w.attempted +} + +func (w *wedgedUntilRestartCompanion) Launch(ctx context.Context, _ string, _ bool) error { + w.mutex.Lock() + w.attempted++ + replaced := w.replaced + w.mutex.Unlock() + if replaced { + return nil + } + <-ctx.Done() + return w.blownBudget +} + +// TestLaunchReplacesTheSessionAfterALaunchBlowsItsBound covers the FrontBoard +// race: a clear-state reinstall the simulator has not finished registering +// makes the session refuse the launch, and XCTest answers that refusal with +// minutes of diagnostics instead of an error, so the bound expires and every +// later call queues behind the same wedge. Calling launch again on that session +// cannot work; the run only recovers if the session is replaced first. The +// recovery has to fire on every shape the driver's transports report that +// expiry in, because which one arrives is a race the driver does not control. +func TestLaunchReplacesTheSessionAfterALaunchBlowsItsBound(t *testing.T) { + for _, shape := range blownBudgetShapes(t) { + t.Run(shape.name, func(t *testing.T) { + previous := launchTimeout + launchTimeout = 100 * time.Millisecond + defer func() { launchTimeout = previous }() + + companion := &wedgedUntilRestartCompanion{blownBudget: shape.err} + output := &bytes.Buffer{} + d := newTestDriver(companion) + d.output = output + restarts := 0 + d.restart = func(context.Context) error { + restarts++ + companion.replaceSession() + return nil + } + + if err := d.Launch(context.Background(), "", false, nil); err != nil { + t.Fatalf("Launch: %v", err) + } + if restarts != 1 { + t.Fatalf("session restarts = %d, want exactly 1 (the session reported %v)", restarts, shape.err) + } + if attempts := companion.launchAttempts(); attempts != 2 { + t.Fatalf("launch attempts = %d, want 2: one that wedged and one on the replaced session", attempts) + } + if !strings.Contains(output.String(), "restarting the session") { + t.Fatalf("the recovery was silent, so a run that needed it never says so; output was %q", output.String()) + } + }) + } +} + +// TestLaunchBoundsTheSessionRestartItTriggers keeps the recovery inside a +// budget of its own. The restart deliberately runs on the driver's lifetime +// context rather than the caller's, so without a deadline a session that never +// comes back would hang the launch path exactly the way #73 stopped it hanging. +func TestLaunchBoundsTheSessionRestartItTriggers(t *testing.T) { + previousLaunch, previousRecovery := launchTimeout, launchRecoveryTimeout + launchTimeout = 100 * time.Millisecond + launchRecoveryTimeout = 200 * time.Millisecond + defer func() { launchTimeout, launchRecoveryTimeout = previousLaunch, previousRecovery }() + + d := newTestDriver(&wedgedUntilRestartCompanion{blownBudget: runnerContextDeadlineError(t)}) + d.restart = func(restartCtx context.Context) error { + <-restartCtx.Done() + return restartCtx.Err() + } + + done := make(chan error, 1) + go func() { done <- d.Launch(context.Background(), "", false, nil) }() + select { + case err := <-done: + if err == nil || !strings.Contains(err.Error(), "session restart failed") { + t.Fatalf("err = %v, want the failed restart named", err) + } + case <-time.After(10 * time.Second): + t.Fatal("Launch never returned: a session that never comes back hangs the launch path") + } +} + +// TestLaunchKeepsTheSessionWhenTheCallersOwnDeadlineExpires holds the recovery +// to the driver's own bound. Spending a session restart on a caller that has +// already run out of budget cannot produce a launch, only a later failure. +func TestLaunchKeepsTheSessionWhenTheCallersOwnDeadlineExpires(t *testing.T) { + previous := launchTimeout + launchTimeout = 30 * time.Second + defer func() { launchTimeout = previous }() + + companion := &wedgedUntilRestartCompanion{blownBudget: runnerContextDeadlineError(t)} + d := newTestDriver(companion) + restarts := 0 + d.restart = func(context.Context) error { + restarts++ + companion.replaceSession() + return nil + } + + ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond) + defer cancel() + if err := d.Launch(ctx, "", false, nil); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("err = %v, want a deadline-exceeded error", err) + } + if restarts != 0 { + t.Fatalf("session restarts = %d, want 0", restarts) + } +} + +// refusedLaunchCompanion answers a launch the way the runner does once it +// checks the app's state after activating it: promptly, naming the app and the +// state it reached, over a session that is still serving. +type refusedLaunchCompanion struct { + fakeCompanion + attempts int +} + +func (r *refusedLaunchCompanion) Launch(context.Context, string, bool) error { + r.attempts++ + return errors.New(`runner launch: failed("com.example.app is not running after launch")`) +} + +// TestLaunchKeepsTheSessionWhenTheRunnerNamesTheRefusal separates a launch that +// answers from a launch that never does. The session restart is the only +// recovery from a wedged session, and it costs a cold start; a runner that +// reports the app's state has already said what a fresh session would say, so +// restarting to hear it again only delays the error and hides the app under it. +func TestLaunchKeepsTheSessionWhenTheRunnerNamesTheRefusal(t *testing.T) { + companion := &refusedLaunchCompanion{} + output := &bytes.Buffer{} + d := newTestDriver(companion) + d.output = output + restarts := 0 + d.restart = func(context.Context) error { + restarts++ + return nil + } + + err := d.Launch(context.Background(), "", false, nil) + if err == nil || !strings.Contains(err.Error(), "com.example.app is not running after launch") { + t.Fatalf("err = %v, want the runner's refusal reaching the caller intact", err) + } + if restarts != 0 { + t.Fatalf("session restarts = %d, want 0: a refusal the runner reported is not a wedged session", restarts) + } + if companion.attempts != 1 { + t.Fatalf("launch attempts = %d, want 1: relaunching an app the runner just refused cannot launch it", companion.attempts) + } + if strings.Contains(output.String(), "restarting the session") { + t.Fatalf("the driver announced a recovery it must not spend here; output was %q", output.String()) + } +} + // newLockTestOptions builds New options that dial a seamed companion, so the // device-lock tests exercise New without spawning anything. func newLockTestOptions(t *testing.T, udid string) Options { diff --git a/internal/hierarchy/hierarchy.go b/internal/hierarchy/hierarchy.go index cc3277a..09c735c 100644 --- a/internal/hierarchy/hierarchy.go +++ b/internal/hierarchy/hierarchy.go @@ -12,7 +12,8 @@ // descPrefix: - starts-with on content-desc / accessibilityText // // Object selectors (multi-attribute AND, element-scoped or global): -// { attr: value, ... } - all key/value pairs must match; substring / boolean semantics +// { attr: value, ... } - all key/value pairs must match, each key resolved by +// the same rule its string form above uses // // Path queries (global scan only, string form): // > > ... - each segment matched within subtree of previous match @@ -445,7 +446,14 @@ func matchAttr(element *Element, attr, value string) bool { return false } -// matchSelector returns true when all filters in sel match the element (AND semantics). +// matchSelector returns true when all filters in sel match the element (AND +// semantics). Each filter goes through matchAttr, the same rule the string form +// resolves a "kind:value" segment by, so {id: "Submit"} and "id:Submit" can +// never resolve to different elements. Reaching the attribute map directly here +// made the object form skip the kind arms entirely: id, desc and descPrefix +// name no attribute any producer writes, so those keys matched NOTHING through +// an object selector while the string form matched, and every property over the +// missing element passed vacuously. func matchSelector(element *Element, sel Selector) bool { for _, f := range sel.Filters { if !matchAttr(element, f.Attr, f.Value) { diff --git a/internal/hierarchy/hierarchy_test.go b/internal/hierarchy/hierarchy_test.go index 4265df1..8a2632c 100644 --- a/internal/hierarchy/hierarchy_test.go +++ b/internal/hierarchy/hierarchy_test.go @@ -1471,3 +1471,83 @@ func TestParseUnreadableFlagKeepsTheRestOfTheTree(t *testing.T) { t.Errorf("stored UnreadableFlags = %d, want 1: the trace has to carry it", decoded.UnreadableFlags) } } + +// selectorFormsDump carries one node per id shape a real dump produces, plus +// nodes carrying a description in the ", " form the desc rule knows about and a +// text the text rule matches on a substring. +const selectorFormsDump = `{ + "attributes": {"resource-id": "root", "bounds": "[0,0,400,800]"}, + "children": [ + {"attributes": {"resource-id": "BareThing", "bounds": "[0,0,100,50]"}, "children": []}, + {"attributes": {"resource-id": "com.example.app:id/AndroidThing", "bounds": "[0,50,100,100]"}, + "children": []}, + {"attributes": {"accessibilityIdentifier": "IosThing", "bounds": "[0,100,100,150]"}, + "children": []}, + {"attributes": {"resource-id": "Described", "content-desc": "Save, button", "bounds": "[0,150,100,200]"}, + "children": []}, + {"attributes": {"resource-id": "Labelled", "text": "Total balance", "bounds": "[0,200,100,250]"}, + "children": []} + ] +}` + +// TestSelectorFormsResolveTheSameElement holds the two selector forms a spec can +// write to ONE rule per key. A spec reaches these through state.ax.find: a +// string goes to FindNode, an object to FindBySelector, and the two ran +// different matchers. `id` has a kind arm that knows an Android resource id is +// package-qualified (com.example.app:id/Thing) and that a spec names the bare +// tail; the object form had no such arm and looked for a literal `id` attribute +// no producer writes, so {id: "Thing"} silently matched nothing on every +// platform while "id:Thing" matched. `desc` and `descPrefix` had the same +// split. A selector that resolves nothing makes every property over it +// vacuously true, which is the failure that reports a green run while checking +// nothing. +func TestSelectorFormsResolveTheSameElement(t *testing.T) { + tree, err := Parse(selectorFormsDump) + if err != nil { + t.Fatal(err) + } + for _, test := range []struct { + key string + value string + want string + }{ + {"id", "BareThing", "BareThing"}, + // A spec names the tail; an Android dump carries the package prefix. + {"id", "AndroidThing", "com.example.app:id/AndroidThing"}, + {"id", "com.example.app:id/AndroidThing", "com.example.app:id/AndroidThing"}, + {"id", "IosThing", "IosThing"}, + {"desc", "Save, button", "Described"}, + // The ", " form an accessibility label takes when a role is appended. + {"desc", "Save", "Described"}, + {"descPrefix", "Sav", "Described"}, + {"text", "Total", "Labelled"}, + {"resource-id", "BareThing", "BareThing"}, + {"testTag", "IosThing", "IosThing"}, + } { + t.Run(test.key+":"+test.value, func(t *testing.T) { + stringForm := test.key + ":" + test.value + fromString := tree.FindNode(stringForm) + if fromString == nil { + t.Fatalf("the string form %q resolved nothing", stringForm) + } + if fromString.ResourceID != test.want { + t.Fatalf("the string form resolved %q, want %q", fromString.ResourceID, test.want) + } + fromObject := tree.Root.FindBySelector( + Selector{Filters: []AttrFilter{{Attr: test.key, Value: test.value}}}, + ) + if fromObject == nil { + t.Fatalf( + "the object form {%s: %q} resolved nothing while %q resolved %q", + test.key, test.value, stringForm, fromString.ResourceID, + ) + } + if fromObject != fromString { + t.Errorf( + "one selector, two answers: {%s: %q} resolved %q and %q resolved %q", + test.key, test.value, fromObject.ResourceID, stringForm, fromString.ResourceID, + ) + } + }) + } +} diff --git a/internal/ltl/formula.go b/internal/ltl/formula.go index 759564d..a40afc6 100644 --- a/internal/ltl/formula.go +++ b/internal/ltl/formula.go @@ -86,8 +86,9 @@ type NextFormula struct { } // EventuallyFormula obliges its inner formula to hold at some step within the -// given bound. An unbounded eventually never triggers a violation within a -// finite run. +// given bound. An unbounded eventually that never fires is violated when the +// run ends, with the reason "eventually never satisfied", so an eventually over +// a state the run may not reach is red on every run that does not reach it. // // When Duration is non-zero and Deadline is the zero time, the evaluator // resolves the absolute deadline on first reduction using the observation diff --git a/internal/runner/composition_reread_test.go b/internal/runner/composition_reread_test.go new file mode 100644 index 0000000..942e602 --- /dev/null +++ b/internal/runner/composition_reread_test.go @@ -0,0 +1,509 @@ +package runner + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/priyanshujain/sanderling/internal/driver" + mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock" + "github.com/priyanshujain/sanderling/internal/trace" +) + +// homeWithRows is one settled route whose list holds rows. A row arriving +// between two reads is what a Compose lazy list mounting over several frames +// looks like from the runner's side. +func homeWithRows(rows int) string { + var children strings.Builder + for row := range rows { + fmt.Fprintf(&children, + `,{"attributes":{"resource-id":"TxnRow%d","class":"android.view.View"},"children":[]}`, row) + } + return fmt.Sprintf( + `{"attributes":{"resource-id":"HomeScreen","class":"android.view.View"},"children":[ + {"attributes":{"resource-id":"TxnList","class":"android.view.View"},"children":[]}%s + ]}`, children.String()) +} + +// composesLateDriver answers the paired Snapshot with the frame the step +// records and the hierarchy read that follows with a tree that has grown a row, +// for the first composingReads reads of the run. After that both reads describe +// the same screen. +type composesLateDriver struct { + *mockdriver.Driver + composingReads int64 + reads atomic.Int64 +} + +func (d *composesLateDriver) Snapshot(context.Context) (string, driver.Image, error) { + return homeWithRows(1), driver.Image{PNG: []byte("png"), Width: 1, Height: 1}, nil +} + +func (d *composesLateDriver) Hierarchy(context.Context) (string, error) { + if d.reads.Add(1) <= d.composingReads { + return homeWithRows(2), nil + } + return homeWithRows(1), nil +} + +// A route can settle before its content composes, so a tree read the moment the +// route arrives can describe a screen that is still filling in. Verifying that +// step compares a half-composed frame against a settled one and convicts an app +// that did nothing wrong. Two reads a read apart see it happening, and the step +// they disagree on is one the verifier must never be handed. +// +// The always-false property is the witness: it fires on the first step the +// verifier evaluates, so the step index of its violation says exactly which +// step reached the verifier. +func TestRunner_AStepWhoseTreeChangedBetweenReadsIsNotVerified(t *testing.T) { + run := func(t *testing.T, composingReads int64) (Summary, string) { + t.Helper() + state := newHarnessWithSpec(t, violationSpec) + device := &composesLateDriver{Driver: state.mock, composingReads: composingReads} + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 3, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 3 { + t.Fatalf("steps = %d, want 3", summary.Steps) + } + return summary, state.writer.Directory() + } + + t.Run("the step it changed on is skipped, the next one is judged", func(t *testing.T) { + summary, directory := run(t, 1) + if len(summary.Violations) != 1 { + t.Fatalf("violations = %v, want exactly one", summary.Violations) + } + violation := summary.Violations[0] + if violation.Properties[0] != "balanceNonNegative" { + t.Fatalf("violated %v, want balanceNonNegative", violation.Properties) + } + if violation.StepIndex != 2 { + t.Errorf("the property first judged step %d, want 2; the verifier was handed "+ + "a screen that grew a row while the runner was reading it", + violation.StepIndex) + } + if summary.SkippedVerification != 1 { + t.Errorf("the run reports %d step(s) judged by nothing, want 1", + summary.SkippedVerification) + } + // Skipped is not lost: the step is still recorded, screenshot and all, + // so the run can be replayed over the frame nothing judged. + steps := traceSteps(t, directory) + if len(steps) != 3 { + t.Fatalf("trace holds %d step(s), want 3", len(steps)) + } + if len(steps[0].Violations) != 0 { + t.Errorf("step 1 recorded violations %v; it was never verified", steps[0].Violations) + } + screenshot := filepath.Join(directory, "screenshots", "step-00001.png") + if _, err := os.Stat(screenshot); err != nil { + t.Errorf("expected the skipped step's screenshot at %s: %v", screenshot, err) + } + }) + + // The control. Two reads that agree must verify as they always did, + // otherwise the case above is just a runner that verifies nothing. + t.Run("two reads that agree verify the step", func(t *testing.T) { + summary, _ := run(t, 0) + if len(summary.Violations) != 1 { + t.Fatalf("violations = %v, want exactly one", summary.Violations) + } + if got := summary.Violations[0].StepIndex; got != 1 { + t.Errorf("the property first judged step %d, want 1; a settled screen must be "+ + "verified on the step it was read", got) + } + }) +} + +// submitsOnTapDriver commits commitsPerTap transactions on every tap, shows the +// running total in the tree, and grows a row under the hierarchy read that +// follows the paired Snapshot: on one chosen step, on the run of steps from +// composingRead through composingThrough, or on every one of them. +type submitsOnTapDriver struct { + *mockdriver.Driver + commitsPerTap int64 + composingRead int64 + composingThrough int64 + everyRead bool + reads atomic.Int64 + committed atomic.Int64 +} + +func (d *submitsOnTapDriver) Tap(context.Context, int, int) error { return d.commit() } +func (d *submitsOnTapDriver) TapSelector(context.Context, string) error { return d.commit() } + +func (d *submitsOnTapDriver) commit() error { + d.committed.Add(d.commitsPerTap) + return nil +} + +func (d *submitsOnTapDriver) Snapshot(context.Context) (string, driver.Image, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil +} + +func (d *submitsOnTapDriver) Hierarchy(context.Context) (string, error) { + read := d.reads.Add(1) + composing := d.everyRead || read == d.composingRead || + (read >= d.composingRead && read <= d.composingThrough) + if composing { + return fmt.Sprintf(homeWithTxnCountComposing, d.committed.Load()), nil + } + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), nil +} + +// The same tree with one more row in it, which is what the reread sees while +// the screen is still filling in. +const homeWithTxnCountComposing = `{"attributes":{"resource-id":"HomeScreen"},"children":[ + {"attributes":{"resource-id":"TxnCount","text":"%d"},"children":[]}, + {"attributes":{"resource-id":"TxnSubmit","bounds":"[40,80,240,160]"},"children":[],"clickable":true,"enabled":true}, + {"attributes":{"resource-id":"TxnRowLate"},"children":[]} +]}` + +// Skipping a step is only free if nothing the spec needs goes missing with it. +// The action a step applies is reported to the spec on the NEXT step the +// verifier accepts, so a skipped step in between swallows the action before it: +// the transaction it committed still turns up in the next reading, and +// submitCommitsOneTransactionPerAction sees a rise nothing in its window +// accounts for. That is the conviction #77 and #78 are about, arriving through +// the skip rather than through the runner's report. +// +// So a frame the verifier will not look at is not one to act on either, which +// is also what #75 asked for: the fuzzer must not tap into a screen that is +// still filling in. +func TestRunner_ASkippedStepDoesNotSwallowTheActionBeforeIt(t *testing.T) { + spec := specWithFolioPredicates(t) + + run := func(t *testing.T, composingRead int64) (Summary, int64) { + t.Helper() + state := newHarnessWithSpec(t, spec) + device := &submitsOnTapDriver{ + Driver: state.mock, + commitsPerTap: 1, + composingRead: composingRead, + } + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 3, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 3 { + t.Fatalf("steps = %d, want 3", summary.Steps) + } + return summary, device.committed.Load() + } + + t.Run("a submit is not lost to the step that follows it", func(t *testing.T) { + summary, committed := run(t, 2) + if summary.SkippedVerification != 1 { + t.Fatalf("the run skipped %d step(s), want 1; the reread never fired, so this "+ + "proves nothing", summary.SkippedVerification) + } + if committed == 0 { + t.Fatal("the device committed nothing; a runner that never acts passes this " + + "test without meaning anything") + } + if len(summary.Violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v\n"+ + "one transaction per submit rose, and a submit went unreported because "+ + "the step after it was skipped", summary.Violations) + } + }) + + // The control: with nothing composing, every step is verified and the same + // app is judged clean, so the case above is not just a runner that stopped + // judging. + t.Run("every step verified, same app, no violation", func(t *testing.T) { + summary, committed := run(t, 0) + if summary.SkippedVerification != 0 { + t.Fatalf("the run skipped %d step(s), want 0", summary.SkippedVerification) + } + if committed != 3 { + t.Fatalf("the device committed %d transaction(s), want 3", committed) + } + if len(summary.Violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v", summary.Violations) + } + }) +} + +// One skipped step is held; a run of them has to be held too. A bound that lets +// the runner act again while the verifier is still being skipped puts back the +// exact swallow the hold exists to prevent, only later: the action drawn on the +// step past the bound overwrites the one the hold was carrying, and the carried +// action is never reported to any spec. +// +// The screen composes on steps 3 through 5 of 6 and the device commits one +// transaction per tap throughout, so the property has a clean pair to judge +// (step 2 to step 6) and nothing in between it can be told about except the +// action step 2 applied. +func TestRunner_ARunOfSkippedStepsReportsEveryActionItApplied(t *testing.T) { + spec := specWithFolioPredicates(t) + + run := func(t *testing.T, commitsPerTap int64) (Summary, int64) { + t.Helper() + state := newHarnessWithSpec(t, spec) + device := &submitsOnTapDriver{ + Driver: state.mock, + commitsPerTap: commitsPerTap, + composingRead: 3, + composingThrough: 5, + } + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 6, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 6 { + t.Fatalf("steps = %d, want 6", summary.Steps) + } + if summary.SkippedVerification != 3 { + t.Fatalf("the run skipped %d step(s), want 3; the reread never fired across "+ + "the run this test is about", summary.SkippedVerification) + } + return summary, device.committed.Load() + } + + t.Run("no submit is lost to the run of skipped steps", func(t *testing.T) { + summary, committed := run(t, 1) + if committed == 0 { + t.Fatal("the device committed nothing; a runner that never acts passes this " + + "test without meaning anything") + } + if len(summary.Violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v\n"+ + "one transaction per submit rose, and a submit went unreported because "+ + "the runner acted on a step the verifier skipped", summary.Violations) + } + if committed != 3 { + t.Errorf("the device committed %d transaction(s), want 3: one per verified "+ + "step (1, 2 and 6) and none from a step nothing would judge", committed) + } + }) + + // The control. Without it a green above proves nothing: a property handed + // no comparable pair is silently vacuous and reports the same empty list. + t.Run("two transactions per tap still convicts across the same run", func(t *testing.T) { + summary, _ := run(t, 2) + if len(summary.Violations) == 0 { + t.Fatal("the counting property missed a double submit; the skipped steps left " + + "it with nothing to judge, so the case above proves nothing") + } + if got := summary.Violations[0].Properties[0]; got != "submitCommitsOneTransactionPerAction" { + t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", + summary.Violations[0].Properties) + } + }) +} + +// A screen that changes shape under every pair of reads (a live list, a spinner +// mounting and unmounting) costs the run its actions: an action applied onto it +// would be the one the next verified step never hears about, and there is no +// next verified step. What the run must not do is come back green off that, +// which is what the "judged by nothing" count and the run's outcome are for. +func TestRunner_AScreenThatNeverSettlesActsOnNothingAndSaysSo(t *testing.T) { + state := newHarnessWithSpec(t, specWithFolioPredicates(t)) + device := &submitsOnTapDriver{Driver: state.mock, commitsPerTap: 1, everyRead: true} + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 5, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 5 { + t.Fatalf("steps = %d, want 5; the run stalled instead of finishing its budget", + summary.Steps) + } + if summary.SkippedVerification != 5 { + t.Fatalf("the run verified some step of a screen that never settled: skipped %d of 5", + summary.SkippedVerification) + } + if got := device.committed.Load(); got != 0 { + t.Errorf("the fuzzer applied %d action(s) onto a screen no property would judge; "+ + "each one is an action no spec will ever be told about", got) + } +} + +// homeWithTicker is one settled route holding a total that ticks and a button +// whose measured bounds shift under it. The nodes, their ids and their classes +// are the same in every rendering of it. +const homeWithTicker = `{"attributes":{"resource-id":"HomeScreen","class":"android.view.View"},"children":[ + {"attributes":{"resource-id":"Total","class":"android.widget.TextView","text":"%s"},"children":[]}, + {"attributes":{"resource-id":"TxnSubmit","class":"android.widget.Button","bounds":"%s"},"children":[]} +]}` + +// The same route with a node in it that was not there a read ago. +const homeWithTickerAndRow = `{"attributes":{"resource-id":"HomeScreen","class":"android.view.View"},"children":[ + {"attributes":{"resource-id":"Total","class":"android.widget.TextView","text":"120.00"},"children":[]}, + {"attributes":{"resource-id":"TxnSubmit","class":"android.widget.Button","bounds":"[40,80,240,160]"},"children":[]}, + {"attributes":{"resource-id":"TxnRowLate","class":"android.view.View"},"children":[]} +]}` + +// rereadsDriver answers the paired Snapshot with one fixed tree and the +// hierarchy read that follows with another, so a test can say exactly what +// moved between the two reads the detector compares. +type rereadsDriver struct { + *mockdriver.Driver + snapshotTree string + rereadTree string +} + +func (d *rereadsDriver) Snapshot(context.Context) (string, driver.Image, error) { + return d.snapshotTree, driver.Image{}, nil +} + +func (d *rereadsDriver) Hierarchy(context.Context) (string, error) { + return d.rereadTree, nil +} + +// What the two reads are compared ON is the whole feature. Comparing the values +// in the tree instead of the nodes in it would fire on every step of a screen +// with a total on it or a measure pass in flight, and a runner that skips every +// step verifies nothing while reporting no violations: green and vacuous, which +// is a worse answer than the composition the comparison set out to catch. +// +// The always-false property is the witness: it fires on the first step that +// reaches the verifier, so its presence and its step index say whether the step +// was judged at all. +func TestRunner_OnlyAChangeOfShapeCostsAStepItsVerdict(t *testing.T) { + settled := fmt.Sprintf(homeWithTicker, "120.00", "[40,80,240,160]") + cases := []struct { + name string + reread string + verified bool + }{ + { + name: "a total that ticked between the two reads", + reread: fmt.Sprintf(homeWithTicker, "121.00", "[40,80,240,160]"), + verified: true, + }, + { + name: "a measure pass that moved the button", + reread: fmt.Sprintf(homeWithTicker, "120.00", "[40,84,240,164]"), + verified: true, + }, + { + name: "a node that was not in the tree a read ago", + reread: homeWithTickerAndRow, + verified: false, + }, + } + + for _, testCase := range cases { + t.Run(testCase.name, func(t *testing.T) { + state := newHarnessWithSpec(t, violationSpec) + device := &rereadsDriver{ + Driver: state.mock, + snapshotTree: settled, + rereadTree: testCase.reread, + } + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 3, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 3 { + t.Fatalf("steps = %d, want 3", summary.Steps) + } + + if !testCase.verified { + if summary.SkippedVerification != 3 { + t.Errorf("the run judged %d of 3 steps whose tree grew a node between "+ + "the two reads; a screen still composing must reach no property", + 3-summary.SkippedVerification) + } + if len(summary.Violations) != 0 { + t.Errorf("a skipped step reached the verifier anyway: %v", + summary.Violations) + } + return + } + if summary.SkippedVerification != 0 { + t.Fatalf("the run judged nothing: %d of 3 steps were skipped over a tree "+ + "whose nodes never changed", summary.SkippedVerification) + } + if len(summary.Violations) != 1 { + t.Fatalf("violations = %v, want exactly one", summary.Violations) + } + if got := summary.Violations[0].StepIndex; got != 1 { + t.Errorf("the property first judged step %d, want 1", got) + } + }) + } +} + +type traceLine struct { + Step int `json:"step"` + Violations []string `json:"violations"` + ExtractorChanges map[string]trace.ExtractorChange `json:"extractor_changes"` +} + +func traceSteps(t *testing.T, directory string) []traceLine { + t.Helper() + body, err := os.ReadFile(filepath.Join(directory, "trace.jsonl")) + if err != nil { + t.Fatal(err) + } + var steps []traceLine + for _, raw := range bytes.Split(bytes.TrimSpace(body), []byte("\n")) { + var line traceLine + if err := json.Unmarshal(raw, &line); err != nil { + t.Fatalf("decode trace line: %v", err) + } + steps = append(steps, line) + } + return steps +} diff --git a/internal/runner/element_extractor_trace_test.go b/internal/runner/element_extractor_trace_test.go index 52dee59..4eb7611 100644 --- a/internal/runner/element_extractor_trace_test.go +++ b/internal/runner/element_extractor_trace_test.go @@ -4,6 +4,7 @@ import ( "bufio" "context" "encoding/json" + "fmt" "os" "path/filepath" "testing" @@ -12,124 +13,109 @@ import ( "github.com/priyanshujain/sanderling/internal/trace" ) -// elementExtractorSpec records accessibility elements directly, the shape an -// author reaches for when the question is "what was on screen at this step": -// one element and the list a generator would have been offered. +// elementExtractorSpec reads a live ax element, the shape every field and +// button in examples/folio/sanderling/spec.ts is extracted with. The property +// is false the moment the field is on screen, so the run records a witness +// whose only interesting content is that element. const elementExtractorSpec = ` -import { actions, extract } from "@sanderling/spec"; -extract("field", state => state.ax.find({ testTag: "LoginEmail" })); -extract("rows", state => state.ax.findAll({ testTag: "Row" })); -globalThis.properties = {}; +import { actions, always, extract } from "@sanderling/spec"; +const amountField = extract("amountField", s => s.ax.find({ "resource-id": "TxnAmountField" })); +globalThis.properties = { + noAmountField: always(() => amountField.current === undefined), +}; globalThis.actions = actions(() => []); ` -const elementExtractorHierarchy = `{ - "attributes": {"class": "android.widget.LinearLayout", "bounds": "[0,0,1080,2340]"}, +const amountFieldTreeJSON = `{ + "attributes": {"resource-id": "root", "bounds": "[0,0,400,800]"}, + "enabled": true, "children": [ - {"attributes": {"resource-id": "LoginEmail", "class": "android.widget.EditText", - "text": "you@example.com", "bounds": "[10,20,200,60]"}, "children": []}, - {"attributes": {"resource-id": "Row", "class": "android.widget.TextView", - "text": "first", "bounds": "[0,100,1080,200]"}, "children": []}, - {"attributes": {"resource-id": "Row", "class": "android.widget.TextView", - "text": "second", "bounds": "[0,200,1080,300]"}, "children": []} + {"attributes": {"resource-id": "TxnAmountField", "text": "199", "bounds": "[0,100,400,160]"}, + "editable": true, "enabled": true, "children": []} ] }` -// TestRunner_TraceRecordsElementValuedExtractors pins that an extractor holding -// an accessibility element reaches the trace. Elements carry host functions -// (find/findAll), which json.Marshal refuses; the encoder used to answer nil -// and the diff then emitted no entry, so the run finished clean with the -// extractor missing from every step and no error anywhere. +// TestRunner_TraceRecordsElementValuedExtractors is the guard on the artifact a +// person opens to decide whether a conviction is real. An element-valued +// extractor used to reach the trace as null on the goja hosts (ios, android): +// its exported value carries the element's find/findAll host functions, which +// json.Marshal refuses, so the encoding failed and both the per-step diff and +// the witness recorded nothing. A witness that reads null for the field the +// property fired on describes a state the property could not have fired in, +// which is worse than a blank. func TestRunner_TraceRecordsElementValuedExtractors(t *testing.T) { state := newHarnessWithSpec(t, elementExtractorSpec) - state.mock.HierarchyJSON = elementExtractorHierarchy + state.mock.HierarchyJSON = amountFieldTreeJSON ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) defer cancel() - if _, err := Run(ctx, Options{ - Duration: 100 * time.Millisecond, + summary, err := Run(ctx, Options{ + Duration: time.Hour, IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, Driver: state.mock, Verifier: state.verifier, TraceWriter: state.writer, - }); err != nil { + }) + if err != nil { t.Fatalf("Run: %v", err) } - - field, rows := elementChangesFromTrace(t, state.directory) - if field == nil { - t.Fatal("no extractor change for the element-valued extractor reached the trace") - } - if rows == nil { - t.Fatal("no extractor change for the element-list extractor reached the trace") + if !containsProperty(summary.Violations, "noAmountField") { + t.Fatalf("noAmountField did not violate, so the element never reached a predicate: %v", + summary.Violations) } - var element map[string]any - if err := json.Unmarshal(field, &element); err != nil { - t.Fatalf("field value is not a JSON object: %v (%s)", err, field) - } - if got := element["id"]; got != "LoginEmail" { - t.Errorf("field.id = %v, want LoginEmail", got) - } - if got := element["text"]; got != "you@example.com" { - t.Errorf("field.text = %v, want you@example.com", got) - } - if got := element["class"]; got != "android.widget.EditText" { - t.Errorf("field.class = %v, want android.widget.EditText", got) - } - bounds, ok := element["bounds"].(map[string]any) - if !ok { - t.Fatalf("field.bounds missing or not an object: %v", element["bounds"]) - } - if bounds["left"] != float64(10) || bounds["top"] != float64(20) || - bounds["right"] != float64(200) || bounds["bottom"] != float64(60) { - t.Errorf("field.bounds = %v, want left/top/right/bottom 10/20/200/60", bounds) - } - for _, key := range []string{"find", "findAll"} { - if _, present := element[key]; present { - t.Errorf("field carries the host function %q into the trace", key) - } - } - - var list []map[string]any - if err := json.Unmarshal(rows, &list); err != nil { - t.Fatalf("rows value is not a JSON array: %v (%s)", err, rows) - } - if len(list) != 2 { - t.Fatalf("rows recorded %d elements, want 2", len(list)) - } - if list[0]["text"] != "first" || list[1]["text"] != "second" { - t.Errorf("rows recorded %v, want the two Row elements in tree order", list) - } -} - -// elementChangesFromTrace returns the first recorded value of each extractor. -func elementChangesFromTrace(t *testing.T, directory string) (field, rows json.RawMessage) { - t.Helper() - file, err := os.Open(filepath.Join(directory, "trace.jsonl")) + file, err := os.Open(filepath.Join(state.writer.Directory(), "trace.jsonl")) if err != nil { t.Fatal(err) } defer file.Close() + type traceLine struct { + Step int `json:"step"` + ExtractorChanges map[string]trace.ExtractorChange `json:"extractor_changes"` + Witnesses map[string]trace.Witness `json:"witnesses"` + } + changes, witnesses := 0, 0 scanner := bufio.NewScanner(file) scanner.Buffer(make([]byte, 0, 64*1024), 8*1024*1024) for scanner.Scan() { - var line struct { - ExtractorChanges map[string]trace.ExtractorChange `json:"extractor_changes"` - } + var line traceLine if err := json.Unmarshal(scanner.Bytes(), &line); err != nil { t.Fatalf("trace line decode: %v", err) } - if change, ok := line.ExtractorChanges["field"]; ok && field == nil { - field = change.Curr + if change, ok := line.ExtractorChanges["amountField"]; ok { + changes++ + assertAmountField(t, fmt.Sprintf("step %d extractor_changes", line.Step), change.Curr) } - if change, ok := line.ExtractorChanges["rows"]; ok && rows == nil { - rows = change.Curr + for name, witness := range line.Witnesses { + witnesses++ + assertAmountField(t, fmt.Sprintf("step %d %s witness", line.Step, name), + witness.Extractors["amountField"]) } } if err := scanner.Err(); err != nil { t.Fatalf("scan trace: %v", err) } - return field, rows + if changes == 0 { + t.Error("amountField never appears in extractor_changes; the element the run read is not in the trace") + } + if witnesses == 0 { + t.Error("no witness reached the trace; nothing was compared") + } +} + +// assertAmountField reads the recorded element the way a person opening the +// trace would: the field's text is the number the property was judged on. +func assertAmountField(t *testing.T, where string, recorded json.RawMessage) { + t.Helper() + var element struct { + Text string `json:"text"` + } + if err := json.Unmarshal(recorded, &element); err != nil { + t.Fatalf("%s: decode %s: %v", where, recorded, err) + } + if element.Text != "199" { + t.Errorf("%s: recorded element is %s, want its text to read 199", where, recorded) + } } diff --git a/internal/runner/foreground_guard_last_action_test.go b/internal/runner/foreground_guard_last_action_test.go new file mode 100644 index 0000000..426f672 --- /dev/null +++ b/internal/runner/foreground_guard_last_action_test.go @@ -0,0 +1,365 @@ +package runner + +import ( + "context" + "fmt" + "path/filepath" + "sync/atomic" + "testing" + "time" + + "github.com/priyanshujain/sanderling/internal/driver" + mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock" +) + +const guardedBundleID = "app.folio" + +// committingDevice is a device whose submit taps commit transactions the next +// hierarchy read shows, and which can say how many it has committed so a test +// can prove the taps landed before reading anything into a verdict. +type committingDevice interface { + driver.DeviceDriver + commits() int64 +} + +// leavesForegroundAfterSubmitDriver is the condition the app-scope guard exists +// for: the submit tap lands and commits, and the app is no longer the +// foreground app by the time the next step looks. Folio's transactions are in +// sqlite, so the commit survives the relaunch and the next reading shows it. +type leavesForegroundAfterSubmitDriver struct { + *mockdriver.Driver + commitsPerTap int64 + committed atomic.Int64 + away atomic.Bool +} + +func (d *leavesForegroundAfterSubmitDriver) Tap(context.Context, int, int) error { + return d.commitThenLeave() +} + +func (d *leavesForegroundAfterSubmitDriver) TapSelector(context.Context, string) error { + return d.commitThenLeave() +} + +func (d *leavesForegroundAfterSubmitDriver) commitThenLeave() error { + d.committed.Add(d.commitsPerTap) + d.away.Store(true) + return nil +} + +func (d *leavesForegroundAfterSubmitDriver) commits() int64 { return d.committed.Load() } + +func (d *leavesForegroundAfterSubmitDriver) Launch( + ctx context.Context, + bundleID string, + clearState bool, + env map[string]string, +) error { + d.away.Store(false) + return d.Driver.Launch(ctx, bundleID, clearState, env) +} + +func (d *leavesForegroundAfterSubmitDriver) ForegroundApp(context.Context) (string, error) { + if d.away.Load() { + return "com.android.launcher", nil + } + return guardedBundleID, nil +} + +func (d *leavesForegroundAfterSubmitDriver) FocusedWindowApp(ctx context.Context) (string, error) { + return d.ForegroundApp(ctx) +} + +func (d *leavesForegroundAfterSubmitDriver) Snapshot(context.Context) (string, driver.Image, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil +} + +// The runner reads both per step and compares them, so a device that answered +// them off different trees would make every step of this test transitional and +// judged by nothing. The sidecar serves both off one read path (snapshotTree) +// for the same reason; this one answers them off the same commit count. +func (d *leavesForegroundAfterSubmitDriver) Hierarchy(context.Context) (string, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), nil +} + +// obscuredAfterSubmitDriver is the other half of the same guard: the app stays +// the resumed activity, but a system window (the notification shade) owns the +// focused window when the next step looks, and the guard presses back to +// collapse it. +type obscuredAfterSubmitDriver struct { + *mockdriver.Driver + commitsPerTap int64 + committed atomic.Int64 + obscured atomic.Bool +} + +func (d *obscuredAfterSubmitDriver) Tap(context.Context, int, int) error { + return d.commitThenObscure() +} + +func (d *obscuredAfterSubmitDriver) TapSelector(context.Context, string) error { + return d.commitThenObscure() +} + +func (d *obscuredAfterSubmitDriver) commitThenObscure() error { + d.committed.Add(d.commitsPerTap) + d.obscured.Store(true) + return nil +} + +func (d *obscuredAfterSubmitDriver) commits() int64 { return d.committed.Load() } + +func (d *obscuredAfterSubmitDriver) PressKey(ctx context.Context, key string) error { + if key == "back" { + d.obscured.Store(false) + } + return d.Driver.PressKey(ctx, key) +} + +func (d *obscuredAfterSubmitDriver) ForegroundApp(context.Context) (string, error) { + return guardedBundleID, nil +} + +func (d *obscuredAfterSubmitDriver) FocusedWindowApp(context.Context) (string, error) { + if d.obscured.Load() { + return "com.android.systemui", nil + } + return guardedBundleID, nil +} + +func (d *obscuredAfterSubmitDriver) Snapshot(context.Context) (string, driver.Image, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil +} + +func (d *obscuredAfterSubmitDriver) Hierarchy(context.Context) (string, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), nil +} + +// runTwoSubmitSteps drives two steps of the shipped folio counting property +// against a device that commits on every tap, and hands back what the property +// decided. Both steps have to run: the first arms the comparison, the second is +// where the guard fires and the pair is judged. +func runTwoSubmitSteps( + t *testing.T, + state *harness, + device committingDevice, + commitsPerTap int64, +) []ViolationRecord { + t.Helper() + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + BundleID: guardedBundleID, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 2 { + t.Fatalf("steps = %d, want 2; the run never reached the step that judges the pair", + summary.Steps) + } + if got := device.commits(); got != commitsPerTap*2 { + t.Fatalf("the device committed %d transaction(s), want %d; the taps never reached it", + got, commitsPerTap*2) + } + return summary.Violations +} + +func countMockActions(state *harness, kind mockdriver.ActionKind, key string) int { + count := 0 + for _, action := range state.mock.Actions() { + if action.Kind != kind { + continue + } + if key != "" && action.Key != key { + continue + } + count++ + } + return count +} + +func specWithFolioPredicates(t *testing.T) string { + t.Helper() + predicates, err := filepath.Abs("../../examples/folio/sanderling/predicates.ts") + if err != nil { + t.Fatal(err) + } + return fmt.Sprintf(submitCountingSpecTemplate, predicates) +} + +// A relaunch is not proof that nothing ran before it. The submit was dispatched +// and confirmed; what the relaunch changed is that the app restarted between +// the two readings the property compares. Reporting "no action" for it hands +// submitCommitsOneTransactionPerAction a transaction rise of one against a +// window of zero submits, which is the conviction #77 fixed for the apply-error +// path, manufactured here out of the scope guard instead. +func TestRunner_ARelaunchDoesNotConvictTheSubmitCountingProperty(t *testing.T) { + spec := specWithFolioPredicates(t) + + run := func(t *testing.T, commitsPerTap int64) []ViolationRecord { + t.Helper() + state := newHarnessWithSpec(t, spec) + device := &leavesForegroundAfterSubmitDriver{ + Driver: state.mock, + commitsPerTap: commitsPerTap, + } + violations := runTwoSubmitSteps(t, state, device, commitsPerTap) + if countMockActions(state, mockdriver.ActionLaunch, "") == 0 { + t.Fatal("the app was never relaunched, so the guard this test is about never ran") + } + return violations + } + + t.Run("one transaction per tap is not a double submit", func(t *testing.T) { + if violations := run(t, 1); len(violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v\n"+ + "one transaction rose against a submit the runner confirmed, and the "+ + "spec was told no action happened because the app was relaunched", + violations) + } + }) + + // The control. Without it a green above proves nothing: a property that + // never sees a comparable pair is silently vacuous and reports the same + // empty violation list. + t.Run("two transactions per tap still convicts", func(t *testing.T) { + violations := run(t, 2) + if len(violations) == 0 { + t.Fatal("the counting property missed a double submit; the harness never " + + "put the property in a position to fire, so the case above proves nothing") + } + if violations[0].Properties[0] != "submitCommitsOneTransactionPerAction" { + t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", violations[0].Properties) + } + }) +} + +// The same hole through the guard's other branch. A system window holding the +// focus says nothing about whether the tap under it ran: it was dispatched, and +// what nobody can say afterwards is whether the app received it. That is the +// unknown `applied` already carries, and it counts toward the submits a window +// could hold. Reporting no action instead convicts the app of a transaction +// with no cause. +func TestRunner_AnOverlayDoesNotConvictTheSubmitCountingProperty(t *testing.T) { + spec := specWithFolioPredicates(t) + + run := func(t *testing.T, commitsPerTap int64) []ViolationRecord { + t.Helper() + state := newHarnessWithSpec(t, spec) + device := &obscuredAfterSubmitDriver{ + Driver: state.mock, + commitsPerTap: commitsPerTap, + } + violations := runTwoSubmitSteps(t, state, device, commitsPerTap) + if countMockActions(state, mockdriver.ActionPressKey, "back") == 0 { + t.Fatal("the overlay was never dismissed, so the guard this test is about never ran") + } + if countMockActions(state, mockdriver.ActionLaunch, "") != 0 { + t.Fatal("a resumed-but-obscured app must not be relaunched") + } + return violations + } + + t.Run("one transaction per tap is not a double submit", func(t *testing.T) { + if violations := run(t, 1); len(violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v\n"+ + "one transaction rose against a submit the runner dispatched, and the "+ + "spec was told no action happened because a system window took the focus", + violations) + } + }) + + t.Run("two transactions per tap still convicts", func(t *testing.T) { + violations := run(t, 2) + if len(violations) == 0 { + t.Fatal("the counting property missed a double submit; the harness never " + + "put the property in a position to fire, so the case above proves nothing") + } + if violations[0].Properties[0] != "submitCommitsOneTransactionPerAction" { + t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", violations[0].Properties) + } + }) +} + +// reportedActionSpec puts what the runner told the spec about the last action +// into an extractor, so a test can read it out of the trace. `applied` and +// `relaunched` have no other producer: the runner's two guard writes are the +// only thing that ever sets them, and every spec-side guard built on them (see +// acrossRelaunch and confirmedApplied in the folio predicates) reads nothing +// else. A regression in either write leaves those guards permanently off with +// no property anywhere able to notice. +const reportedActionSpec = ` +import { actions, always, extract, Tap } from "@sanderling/spec"; +const reportedAction = extract("reportedAction", state => { + const last = state.lastAction; + if (last == null) return "none"; + const dispatch = last.applied === true ? "applied" : "unconfirmed"; + const process = last.relaunched === true ? "relaunched" : "same-process"; + return dispatch + "/" + process; +}); +globalThis.properties = { + theGuardTheRunnerRanReachesTheSpec: always( + () => reportedAction.current !== "applied/same-process", + ), +}; +globalThis.actions = actions(() => [Tap({ on: "id:TxnSubmit" })]); +` + +// runReportingTheGuard drives two steps against a device whose submit tap trips +// one of the foreground guards, and hands back what the spec read off +// state.lastAction on the step the guard fired. +func runReportingTheGuard(t *testing.T, device committingDevice, state *harness) string { + t.Helper() + if violations := runTwoSubmitSteps(t, state, device, 1); len(violations) != 0 { + t.Errorf("the spec was told the action ran untouched by any guard: %v", violations) + } + steps := traceSteps(t, state.writer.Directory()) + if len(steps) != 2 { + t.Fatalf("trace holds %d step(s), want 2", len(steps)) + } + change, ok := steps[1].ExtractorChanges["reportedAction"] + if !ok { + t.Fatalf("step 2 recorded no reading of the reported action: %+v", steps[1]) + } + return string(change.Curr) +} + +func TestRunner_TheSpecIsToldTheAppWasRelaunchedUnderTheAction(t *testing.T) { + state := newHarnessWithSpec(t, reportedActionSpec) + device := &leavesForegroundAfterSubmitDriver{Driver: state.mock, commitsPerTap: 1} + + reported := runReportingTheGuard(t, device, state) + + if countMockActions(state, mockdriver.ActionLaunch, "") == 0 { + t.Fatal("the app was never relaunched, so the write this test is about never ran") + } + if reported != `"applied/relaunched"` { + t.Errorf("the spec read %s off state.lastAction, want \"applied/relaunched\"; "+ + "a property relaxed across a relaunch cannot fire on a run that never "+ + "tells it one happened", reported) + } +} + +func TestRunner_TheSpecIsToldAnObscuredActionWasNotConfirmed(t *testing.T) { + state := newHarnessWithSpec(t, reportedActionSpec) + device := &obscuredAfterSubmitDriver{Driver: state.mock, commitsPerTap: 1} + + reported := runReportingTheGuard(t, device, state) + + if countMockActions(state, mockdriver.ActionPressKey, "back") == 0 { + t.Fatal("the overlay was never dismissed, so the write this test is about never ran") + } + if reported != `"unconfirmed/same-process"` { + t.Errorf("the spec read %s off state.lastAction, want "+ + "\"unconfirmed/same-process\"; a system window held the focused window, "+ + "so whether the app received the tap is exactly what nobody can say", + reported) + } +} diff --git a/internal/runner/runner.go b/internal/runner/runner.go index f26528a..d0f1817 100644 --- a/internal/runner/runner.go +++ b/internal/runner/runner.go @@ -57,6 +57,11 @@ type Summary struct { EndTime time.Time Steps int Violations []ViolationRecord + // SkippedVerification counts the steps whose tree was still moving when it + // was read, so no property judged them. A green run that skipped most of + // its steps checked almost nothing, and nothing else in the output would + // say so. + SkippedVerification int // UnsupportedVerbs lists verbs the picker requested that the platform // could not dispatch, deduped, so the report can flag a spec exercising // gestures this target does not support. @@ -105,6 +110,7 @@ func Run(ctx context.Context, options Options) (Summary, error) { _, pageExtractors := extractorSource.(webSource) exceptionReporter, _ := options.Driver.(driver.ExceptionReporter) navigationReporter, _ := options.Driver.(driver.NavigationReporter) + rereadHierarchy := driverIsAndroid(ctx, options, logger) summary := Summary{StartTime: time.Now()} deadline := summary.StartTime.Add(options.Duration) @@ -126,8 +132,23 @@ func Run(ctx context.Context, options Options) (Summary, error) { // backed out of (or otherwise left) the app, relaunch it before we // observe or act, so properties never evaluate against a foreign app // and actions never land outside the app. - if ensureForeground(ctx, options, logger, stepIndex) { - lastAction = nil + // + // What the guard did is reported to the spec on the action it followed, + // because dropping that action says "nothing ran between these two + // readings" and the runner has no business saying that: the action ran, + // and a property told otherwise convicts the app of an effect with no + // cause. See foreground_guard_last_action_test.go. + guard := ensureForeground(ctx, options, logger, stepIndex) + if lastAction != nil { + switch guard { + case foregroundRelaunched: + lastAction.Relaunched = true + case foregroundOverlayDismissed: + // A system window owned the focused window, so whether the app + // itself ever received this action is exactly the unknown + // Applied already has a state for. + lastAction.Applied = false + } } // Hierarchy, metrics, and logs are independent device reads. Run @@ -150,7 +171,8 @@ func Run(ctx context.Context, options Options) (Summary, error) { // screenshot describe the same frame, then re-fetches the pair // while the tree still looks transitional. g.Go(func() error { - tree, screenshotPNG, transitional, hierarchyErr = fetchSyncedState(gctx, options, logger, si) + tree, screenshotPNG, transitional, hierarchyErr = fetchSyncedState( + gctx, options, logger, si, rereadHierarchy) return nil }) g.Go(func() error { @@ -159,7 +181,7 @@ func Run(ctx context.Context, options Options) (Summary, error) { }) logSince := lastLogTime g.Go(func() error { - logs = collectLogs(gctx, options.Driver, logSince) + logs = collectLogs(gctx, options.Driver, logger, si, logSince) return nil }) // All goroutines write to local variables and return nil, so the Wait @@ -197,8 +219,10 @@ func Run(ctx context.Context, options Options) (Summary, error) { screen = tree.Elements[0].Screen } - // Transitional trees describe a NavHost mid cross-fade. Pushing - // one would poison the verifier's previous/current extractor + // A transitional tree is one nothing can vouch for: a NavHost mid + // cross-fade, a screen that changed shape between two reads, or a + // hierarchy that came back empty. Pushing one would poison the + // verifier's previous/current extractor // advance, so the next clean step would compare against this // transient state and emit false-positive violations. We still // record the step (hierarchy + screenshot) for replay-side @@ -223,10 +247,11 @@ func Run(ctx context.Context, options Options) (Summary, error) { // behind the hierarchy fetch; the fetch is what decides whether this // step counts at all, so it has to go first. // - // lastAction is the same value PushSnapshot hands the goja state - // below: the two engines evaluate this step against one action. + // lastAction and logs are the same values PushSnapshot hands the + // goja state below: the two engines evaluate this step against one + // action and one set of log entries. overridesCtx, overridesCancel := context.WithTimeout(ctx, observationTimeout) - v8Overrides, overridesErr := extractorSource.ExtractorOverrides(overridesCtx, lastAction) + v8Overrides, overridesErr := extractorSource.ExtractorOverrides(overridesCtx, lastAction, logs) overridesCancel() if overridesErr != nil { // Not a warning. Without the page's values this step's @@ -281,18 +306,44 @@ func Run(ctx context.Context, options Options) (Summary, error) { extractorChanges = encodeExtractorChanges(options.Verifier.ChangedExtractors()) } else { skippedVerification = true - logger.Warn("transitional tree after retry budget; skipping verifier", + summary.SkippedVerification++ + logger.Warn("unsettled tree; skipping verifier", "step", stepIndex, "screen", screen, "nodes", treeSize) } logger.Info("step", "index", stepIndex, "screen", screen, "nodes", treeSize) - nextAction, nextErr := actionSource.NextAction(ctx, stepIndex) + // A frame the verifier would not look at is not one to act on either. + // #75 is the fuzzer tapping into a screen that is still filling in, and + // holding the action back is also what keeps the spec's view of the run + // continuous: the action a step applies is reported on the NEXT step the + // verifier accepts, so acting here would leave the action applied last + // step unreported for good, and a property counting actions against + // their effects would then see an effect whose cause the runner + // swallowed. See TestRunner_ASkippedStepDoesNotSwallowTheActionBeforeIt. + // + // Unbounded, because lastAction holds exactly one action: any bound that + // let the runner act again while the verifier was still being skipped + // would overwrite the action the hold was carrying, and that is the same + // swallow arriving one step later. A screen that keeps moving therefore + // costs the run its actions rather than its soundness, and a run that + // verified nothing says so in its outcome (internal/testrun). + held := skippedVerification + if held { + logger.Warn("screen still moving; holding this step's action back", + "step", stepIndex) + } + + var nextAction verifier.Action + nextErr := verifier.ErrNoAction var traceAction *trace.Action - if nextErr == nil { - traceAction = traceActionFor(nextAction, tree) - stampActionSource(traceAction, actionSource) - } else if !errors.Is(nextErr, verifier.ErrNoAction) { - return summary, fmt.Errorf("step %d next action: %w", stepIndex, nextErr) + if !held { + nextAction, nextErr = actionSource.NextAction(ctx, stepIndex) + if nextErr == nil { + traceAction = traceActionFor(nextAction, tree) + stampActionSource(traceAction, actionSource) + } else if !errors.Is(nextErr, verifier.ErrNoAction) { + return summary, fmt.Errorf("step %d next action: %w", stepIndex, nextErr) + } } residuals, residualErr := encodeResiduals(options.Verifier.Residuals()) @@ -300,7 +351,7 @@ func Run(ctx context.Context, options Options) (Summary, error) { logger.Warn("residual encode failed", "step", stepIndex, "err", residualErr) } - applySkipped := false + applySkipped := held var actionSkipped actionSkipReason if nextErr == nil && !appIsForeground(ctx, options) { // The app left the foreground between observe and apply (a prior @@ -365,7 +416,13 @@ func Run(ctx context.Context, options Options) (Summary, error) { "step", stepIndex, "reason", actionSkipped, "err", err) transitional = true applySkipped = true - lastAction = nil + // The error says the call failed, not that the gesture never + // reached the app: a deadline that fires after dispatch leaves + // the effect committed. Reporting no action here would let a + // property convict the app for an effect with no cause, so the + // action is reported with its fate unknown instead. + unconfirmed := nextAction + lastAction = &unconfirmed } else if notDispatched != "" { // The action was chosen but nothing reached the driver, so the // screen is exactly the one already verified: the step stays @@ -379,12 +436,16 @@ func Run(ctx context.Context, options Options) (Summary, error) { lastAction = nil } else { consecutiveApplyFailures = 0 - actionCopy := nextAction - lastAction = &actionCopy + applied := nextAction + applied.Applied = true + lastAction = &applied } - } else { + } else if !held { lastAction = nil } + // A held step leaves lastAction alone on purpose: nothing ran here, and + // the action it points at is still the one the next verified step has to + // be told about. step := trace.Step{ Index: stepIndex, @@ -429,7 +490,16 @@ func Run(ctx context.Context, options Options) (Summary, error) { // concurrent fetches observe a stable post-action state. A transient // apply error means nothing landed, so the idle poll has nothing to // settle and may itself hang on the same device condition. - if nextErr == nil && !applySkipped && nextAction.Kind != verifier.ActionKindWait { + // + // A held step settles too, and it is the only case here that waits with + // nothing applied. The reread that held it takes its two reads a round + // trip apart, which is a tighter window than the one the detector was + // measured over (an action and a settle); looping straight back into it + // would compare two reads of a composing screen closer together still, + // so the screen that most needs to settle is the one given least room. + mutated := nextErr == nil && !applySkipped && + nextAction.Kind != verifier.ActionKindWait + if held || mutated { idleCtx, idleCancel := context.WithTimeout(ctx, options.IdleTimeout) idleErr := options.Driver.WaitForIdle(idleCtx, options.IdleTimeout) if idleErr != nil && idleCtx.Err() == nil { @@ -493,6 +563,10 @@ func RenderSummary(w io.Writer, summary Summary, platform string) { fmt.Fprintf(w, "%d step(s) observed nothing: the device state could not be read\n", summary.FailedObservations) } + if summary.SkippedVerification > 0 { + fmt.Fprintf(w, "%d step(s) judged by nothing: the screen was still moving when it was read\n", + summary.SkippedVerification) + } if len(summary.UnsupportedVerbs) > 0 { fmt.Fprintf(w, "unsupported on %s: %s\n", platform, strings.Join(summary.UnsupportedVerbs, ", ")) @@ -542,20 +616,38 @@ func resolveIdleTimeout(options Options) time.Duration { return timeout } +// foregroundGuard is what ensureForeground had to do to put the app back in +// front. The two interventions are separate values because they are separate +// facts about the action they follow: a relaunch leaves it confirmed but +// straddling a restart, while a system window holding the focus leaves it +// dispatched with no way to tell whether the app received it. +type foregroundGuard int + +const ( + foregroundIntact foregroundGuard = iota + foregroundOverlayDismissed + foregroundRelaunched +) + // ensureForeground keeps the app under test in the foreground. When the driver // can report the foreground app and it no longer matches the bundle under test, -// the app is relaunched. Returns true when a relaunch happened so the caller -// can drop the now-stale lastAction. Drivers without ForegroundChecker (web, +// the app is relaunched. Reports what it did so the caller can pass that on to +// the spec through the previous action. Drivers without ForegroundChecker (web, // iOS) are a no-op. -func ensureForeground(ctx context.Context, options Options, logger *slog.Logger, stepIndex int) bool { +func ensureForeground( + ctx context.Context, + options Options, + logger *slog.Logger, + stepIndex int, +) foregroundGuard { checker, ok := options.Driver.(driver.ForegroundChecker) if !ok || options.BundleID == "" { - return false + return foregroundIntact } foreground, err := checker.ForegroundApp(ctx) if err != nil { logger.Warn("foreground check failed", "step", stepIndex, "err", err) - return false + return foregroundIntact } if foreground != "" && foreground != options.BundleID { logger.Warn("app left foreground; relaunching", @@ -568,7 +660,7 @@ func ensureForeground(ctx context.Context, options Options, logger *slog.Logger, // window, so it never acts outside the app no matter how slow the // relaunch settles. awaitForeground(ctx, options, logger, stepIndex) - return true + return foregroundRelaunched } // The app is the resumed activity, but a system overlay can still own the // focused window while the app stays resumed: a fuzzer swipe starting in the @@ -578,15 +670,15 @@ func ensureForeground(ctx context.Context, options Options, logger *slog.Logger, // the app again. focusChecker, hasFocus := options.Driver.(driver.FocusedWindowChecker) if !hasFocus { - return false + return foregroundIntact } focused, err := focusChecker.FocusedWindowApp(ctx) if err != nil { logger.Warn("focus check failed", "step", stepIndex, "err", err) - return false + return foregroundIntact } if focused == "" || focused == options.BundleID { - return false + return foregroundIntact } logger.Warn("system window obscuring app; dismissing", "step", stepIndex, "focused", focused, "want", options.BundleID) @@ -594,7 +686,7 @@ func ensureForeground(ctx context.Context, options Options, logger *slog.Logger, logger.Warn("dismiss overlay failed", "step", stepIndex, "err", err) } settleForForeground(ctx, options) - return true + return foregroundOverlayDismissed } // appIsForeground reports whether the app under test currently owns the @@ -840,11 +932,23 @@ func applyAction(ctx context.Context, drv driver.DeviceDriver, action verifier.A } // collectLogs pulls recent error-level log entries from the driver since the -// previous fetch. A failure is warned-on but not fatal: log capture is a -// best-effort observability channel, not a correctness dependency. -func collectLogs(ctx context.Context, drv driver.DeviceDriver, since time.Time) []verifier.LogEntry { +// previous fetch. A failure is warned-on but not fatal: one unreadable fetch on +// a flaky device should not end a run. It is not free either. This fetch is the +// whole evidence base for state.logs, so a step that could not make it leaves +// every log property (the default noLogcatErrors included) holding on an empty +// slice, and that has to be visible in the run's output rather than read as the +// app having logged nothing. +func collectLogs( + ctx context.Context, + drv driver.DeviceDriver, + logger *slog.Logger, + step int, + since time.Time, +) []verifier.LogEntry { entries, err := drv.RecentLogs(ctx, since, "E") if err != nil { + logger.Warn("log fetch failed; log properties hold vacuously this step", + "step", step, "err", err) return nil } result := make([]verifier.LogEntry, 0, len(entries)) @@ -1200,10 +1304,17 @@ const ( // orthogonal case where the frame itself is transitional. // // The transitional return reports whether the retry budget was exhausted -// on a still-transitional tree. Callers use it to skip the verifier for -// that step so the previous/current extractor advance does not absorb +// on a still-transitional tree, or (when reread is set) whether a second +// hierarchy read disagreed with the first. Callers use it to skip the verifier +// for that step so the previous/current extractor advance does not absorb // transient state. -func fetchSyncedState(ctx context.Context, options Options, logger *slog.Logger, stepIndex int) (tree *hierarchy.Tree, png []byte, transitional bool, err error) { +func fetchSyncedState( + ctx context.Context, + options Options, + logger *slog.Logger, + stepIndex int, + reread bool, +) (tree *hierarchy.Tree, png []byte, transitional bool, err error) { var pngBytes []byte var previousJSON string retryLoop: @@ -1239,6 +1350,9 @@ retryLoop: case <-timer.C: } } + if reread && err == nil && !transitional && changedOnReread(ctx, options, logger, stepIndex, tree) { + transitional = true + } if len(pngBytes) > 0 { if writeErr := options.TraceWriter.WriteScreenshot(stepIndex, pngBytes); writeErr != nil { logger.Warn("screenshot write failed", "step", stepIndex, "err", writeErr) @@ -1247,6 +1361,98 @@ retryLoop: return tree, pngBytes, transitional, err } +// changedOnReread reads the hierarchy once more and reports whether the screen +// changed shape while we were looking at it. A Compose route can settle before +// its content composes (a lazy list mounts over several frames, a query lands a +// frame late), and a tree read in that window describes a screen that is still +// filling in. Two reads a read apart are the cheapest thing that can see it +// happening: the round trip IS the interval, so there is no sleep here. +// +// The comparison only means anything because the Hierarchy RPC serves the tree +// the snapshot's own read produces (see snapshotTree in the sidecar). Off the +// bare device read it does not: with an IME standing open, the snapshot answers +// with 134 nodes and the bare read with 489, and the pair then differs over +// whether the sidecar closed a keyboard between them rather than over anything +// the app did. +// +// Waiting for the change to stop was measured on an API 34 device and refused: +// a 750ms-quiet poll capped at 2s cost a median 1434ms against 76ms for one +// read, hit its cap on every frame it fired for, and still handed back a frame +// that might be filling. Detecting is what the runner can act on, because a +// step it declines to verify is at worst a missed conviction, never a false +// one. +// +// A read that fails reports no change. Nothing about a dropped RPC says the +// screen was moving, and skipping verification on it would quietly spend the +// run's evidence on a flaky link. +func changedOnReread( + ctx context.Context, + options Options, + logger *slog.Logger, + stepIndex int, + first *hierarchy.Tree, +) bool { + // An empty tree is skipped by the caller anyway, so the read buys nothing. + if first == nil || len(first.Elements) == 0 { + return false + } + hierarchyJSON, err := options.Driver.Hierarchy(ctx) + if err != nil { + logger.Warn("second hierarchy read failed", "step", stepIndex, "err", err) + return false + } + second, err := hierarchy.Parse(hierarchyJSON) + if err != nil || second == nil { + logger.Warn("second hierarchy parse failed", "step", stepIndex, "err", err) + return false + } + if structuralShape(first) == structuralShape(second) { + return false + } + logger.Warn("screen changed between two reads; skipping verifier", + "step", stepIndex, "nodes", len(first.Elements), "then", len(second.Elements)) + return true +} + +// structuralShape renders what is on screen as its nodes' identities in tree +// order: how many there are, and which ids and classes they carry. +// +// Text and bounds are deliberately absent. A measure pass that moves pixels is +// not a screen still composing, and neither is a value arriving into a node +// that already exists, which this cannot tell apart from a clock ticking. This +// decides whether a property gets to judge at all, so it reads only what a +// change in what is on screen can move: a detector that fires on every step of +// a screen with a timer on it would leave the run green and vacuous, which is +// worse than the composition it set out to catch. The trade is measured rather +// than assumed: over 100 folio steps on an API 35 emulator, text moved under +// an unchanged shape on 1 step, and the shape itself moved on 1 other. +// +// TestRunner_OnlyAChangeOfShapeCostsAStepItsVerdict is what holds the line: +// adding either field back to the shape turns one of its cases red. +func structuralShape(tree *hierarchy.Tree) string { + var shape strings.Builder + for _, element := range tree.Elements { + shape.WriteString(element.ResourceID) + shape.WriteByte(0x1f) + shape.WriteString(element.Class) + shape.WriteByte(0x1e) + } + return shape.String() +} + +// driverIsAndroid asks the driver what it is, once per run, so the step loop +// never repeats the RPC. It gates the reread: #75 is about Compose composition, +// and web and iOS have their own settle paths and no measurement saying an +// extra hierarchy read there is cheap. An unreadable answer is not android. +func driverIsAndroid(ctx context.Context, options Options, logger *slog.Logger) bool { + health, err := options.Driver.Health(ctx) + if err != nil { + logger.Warn("health read failed; not rereading the hierarchy", "err", err) + return false + } + return health.Platform == "android" +} + func traceActionFor(action verifier.Action, tree *hierarchy.Tree) *trace.Action { traceAction := &trace.Action{Kind: string(action.Kind), X: action.X, Y: action.Y} switch action.Kind { diff --git a/internal/runner/runner_test.go b/internal/runner/runner_test.go index 61c0e90..9c75a68 100644 --- a/internal/runner/runner_test.go +++ b/internal/runner/runner_test.go @@ -253,6 +253,23 @@ func TestRenderSummary_OmitsUnsupportedLineWhenNone(t *testing.T) { } } +// A step nothing judged is not a step that passed. The run prints its count so +// a green summary cannot hide a run that skipped most of its steps, which is +// what a screen that keeps moving under the reads would produce. +func TestRenderSummary_CountsTheStepsNothingJudged(t *testing.T) { + var out bytes.Buffer + RenderSummary(&out, Summary{Steps: 10, SkippedVerification: 4}, "android") + if !strings.Contains(out.String(), "4 step(s) judged by nothing") { + t.Errorf("expected the unjudged-step count, got:\n%s", out.String()) + } + + out.Reset() + RenderSummary(&out, Summary{Steps: 10}, "android") + if strings.Contains(out.String(), "judged by nothing") { + t.Errorf("a run that judged every step must not print the line, got:\n%s", out.String()) + } +} + func TestRunner_ViolationSurfacesInSummary(t *testing.T) { state := newHarnessWithSpec(t, violationSpec) @@ -1507,8 +1524,15 @@ func TestRunner_UsesAtomicSnapshot(t *testing.T) { if snapshotCalls == 0 { t.Errorf("expected at least one Snapshot call, got %d", snapshotCalls) } - if hierarchyCalls != 0 { - t.Errorf("expected zero standalone Hierarchy calls (runner must use Snapshot), got %d", hierarchyCalls) + // The recorded pair still comes from Snapshot. The standalone hierarchy + // reads are the composition detector (changedOnReread), one per step at + // most, and they are never the source of what the step records. + if hierarchyCalls > summary.Steps { + t.Errorf("expected at most one standalone Hierarchy call per step (runner must observe through Snapshot), got %d over %d steps", + hierarchyCalls, summary.Steps) + } + if snapshotCalls < summary.Steps { + t.Errorf("expected a Snapshot per step, got %d over %d steps", snapshotCalls, summary.Steps) } if screenshotCalls != 0 { t.Errorf("expected zero standalone Screenshot calls (runner must use Snapshot), got %d", screenshotCalls) @@ -2339,8 +2363,10 @@ func TestEnsureForeground_DismissesSystemOverlay(t *testing.T) { logger := slog.New(slog.NewTextHandler(io.Discard, &slog.HandlerOptions{Level: slog.LevelWarn})) options := Options{BundleID: "app.folio", Driver: m, IdleTimeout: 10 * time.Millisecond} - if !ensureForeground(context.Background(), options, logger, 5) { - t.Fatal("expected the guard to act on the focus-stealing overlay") + got := ensureForeground(context.Background(), options, logger, 5) + if got != foregroundOverlayDismissed { + t.Fatalf("the guard reported %v, want foregroundOverlayDismissed; "+ + "an obscured app is not a relaunched one", got) } backs, relaunches := 0, 0 for _, a := range m.Actions() { diff --git a/internal/runner/source.go b/internal/runner/source.go index c57acf3..df8b9f3 100644 --- a/internal/runner/source.go +++ b/internal/runner/source.go @@ -26,14 +26,15 @@ type ActionSource interface { // PushSnapshot. The mobile path has none (returns nil); the web path returns the // values its extractors computed in V8 against the real DOM. // -// lastAction is the action the previous step actually applied, the same value -// PushSnapshot hands the goja state. The web path has to install it in the page -// before its extractors run: a spec extractor reading state.lastAction runs in -// V8 there, and V8 has no way to know what the runner dispatched. +// lastAction and logs are what PushSnapshot hands the goja state. The web path +// has to install both in the page before its extractors run: a spec extractor +// reading state.lastAction or state.logs runs in V8 there, and V8 knows neither +// what the runner dispatched nor what the driver's log fetch returned. type ExtractorSource interface { ExtractorOverrides( ctx context.Context, lastAction *verifier.Action, + logs []verifier.LogEntry, ) (map[int]json.RawMessage, error) } @@ -44,6 +45,13 @@ type lastActionInstaller interface { SetLastAction(ctx context.Context, encoded json.RawMessage) error } +// logInstaller is the same channel for the entries this step's log fetch +// returned. Console output reaches the driver over CDP, so the page can only +// learn about it from the runner. +type logInstaller interface { + SetLogs(ctx context.Context, encoded json.RawMessage) error +} + // gojaSource drives both action selection and (trivially) extractor overrides // for the mobile path, where the goja-bundled picker runs in-process and no V8 // extractor values exist. @@ -58,6 +66,7 @@ func (s gojaSource) NextAction(context.Context, int) (verifier.Action, error) { func (gojaSource) ExtractorOverrides( context.Context, *verifier.Action, + []verifier.LogEntry, ) (map[int]json.RawMessage, error) { return nil, nil } @@ -79,24 +88,35 @@ func (s webSource) NextAction(ctx context.Context, _ int) (verifier.Action, erro return verifier.DecodeAction(raw) } -// ExtractorOverrides installs the previous step's action in the page, then -// reads back what the spec's extractors computed against the live DOM. The -// install is not best-effort: a web driver that cannot take it leaves -// state.lastAction null in V8, which silently turns every action-gated -// property vacuously true, so it is reported as an error instead. +// ExtractorOverrides installs the previous step's action and this step's log +// entries in the page, then reads back what the spec's extractors computed +// against the live DOM. Neither install is best-effort: a web driver that +// cannot take them leaves state.lastAction null and state.logs empty in V8, +// which silently turns every action-gated property and every log property +// vacuously true, so both are reported as errors instead. func (s webSource) ExtractorOverrides( ctx context.Context, lastAction *verifier.Action, + logs []verifier.LogEntry, ) (map[int]json.RawMessage, error) { - installer, ok := s.web.(lastActionInstaller) + actions, ok := s.web.(lastActionInstaller) if !ok { return nil, fmt.Errorf( "web driver %T cannot install state.lastAction; every property gated "+ "on the last action would be vacuously true", s.web) } - if err := installer.SetLastAction(ctx, verifier.EncodeLastAction(lastAction)); err != nil { + if err := actions.SetLastAction(ctx, verifier.EncodeLastAction(lastAction)); err != nil { return nil, fmt.Errorf("install last action: %w", err) } + entries, ok := s.web.(logInstaller) + if !ok { + return nil, fmt.Errorf( + "web driver %T cannot install state.logs; every property reading the "+ + "log stream would be vacuously true", s.web) + } + if err := entries.SetLogs(ctx, verifier.EncodeLogs(logs)); err != nil { + return nil, fmt.Errorf("install logs: %w", err) + } return s.web.EvaluateExtractors(ctx) } diff --git a/internal/runner/uncertain_last_action_test.go b/internal/runner/uncertain_last_action_test.go new file mode 100644 index 0000000..cd122f5 --- /dev/null +++ b/internal/runner/uncertain_last_action_test.go @@ -0,0 +1,154 @@ +package runner + +import ( + "context" + "errors" + "fmt" + "path/filepath" + "sync/atomic" + "testing" + "time" + + "github.com/priyanshujain/sanderling/internal/driver" + mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock" +) + +// An apply error is not proof that nothing landed. An RPC deadline that fires +// after the tap was dispatched leaves the transaction committed, and a runner +// that reports "no action" for it hands +// submitCommitsOneTransactionPerAction a rise of one transaction against a +// window of zero submits: a conviction manufactured out of the runner's own +// uncertainty, on the property carrying most of the detection on android. +// +// The spec below is the real folio predicate pair, imported from the example, +// so what this asserts is the verdict the shipped property reaches. +const submitCountingSpecTemplate = ` +import { actions, always, extract, next, Tap } from "@sanderling/spec"; +import { + committedTransactionsExceedSubmits, + countSubmitsInWindow, +} from "%s"; + +let submits = 0; +const submitsInWindow = extract("submitsInWindow", state => { + const window = countSubmitsInWindow({ + previousCount: submits, + lastAction: state.lastAction, + fresh: true, + }); + submits = window.next; + return window.reported; +}); + +const counts = extract("counts", state => { + const text = state.ax.find("id:TxnCount")?.text; + return text ? { Travel: parseInt(text, 10) } : null; +}); + +globalThis.properties = { + submitCommitsOneTransactionPerAction: always( + next(() => + !committedTransactionsExceedSubmits({ + countsBefore: counts.previous ?? null, + countsAfter: counts.current, + submitsInWindow: submitsInWindow.current, + }), + ), + ), +}; +globalThis.actions = actions(() => [Tap({ on: "id:TxnSubmit" })]); +` + +const homeWithTxnCount = `{"attributes":{"resource-id":"HomeScreen"},"children":[ + {"attributes":{"resource-id":"TxnCount","text":"%d"},"children":[]}, + {"attributes":{"resource-id":"TxnSubmit","bounds":"[40,80,240,160]"},"children":[],"clickable":true,"enabled":true} +]}` + +// dispatchThenFailDriver is the device condition the runner cannot see through: +// the tap reaches the app and commits, then the call the runner is waiting on +// times out. Every later hierarchy read shows the committed transactions. +type dispatchThenFailDriver struct { + *mockdriver.Driver + commitsPerTap int64 + committed atomic.Int64 +} + +func (d *dispatchThenFailDriver) Tap(context.Context, int, int) error { + return d.dispatchThenFail() +} + +func (d *dispatchThenFailDriver) TapSelector(context.Context, string) error { + return d.dispatchThenFail() +} + +func (d *dispatchThenFailDriver) dispatchThenFail() error { + d.committed.Add(d.commitsPerTap) + return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded") +} + +func (d *dispatchThenFailDriver) Snapshot(context.Context) (string, driver.Image, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil +} + +func (d *dispatchThenFailDriver) Hierarchy(context.Context) (string, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), nil +} + +func TestRunner_ApplyErrorAfterDispatchDoesNotConvictTheSubmitCountingProperty(t *testing.T) { + predicates, err := filepath.Abs("../../examples/folio/sanderling/predicates.ts") + if err != nil { + t.Fatal(err) + } + spec := fmt.Sprintf(submitCountingSpecTemplate, predicates) + + run := func(t *testing.T, commitsPerTap int64) []ViolationRecord { + t.Helper() + state := newHarnessWithSpec(t, spec) + device := &dispatchThenFailDriver{Driver: state.mock, commitsPerTap: commitsPerTap} + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 2 { + t.Fatalf("steps = %d, want 2; the run never reached the step that judges the pair", summary.Steps) + } + if got := device.committed.Load(); got != commitsPerTap*2 { + t.Fatalf("the device committed %d transaction(s), want %d; the taps never reached it", + got, commitsPerTap*2) + } + return summary.Violations + } + + t.Run("one transaction per tap is not a double submit", func(t *testing.T) { + if violations := run(t, 1); len(violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v\n"+ + "one transaction rose against a submit the runner dispatched but "+ + "could not confirm, and the spec was told no action happened", + violations) + } + }) + + // The control. Without it a green above proves nothing: a property that + // never sees a comparable pair is silently vacuous and reports the same + // empty violation list. + t.Run("two transactions per tap still convicts", func(t *testing.T) { + violations := run(t, 2) + if len(violations) == 0 { + t.Fatal("the counting property missed a double submit; the harness never " + + "put the property in a position to fire, so the case above proves nothing") + } + if violations[0].Properties[0] != "submitCommitsOneTransactionPerAction" { + t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", violations[0].Properties) + } + }) +} diff --git a/internal/runner/web_carrier_test.go b/internal/runner/web_carrier_test.go index d4776a5..e47a8fb 100644 --- a/internal/runner/web_carrier_test.go +++ b/internal/runner/web_carrier_test.go @@ -55,6 +55,12 @@ func (d *carrierWebDriver) Snapshot(ctx context.Context) (string, driver.Image, func (d *carrierWebDriver) InstallBundle(context.Context, []byte) error { return nil } +// A web target says so. The runner's per-step hierarchy reread is android-only, +// and a fake claiming android would take a path no chrome run takes. +func (d *carrierWebDriver) Health(context.Context) (driver.Health, error) { + return driver.Health{Ready: true, Version: "fake", Platform: "web"}, nil +} + func (d *carrierWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) { d.reads++ return map[int]json.RawMessage{0: json.RawMessage(strconv.Itoa(d.reads))}, nil @@ -69,6 +75,8 @@ func (d *carrierWebDriver) NextActionFromV8(context.Context) (json.RawMessage, e func (d *carrierWebDriver) SetLastAction(context.Context, json.RawMessage) error { return nil } +func (d *carrierWebDriver) SetLogs(context.Context, json.RawMessage) error { return nil } + // TestRunner_TransitionalStepNeverAdvancesThePageCarrier pins the ordering the // web path depends on. The page-side extractors must run only on steps the // verifier accepts: their getters advance spec state every time they evaluate, @@ -166,6 +174,8 @@ func (d *installFailsWebDriver) SetLastAction(context.Context, json.RawMessage) return errors.New("__sanderlingSetLastAction__ is not a function") } +func (d *installFailsWebDriver) SetLogs(context.Context, json.RawMessage) error { return nil } + // TestRunner_LastActionInstallFailureFailsTheRun covers the other half of the // same trust boundary. A run that cannot install lastAction in the page cannot // apply the page's extractor values either, so the step keeps goja's @@ -194,3 +204,49 @@ func TestRunner_LastActionInstallFailureFailsTheRun(t *testing.T) { t.Errorf("Run error = %v, want it to name the failed lastAction install", err) } } + +// logInstallFailsWebDriver takes lastAction and refuses the logs, the shape a +// page carrying an older published @sanderling/spec runtime has: it knows the +// action setter and not the log one. +type logInstallFailsWebDriver struct { + *installFailsWebDriver +} + +func (d *logInstallFailsWebDriver) SetLastAction(context.Context, json.RawMessage) error { + return nil +} + +func (d *logInstallFailsWebDriver) SetLogs(context.Context, json.RawMessage) error { + return errors.New("__sanderlingSetLogs__ is not a function") +} + +// TestRunner_LogInstallFailureFailsTheRun holds the log channel to the same +// standard as the action one. The driver having the console errors decides +// nothing on web: the page's reading of every extractor replaces the host's, so +// a run that cannot put the entries back into the page evaluates noLogcatErrors +// against an empty array and reports green on a console full of errors. +// Continuing past this is the vacuity the whole install exists to prevent. +func TestRunner_LogInstallFailureFailsTheRun(t *testing.T) { + state := newHarnessWithSpec(t, carrierSpec) + web := &logInstallFailsWebDriver{ + installFailsWebDriver: &installFailsWebDriver{Driver: state.mock}, + } + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + _, err := Run(ctx, Options{ + Duration: 2 * time.Second, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 3, + Driver: web, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err == nil { + t.Fatal("Run succeeded with a page that cannot take the step's logs; " + + "every property reading the log stream ran against an empty array") + } + if !bytes.Contains([]byte(err.Error()), []byte("install logs")) { + t.Errorf("Run error = %v, want it to name the failed log install", err) + } +} diff --git a/internal/runner/web_extractor_trace_test.go b/internal/runner/web_extractor_trace_test.go index b31733e..d2c54e4 100644 --- a/internal/runner/web_extractor_trace_test.go +++ b/internal/runner/web_extractor_trace_test.go @@ -47,6 +47,8 @@ func (d *webMockDriver) NextActionFromV8(context.Context) (json.RawMessage, erro func (d *webMockDriver) SetLastAction(context.Context, json.RawMessage) error { return nil } +func (d *webMockDriver) SetLogs(context.Context, json.RawMessage) error { return nil } + // TestRunner_TraceRecordsTheValueTheVerdictUsed fails if the trace and the // verdict disagree about an extractor. A witness is only an explanation of a // violation if it holds the state the violated property was evaluated against. diff --git a/internal/runner/web_last_action_test.go b/internal/runner/web_last_action_test.go index 22ea458..6f66304 100644 --- a/internal/runner/web_last_action_test.go +++ b/internal/runner/web_last_action_test.go @@ -1,11 +1,16 @@ package runner import ( + "bytes" "context" "encoding/json" + "errors" + "log/slog" + "strings" "testing" "time" + "github.com/priyanshujain/sanderling/internal/driver" mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock" ) @@ -25,7 +30,8 @@ globalThis.properties = {}; // control, so the runner has a real applied action to report on the next step. type tappingWebDriver struct { *mockdriver.Driver - installed []string + installed []string + installedLogs []string } func (d *tappingWebDriver) InstallBundle(context.Context, []byte) error { return nil } @@ -43,6 +49,11 @@ func (d *tappingWebDriver) SetLastAction(_ context.Context, encoded json.RawMess return nil } +func (d *tappingWebDriver) SetLogs(_ context.Context, encoded json.RawMessage) error { + d.installedLogs = append(d.installedLogs, string(encoded)) + return nil +} + func TestRunner_WebInstallsLastActionInThePage(t *testing.T) { state := newHarnessWithSpec(t, lastActionSpec) web := &tappingWebDriver{Driver: state.mock} @@ -71,7 +82,114 @@ func TestRunner_WebInstallsLastActionInThePage(t *testing.T) { // Every later step carries what the runner actually applied. The shape is // the goja host's (internal/verifier/marshal.go lastActionFields), pinned // against it by TestLastAction_WebJSONMatchesTheGojaObject. - const want = `{"kind":"Tap","on":"id:TxnSubmit"}` + const want = `{"kind":"Tap","applied":true,"relaunched":null,"on":"id:TxnSubmit"}` + if web.installed[1] != want { + t.Errorf("step 2 installed %s, want %s", web.installed[1], want) + } +} + +// The same hole on the other channel: state.logs was hardcoded [] in +// pkg/spec/src/web-runtime.ts, and because the page's reading of an extractor +// replaces the host's on web, the driver's error-level entries never reached a +// property. The default noLogcatErrors counted an empty array on every run. +func TestRunner_WebInstallsTheStepsLogsInThePage(t *testing.T) { + state := newHarnessWithSpec(t, lastActionSpec) + state.mock.LogEntries = []driver.LogEntry{ + {UnixMillis: 1700000000123, Level: "E", Tag: "console", Message: "boom from the page"}, + } + web := &tappingWebDriver{Driver: state.mock} + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if _, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: web, + Verifier: state.verifier, + TraceWriter: state.writer, + }); err != nil { + t.Fatalf("Run: %v", err) + } + + if len(web.installedLogs) == 0 { + t.Fatal("the page was never handed the step's logs; every property reading " + + "state.logs evaluated against the empty array the page starts with") + } + // The shape is the goja host's (internal/verifier/marshal.go logFields), + // pinned against it by TestLogs_WebJSONMatchesTheGojaObject. + const want = `[{"unixMillis":1700000000123,"level":"E","tag":"console","message":"boom from the page"}]` + if web.installedLogs[0] != want { + t.Errorf("step 1 installed %s, want %s", web.installedLogs[0], want) + } +} + +// A log fetch that fails decides the verdict of every log property: they all +// evaluate against an empty slice and hold. That is not a fact about the app, +// so the step it happened on has to be visible in the run's output. It used to +// be dropped in silence, under a comment claiming it was warned about. +func TestRunner_ReportsALogFetchItCouldNotMake(t *testing.T) { + state := newHarnessWithSpec(t, lastActionSpec) + state.mock.Failures[mockdriver.ActionRecentLogs] = errors.New("adb: device offline") + + var buffer bytes.Buffer + logger := slog.New(slog.NewTextHandler(&buffer, &slog.HandlerOptions{Level: slog.LevelWarn})) + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if _, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: state.mock, + Verifier: state.verifier, + TraceWriter: state.writer, + Logger: logger, + }); err != nil { + t.Fatalf("Run: %v", err) + } + + if !strings.Contains(buffer.String(), "adb: device offline") { + t.Errorf("the run never reported the failed log fetch, so noLogcatErrors "+ + "held on evidence nobody collected; log was %q", buffer.String()) + } +} + +// failingTapWebDriver dispatches the tap and then fails the call, the shape an +// RPC deadline takes: the page has the click, the runner has an error. +type failingTapWebDriver struct { + *tappingWebDriver +} + +func (d *failingTapWebDriver) Tap(context.Context, int, int) error { + return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded") +} + +// The web leg of the same three states the goja host reports. "applied":null is +// not "no action": a property gated on the last action still sees the tap and +// decides for itself, which it cannot do if the page is handed a bare null. +func TestRunner_WebInstallsAnUnconfirmedActionWithItsFateUnknown(t *testing.T) { + state := newHarnessWithSpec(t, lastActionSpec) + web := &failingTapWebDriver{tappingWebDriver: &tappingWebDriver{Driver: state.mock}} + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if _, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: web, + Verifier: state.verifier, + TraceWriter: state.writer, + }); err != nil { + t.Fatalf("Run: %v", err) + } + + if len(web.installed) < 2 { + t.Fatalf("the page was handed lastAction %d time(s); the web path never installed it", + len(web.installed)) + } + const want = `{"kind":"Tap","applied":null,"relaunched":null,"on":"id:TxnSubmit"}` if web.installed[1] != want { t.Errorf("step 2 installed %s, want %s", web.installed[1], want) } diff --git a/internal/testrun/driver.go b/internal/testrun/driver.go index a46a857..caa1531 100644 --- a/internal/testrun/driver.go +++ b/internal/testrun/driver.go @@ -84,6 +84,17 @@ var newDeviceDriver = func(ctx context.Context, options ioscompanion.DeviceOptio return d, d.Close, nil } +// newSimulatorDriver constructs the iOS simulator driver and its cleanup. A +// seam so routing tests assert the run's options reach ioscompanion.Options +// without spawning a companion. +var newSimulatorDriver = func(ctx context.Context, options ioscompanion.Options) (driver.DeviceDriver, func(), error) { + d, err := ioscompanion.New(ctx, options) + if err != nil { + return nil, nil, err + } + return d, d.Close, nil +} + // buildDriver creates the appropriate DeviceDriver for the platform and returns // a cleanup function. For web, ChromeDriver is used directly. An iOS simulator // is driven by the native simulator companion (no JVM). A physical iOS device @@ -99,16 +110,17 @@ func buildDriver(ctx context.Context, options Options, stdout io.Writer) (driver } if options.Platform == "ios" && options.iosIsSimulator { - d, err := ioscompanion.New(ctx, ioscompanion.Options{ + d, cleanup, err := newSimulatorDriver(ctx, ioscompanion.Options{ UniqueDeviceIdentifier: options.iosUDID, BundleID: options.BundleID, AppPath: options.IosAppPath, + ClearState: options.ClearData, Output: stdout, }) if err != nil { return nil, nil, fmt.Errorf("ios simulator driver: %w", err) } - return d, d.Close, nil + return d, cleanup, nil } if options.Platform == "ios" { @@ -117,6 +129,7 @@ func buildDriver(ctx context.Context, options Options, stdout io.Writer) (driver CoreDeviceID: options.iosCoreDeviceID, BundleID: options.BundleID, AppPath: options.IosAppPath, + ClearState: options.ClearData, Output: stdout, }) if err != nil { @@ -148,10 +161,14 @@ func buildDriver(ctx context.Context, options Options, stdout io.Writer) (driver if options.Device != "" { sidecarArgs = append(sidecarArgs, "--serial", options.Device) } + adbPath, err := android.AdbBinary() + if err != nil { + return nil, nil, preflightFailure("android", err) + } sidecarCommand := exec.CommandContext(ctx, "java", sidecarArgs...) sidecarCommand.Stdout = stdout sidecarCommand.Stderr = stdout - sidecarCommand.Env = android.EnvWithAndroidPlatformTools(os.Environ()) + sidecarCommand.Env = android.EnvWithAndroidPlatformTools(os.Environ(), adbPath) // SIGTERM lets the sidecar's shutdown hook stop the iOS XCTest runner. // SIGKILL skips the hook and orphans an xcodebuild session that later // restarts its runner and hijacks the simulator mid-run. @@ -162,11 +179,13 @@ func buildDriver(ctx context.Context, options Options, stdout io.Writer) (driver if err := sidecarCommand.Start(); err != nil { return nil, nil, fmt.Errorf("spawn sidecar: %w", err) } - fmt.Fprintf(stdout, "sidecar pid=%d listening on 127.0.0.1:%d\n", sidecarCommand.Process.Pid, sidecarPort) + sidecarExited := watchSidecar(sidecarCommand) + address := fmt.Sprintf("127.0.0.1:%d", sidecarPort) + fmt.Fprintf(stdout, "sidecar pid=%d listening on %s (adb: %s)\n", sidecarCommand.Process.Pid, address, adbPath) - driverClient, err := driverSidecar.Dial(fmt.Sprintf("127.0.0.1:%d", sidecarPort)) + driverClient, err := driverSidecar.Dial(address) if err != nil { - stopSidecar(sidecarCommand) + stopSidecar(sidecarCommand, sidecarExited) return nil, nil, fmt.Errorf("dial sidecar: %w", err) } driverClient.SetPlatform(options.Platform) @@ -175,22 +194,82 @@ func buildDriver(ctx context.Context, options Options, stdout io.Writer) (driver // (absorbing the XCUITest startup race) runs inside IosDriverBackend.init // in the sidecar - no additional sleep needed here. healthCtx, healthCancel := context.WithTimeout(ctx, sidecarStartupTimeout) - if err := driverClient.WaitForHealth(healthCtx, 250e6); err != nil { - healthCancel() - stopSidecar(sidecarCommand) - _ = driverClient.Close() - return nil, nil, fmt.Errorf("sidecar health check: %w", err) - } + healthErr := awaitSidecar(healthCtx, address, sidecarStartupTimeout, func(pollCtx context.Context) error { + return driverClient.WaitForHealth(pollCtx, 250e6) + }, sidecarExited) healthCancel() + if healthErr != nil { + stopSidecar(sidecarCommand, sidecarExited) + _ = driverClient.Close() + return nil, nil, healthErr + } fmt.Fprintln(stdout, "sidecar is healthy") cleanup := func() { _ = driverClient.Close() - stopSidecar(sidecarCommand) + stopSidecar(sidecarCommand, sidecarExited) } return driverClient, cleanup, nil } +// watchSidecar reaps the sidecar and publishes its exit status. The channel is +// closed after the send so the shutdown path can still receive once the startup +// path has taken the status. +func watchSidecar(sidecarCommand *exec.Cmd) <-chan error { + exited := make(chan error, 1) + go func() { + exited <- sidecarCommand.Wait() + close(exited) + }() + return exited +} + +// awaitSidecar waits for the sidecar to answer a health check, racing that +// against the process exiting so a sidecar that dies during startup is reported +// as the exit it was rather than as a deadline half a minute later. Neither +// failure knows why the sidecar was unhappy, so both name what to look at +// instead of picking a cause. +func awaitSidecar( + ctx context.Context, + address string, + timeout time.Duration, + health func(context.Context) error, + exited <-chan error, +) error { + healthy := make(chan error, 1) + go func() { healthy <- health(ctx) }() + select { + case exitErr := <-exited: + return sidecarExitedError(address, exitErr) + case err := <-healthy: + if err == nil { + return nil + } + select { + case exitErr := <-exited: + return sidecarExitedError(address, exitErr) + default: + return fmt.Errorf( + "sidecar did not answer a health check on %s within %s and is still running\n%s", + address, timeout, sidecarWhatToCheck, + ) + } + } +} + +const sidecarWhatToCheck = "check the sidecar output above, then `sanderling doctor --platform=android` (java 17+, adb, Android SDK)" + +func sidecarExitedError(address string, exitErr error) error { + status := "exit status 0" + if exitErr != nil { + status = exitErr.Error() + } + return fmt.Errorf( + "sidecar exited before it answered a health check on %s: %s\n%s", + address, status, sidecarWhatToCheck, + ) +} + // sidecarShutdownGrace bounds how long the sidecar gets to run its shutdown // hook (terminate the app, stop the XCTest runner) before being killed. const sidecarShutdownGrace = 15 * time.Second @@ -198,25 +277,20 @@ const sidecarShutdownGrace = 15 * time.Second // stopSidecar terminates the sidecar gracefully so its shutdown hook can stop // the device-side runner processes, escalating to SIGKILL when it does not // exit within the grace window. -func stopSidecar(sidecarCommand *exec.Cmd) { +func stopSidecar(sidecarCommand *exec.Cmd, exited <-chan error) { if sidecarCommand.Process == nil { return } if err := sidecarCommand.Process.Signal(syscall.SIGTERM); err != nil { _ = sidecarCommand.Process.Kill() - _ = sidecarCommand.Wait() + <-exited return } - done := make(chan struct{}) - go func() { - _ = sidecarCommand.Wait() - close(done) - }() select { - case <-done: + case <-exited: case <-time.After(sidecarShutdownGrace): _ = sidecarCommand.Process.Kill() - <-done + <-exited } } diff --git a/internal/testrun/driver_test.go b/internal/testrun/driver_test.go index 5e461b3..3bd266b 100644 --- a/internal/testrun/driver_test.go +++ b/internal/testrun/driver_test.go @@ -4,7 +4,10 @@ import ( "context" "errors" "io" + "os/exec" + "strings" "testing" + "time" "github.com/priyanshujain/sanderling/internal/driver" "github.com/priyanshujain/sanderling/internal/driver/ioscompanion" @@ -27,7 +30,7 @@ func TestBuildDriverRoutesPhysicalIOSToDeviceDriver(t *testing.T) { return stubDeviceDriver{}, func() { closed = true }, nil } - options := Options{Platform: "ios", BundleID: "app.folio", IosAppPath: "/tmp/iosApp.app"} + options := Options{Platform: "ios", BundleID: "app.folio", IosAppPath: "/tmp/iosApp.app", ClearData: true} options.iosIsSimulator = false options.iosUDID = "00008140-HW" options.iosCoreDeviceID = "CORE-1" @@ -45,12 +48,41 @@ func TestBuildDriverRoutesPhysicalIOSToDeviceDriver(t *testing.T) { if got.BundleID != "app.folio" || got.AppPath != "/tmp/iosApp.app" { t.Fatalf("DeviceOptions = %+v, want bundle and app path threaded through", got) } + if !got.ClearState { + t.Fatalf("DeviceOptions = %+v, want clear-data threaded through: the driver clears before its session, so a launch cannot", got) + } cleanup() if !closed { t.Fatal("cleanup must close the device driver") } } +func TestBuildDriverThreadsClearStateToTheSimulatorDriver(t *testing.T) { + stubPreflight(t) + original := newSimulatorDriver + t.Cleanup(func() { newSimulatorDriver = original }) + + var got ioscompanion.Options + newSimulatorDriver = func(_ context.Context, options ioscompanion.Options) (driver.DeviceDriver, func(), error) { + got = options + return stubDeviceDriver{}, func() {}, nil + } + + options := Options{Platform: "ios", BundleID: "app.folio", IosAppPath: "/tmp/iosApp.app", ClearData: true} + options.iosIsSimulator = true + options.iosUDID = "SIM-UDID" + + if _, _, err := buildDriver(context.Background(), options, io.Discard); err != nil { + t.Fatalf("buildDriver: %v", err) + } + if got.UniqueDeviceIdentifier != "SIM-UDID" || got.BundleID != "app.folio" || got.AppPath != "/tmp/iosApp.app" { + t.Fatalf("Options = %+v, want the resolved target, bundle and app path", got) + } + if !got.ClearState { + t.Fatalf("Options = %+v, want clear-data threaded through: the driver clears before its session, so a launch cannot", got) + } +} + func TestBuildDriverSurfacesDeviceConstructionError(t *testing.T) { stubPreflight(t) original := newDeviceDriver @@ -65,6 +97,130 @@ func TestBuildDriverSurfacesDeviceConstructionError(t *testing.T) { } } +// A sidecar that dies during startup leaves the health poll with nothing to +// talk to, and reporting that as a deadline sends the reader after a gRPC +// timeout instead of the exit that already happened. +func TestAwaitSidecarReportsTheExitItSaw(t *testing.T) { + exited := make(chan error, 1) + exited <- errors.New("exit status 1") + close(exited) + + err := awaitSidecar( + context.Background(), + "127.0.0.1:54321", + 30*time.Second, + func(ctx context.Context) error { <-ctx.Done(); return ctx.Err() }, + exited, + ) + if err == nil { + t.Fatal("expected an error when the sidecar exits before it is healthy") + } + for _, want := range []string{ + "sidecar exited before it answered a health check on 127.0.0.1:54321: exit status 1", + "check the sidecar output above", + "sanderling doctor --platform=android", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("error %q missing %q", err, want) + } + } +} + +func TestAwaitSidecarTimeoutSaysOnlyWhatItObserved(t *testing.T) { + ctx, cancel := context.WithTimeout(context.Background(), 50*time.Millisecond) + defer cancel() + + err := awaitSidecar( + ctx, + "127.0.0.1:54321", + 50*time.Millisecond, + func(ctx context.Context) error { <-ctx.Done(); return ctx.Err() }, + make(chan error, 1), + ) + if err == nil { + t.Fatal("expected an error when the sidecar never answers") + } + for _, want := range []string{ + "sidecar did not answer a health check on 127.0.0.1:54321 within 50ms and is still running", + "check the sidecar output above", + "sanderling doctor --platform=android", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("error %q missing %q", err, want) + } + } + if strings.Contains(err.Error(), "context deadline exceeded") { + t.Errorf("error %q must not hand the reader a bare gRPC deadline", err) + } +} + +func TestAwaitSidecarHealthyReturnsNil(t *testing.T) { + err := awaitSidecar( + context.Background(), + "127.0.0.1:54321", + 30*time.Second, + func(context.Context) error { return nil }, + make(chan error, 1), + ) + if err != nil { + t.Fatalf("expected a healthy sidecar to pass, got %v", err) + } +} + +func TestStopSidecarTerminatesARunningSidecar(t *testing.T) { + command := exec.Command("sleep", "60") + if err := command.Start(); err != nil { + t.Fatalf("start: %v", err) + } + exited := watchSidecar(command) + + stopped := make(chan struct{}) + go func() { + stopSidecar(command, exited) + close(stopped) + }() + select { + case <-stopped: + case <-time.After(10 * time.Second): + t.Fatal("stopSidecar never returned for a running sidecar") + } + if got := command.ProcessState.String(); got != "signal: terminated" { + t.Errorf("sidecar ended as %q, want the SIGTERM its shutdown hook needs", got) + } +} + +// The startup path takes the exit status to report it, so the shutdown path +// that follows must not sit waiting for a status nobody will send again. +func TestStopSidecarAfterTheStartupPathTookTheExitStatus(t *testing.T) { + command := exec.Command("sh", "-c", "exit 3") + if err := command.Start(); err != nil { + t.Fatalf("start: %v", err) + } + exited := watchSidecar(command) + + err := awaitSidecar( + context.Background(), + "127.0.0.1:54321", + 30*time.Second, + func(ctx context.Context) error { <-ctx.Done(); return ctx.Err() }, + exited, + ) + if err == nil || !strings.Contains(err.Error(), "exit status 3") { + t.Fatalf("expected the sidecar's real exit status, got %v", err) + } + + stopped := make(chan struct{}) + go func() { + stopSidecar(command, exited) + close(stopped) + }() + select { + case <-stopped: + case <-time.After(10 * time.Second): + t.Fatal("stopSidecar blocked on an exit status the startup path had already taken") + } +} + // stubPreflight bypasses the host-readiness checks so routing tests exercise // driver construction on a Linux CI runner that lacks xcrun/java. func stubPreflight(t *testing.T) { diff --git a/internal/testrun/preflight.go b/internal/testrun/preflight.go index a65c88b..e95caf8 100644 --- a/internal/testrun/preflight.go +++ b/internal/testrun/preflight.go @@ -4,6 +4,8 @@ import ( "context" "fmt" "os/exec" + + "github.com/priyanshujain/sanderling/internal/android" ) // Preflight runs platform-specific host checks before sidecar/driver setup. @@ -18,7 +20,15 @@ func Preflight(ctx context.Context, platform string) error { type preflightFunc func(name string) error +// preflightCheck resolves adb through the same helper every adb call in a run +// uses, so a host whose SDK is only reachable through $ANDROID_HOME or a +// standard install location is not turned away here and then driven fine by +// the rest of the pipeline. func preflightCheck(name string) error { + if name == "adb" { + _, err := android.AdbBinary() + return err + } if _, err := exec.LookPath(name); err != nil { return fmt.Errorf("%s not found on PATH: %w", name, err) } diff --git a/internal/testrun/preflight_test.go b/internal/testrun/preflight_test.go index e460d07..d2f32f1 100644 --- a/internal/testrun/preflight_test.go +++ b/internal/testrun/preflight_test.go @@ -3,6 +3,8 @@ package testrun import ( "context" "errors" + "os" + "path/filepath" "strings" "testing" ) @@ -49,6 +51,33 @@ func TestPreflight_AndroidNeedsAdbAndJava(t *testing.T) { } } +// Every adb call in an android run resolves through $ANDROID_HOME and the +// standard SDK locations, so a preflight that only looks at PATH turns away a +// host the run itself would drive. +func TestPreflight_AndroidAcceptsAdbUnderAndroidHome(t *testing.T) { + sdk := t.TempDir() + writeExecutable(t, filepath.Join(sdk, "platform-tools", "adb")) + pathDirectory := t.TempDir() + writeExecutable(t, filepath.Join(pathDirectory, "java")) + t.Setenv("PATH", pathDirectory) + t.Setenv("ANDROID_HOME", sdk) + t.Setenv("ANDROID_SDK_ROOT", "") + + if err := Preflight(context.Background(), "android"); err != nil { + t.Fatalf("Preflight with adb under $ANDROID_HOME: %v", err) + } +} + +func writeExecutable(t *testing.T, path string) { + t.Helper() + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatalf("mkdir %s: %v", filepath.Dir(path), err) + } + if err := os.WriteFile(path, nil, 0o755); err != nil { + t.Fatalf("write %s: %v", path, err) + } +} + func TestPreflight_iOSNeedsXcrun(t *testing.T) { check := func(name string) error { if name == "xcrun" { diff --git a/internal/testrun/testrun.go b/internal/testrun/testrun.go index e822c43..c03d3d8 100644 --- a/internal/testrun/testrun.go +++ b/internal/testrun/testrun.go @@ -247,7 +247,15 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error { // --exit-on-violation a run that found violations is still a successful run // (the summary reports them), which is the behaviour every existing caller // depends on. +// +// A run none of whose steps reached the verifier fails whatever the flags say, +// because it holds no verdict to report. The threshold is every step and not a +// fraction of them: a screen that composes now and then costs a healthy android +// run a step or two, and a check that fired on those would be red on every run. func runOutcome(options Options, summary runner.Summary) error { + if summary.Steps > 0 && summary.SkippedVerification == summary.Steps { + return VacuousRunError{Steps: summary.Steps} + } if options.ExitOnViolation && len(summary.Violations) > 0 { return ViolationsError{Count: len(summary.Violations)} } @@ -283,6 +291,22 @@ func BundleSpec(specPath string, seed int64) (bundler.Result, error) { }) } +// VacuousRunError reports a run in which no step reached the verifier, so no +// property ever judged anything. It is not a clean run and it is not a found +// bug: it is a run that produced no evidence either way, and the absence of +// violations in it says nothing about the app. It stays untyped to the CLI's +// violation path on purpose, so it exits 1 as a broken run rather than 2. +type VacuousRunError struct { + Steps int +} + +func (e VacuousRunError) Error() string { + return fmt.Sprintf( + "%d step(s) ran and none of them reached the verifier: the screen was "+ + "still moving every time it was read, so no property judged this run", + e.Steps) +} + // bundleInputs holds the pre-driver assembly: alias map, seed, esbuild defines, // and the resolved spec-API/goja-runtime paths the bundler consumes. type bundleInputs struct { @@ -372,16 +396,20 @@ func resolveRuntimeSibling(specAPIPath, userSpecPath, filename string) string { return "" } -// resolveSpecAPIPath returns the path to pkg/spec/src/index.ts inside -// a sanderling source checkout, searched upward from the spec file and the cwd. -// Returns "" when not found, in which case esbuild resolves @sanderling/spec via -// node_modules the way a downstream user's project would. +// resolveSpecAPIPath returns the path to the spec API's index.ts: a sanderling +// source checkout first, searched upward from the spec file and the cwd, then +// an installed node_modules/@sanderling/spec. Aliasing the installed copy is +// what keeps the spec and the runtime entry on one module graph; resolving the +// bare specifier through package.json "exports" would load dist/ alongside the +// runtime's src/ and give sampler-rng.ts two instances. func resolveSpecAPIPath(specPath string) string { - var candidates []string + var checkout, installed []string if absoluteSpec, err := filepath.Abs(specPath); err == nil { directory := filepath.Dir(absoluteSpec) for { - candidates = append(candidates, filepath.Join(directory, "pkg/spec/src/index.ts")) + checkout = append(checkout, filepath.Join(directory, "pkg/spec/src/index.ts")) + installed = append(installed, + filepath.Join(directory, "node_modules/@sanderling/spec/src/index.ts")) parent := filepath.Dir(directory) if parent == directory { break @@ -390,9 +418,9 @@ func resolveSpecAPIPath(specPath string) string { } } if cwd, err := os.Getwd(); err == nil { - candidates = append(candidates, filepath.Join(cwd, "pkg/spec/src/index.ts")) + checkout = append(checkout, filepath.Join(cwd, "pkg/spec/src/index.ts")) } - for _, candidate := range candidates { + for _, candidate := range append(checkout, installed...) { if _, err := os.Stat(candidate); err == nil { return candidate } diff --git a/internal/testrun/testrun_test.go b/internal/testrun/testrun_test.go index 3674cef..f00bce2 100644 --- a/internal/testrun/testrun_test.go +++ b/internal/testrun/testrun_test.go @@ -2,7 +2,9 @@ package testrun import ( "context" + "encoding/json" "errors" + "io/fs" "os" "path/filepath" "strings" @@ -287,6 +289,31 @@ func TestRunOutcome_ReportsViolationsOnlyUnderTheFlag(t *testing.T) { } } +// A step the verifier skipped was judged by nothing, so a run whose every step +// was skipped holds no verdict at all: "no violations" there is the absence of +// an answer rather than a clean one. Reporting it as a successful run is the +// green and vacuous outcome structuralShape's own design notes call worse than +// the composition it catches, and the runner's hold is what makes a fully +// skipped run reachable. +func TestRunOutcome_ARunThatJudgedNothingIsNotASuccess(t *testing.T) { + nothingJudged := runner.Summary{Steps: 6, SkippedVerification: 6} + err := runOutcome(Options{}, nothingJudged) + var vacuous VacuousRunError + if !errors.As(err, &vacuous) { + t.Fatalf("a run that judged none of its 6 steps came back %v, want a VacuousRunError", err) + } + if vacuous.Steps != 6 { + t.Errorf("steps: got %d, want 6", vacuous.Steps) + } + + // A screen that composes now and then costs a run steps, not its verdict. A + // check that fired here would turn every healthy android run red. + mostlyJudged := runner.Summary{Steps: 6, SkippedVerification: 5} + if err := runOutcome(Options{}, mostlyJudged); err != nil { + t.Errorf("a run that judged one of its 6 steps must succeed, got %v", err) + } +} + // wedgedLaunchDriver never returns from Launch, standing in for a driver whose // device-side session is stuck. type wedgedLaunchDriver struct { @@ -327,3 +354,153 @@ func TestLaunchAppBoundsWedgedDriver(t *testing.T) { t.Fatal("launchApp never returned: the pre-run launch is unbounded, so a wedged driver hangs the run forever") } } + +// repoFile walks up from the test's working directory and returns the absolute +// path of rel inside the sanderling checkout. +func repoFile(t *testing.T, rel string) string { + t.Helper() + directory, err := os.Getwd() + if err != nil { + t.Fatal(err) + } + for { + candidate := filepath.Join(directory, rel) + if _, err := os.Stat(candidate); err == nil { + return candidate + } + parent := filepath.Dir(directory) + if parent == directory { + t.Fatalf("%s not found above the test directory", rel) + } + directory = parent + } +} + +// publishedFiles returns the "files" entries of pkg/spec/package.json, the +// exact set npm ships in the @sanderling/spec tarball. +func publishedFiles(t *testing.T) []string { + t.Helper() + raw, err := os.ReadFile(repoFile(t, "pkg/spec/package.json")) + if err != nil { + t.Fatal(err) + } + var manifest struct { + Files []string `json:"files"` + } + if err := json.Unmarshal(raw, &manifest); err != nil { + t.Fatal(err) + } + return manifest.Files +} + +// installPublishedPackage reproduces what `npm install @sanderling/spec` +// unpacks into node_modules: only the paths package.json publishes. +func installPublishedPackage(t *testing.T, dest string) { + t.Helper() + specDir := filepath.Dir(repoFile(t, "pkg/spec/package.json")) + for _, entry := range publishedFiles(t) { + source := filepath.Join(specDir, entry) + if _, err := os.Stat(source); err != nil { + continue + } + copyTree(t, source, filepath.Join(dest, entry)) + } +} + +func copyTree(t *testing.T, source, dest string) { + t.Helper() + err := filepath.WalkDir(source, func(path string, entry fs.DirEntry, err error) error { + if err != nil { + return err + } + relative, err := filepath.Rel(source, path) + if err != nil { + return err + } + target := filepath.Join(dest, relative) + if entry.IsDir() { + return os.MkdirAll(target, 0o755) + } + data, err := os.ReadFile(path) + if err != nil { + return err + } + if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil { + return err + } + return os.WriteFile(target, data, 0o644) + }) + if err != nil { + t.Fatal(err) + } +} + +// TestResolveRuntimeSibling_PublishedPackageShipsTheRuntimes pins npm's "files" +// list against the resolver that consumes it. The tarball shipped dist/ alone +// while the node_modules fallback looks for src/goja-runtime.ts, so every +// `npm install @sanderling/spec` user hit "goja-runtime.ts not found". +func TestResolveRuntimeSibling_PublishedPackageShipsTheRuntimes(t *testing.T) { + root := t.TempDir() + installPublishedPackage(t, filepath.Join(root, "node_modules", "@sanderling", "spec")) + specPath := filepath.Join(root, "spec.ts") + if err := os.WriteFile(specPath, []byte(""), 0o644); err != nil { + t.Fatal(err) + } + + for _, filename := range []string{"goja-runtime.ts", "web-runtime.ts"} { + if resolveRuntimeSibling("", specPath, filename) == "" { + t.Errorf("%s unreachable from a published install; package.json publishes %v", + filename, publishedFiles(t)) + } + } +} + +// TestPrepareBundleInputs_InstalledPackageSharesOneModuleGraph pins the +// downstream case: with no sanderling checkout above the spec, the aliases and +// the runtime entry must name the SAME installed copy. An unset alias let +// esbuild resolve @sanderling/spec to dist/ while the runtime came from src/, +// which loads sampler-rng.ts twice; from(), strings(), integers() and emails() +// then read an rng the picker never set and collapse to a fixed default. +func TestPrepareBundleInputs_InstalledPackageSharesOneModuleGraph(t *testing.T) { + root := t.TempDir() + installed := filepath.Join(root, "node_modules", "@sanderling", "spec") + installPublishedPackage(t, installed) + specPath := filepath.Join(root, "sanderling", "spec.ts") + if err := os.MkdirAll(filepath.Dir(specPath), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(specPath, []byte(""), 0o644); err != nil { + t.Fatal(err) + } + + cwd, err := os.Getwd() + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = os.Chdir(cwd) }) + if err := os.Chdir(root); err != nil { + t.Fatal(err) + } + + prep, err := prepareBundleInputs(Options{Spec: specPath}) + if err != nil { + t.Fatal(err) + } + source := filepath.Join(installed, "src") + want := map[string]string{ + "@sanderling/spec": filepath.Join(source, "index.ts"), + "@sanderling/spec/defaults": filepath.Join(source, "defaults/index.ts"), + "@sanderling/spec/defaults/properties": filepath.Join(source, "defaults/properties.ts"), + } + for key, wantValue := range want { + if prep.aliases[key] != wantValue { + t.Errorf("alias %q = %q, want %q", key, prep.aliases[key], wantValue) + } + } + if got := prep.gojaRuntimePath; got != filepath.Join(source, "goja-runtime.ts") { + t.Errorf("gojaRuntimePath = %q, want it beside the aliased index.ts", got) + } + if got := resolveWebRuntimePath(prep.specAPIPath, specPath); got != filepath.Join(source, "web-runtime.ts") { + t.Errorf("webRuntimePath = %q, want it beside the aliased index.ts", got) + } +} diff --git a/internal/verifier/ax_integration_test.go b/internal/verifier/ax_integration_test.go index 5a632c8..23522b5 100644 --- a/internal/verifier/ax_integration_test.go +++ b/internal/verifier/ax_integration_test.go @@ -2,6 +2,7 @@ package verifier import ( "os" + "strconv" "strings" "testing" @@ -176,3 +177,77 @@ func TestStateAxObjectSelectorKeepsCrossPlatformKeysSilent(t *testing.T) { t.Fatalf("probe = %q, want miss", got) } } + +// axSelectorFormsTree carries one node per id shape a dump produces: the bare +// tag Compose and the web driver emit, the package-qualified resource id +// Android emits, and the iOS accessibility identifier. +const axSelectorFormsTree = `{ + "attributes": {"resource-id": "root", "bounds": "[0,0,400,800]"}, + "children": [ + {"attributes": {"resource-id": "BareThing", "text": "bare", "bounds": "[0,0,100,50]"}, + "children": []}, + {"attributes": {"resource-id": "com.example.app:id/AndroidThing", "text": "android", + "bounds": "[0,50,100,100]"}, "children": []}, + {"attributes": {"accessibilityIdentifier": "IosThing", "text": "ios", + "bounds": "[0,100,100,150]"}, "children": []} + ] +}` + +// TestStateAxSelectorFormsAgree drives both selector forms a spec can write +// through state.ax.find and holds them to the same element. The two forms +// dispatch to different lookups (findNodeFromJS sends a string to FindNode and +// an object to FindBySelector), and the object one used to skip the id rule +// that knows an Android resource id is package-qualified, so a spec that wrote +// ax.find({id: "AddAccountSubmit"}) got undefined on Android and every property +// reading it passed while checking nothing. +func TestStateAxSelectorFormsAgree(t *testing.T) { + tree, err := hierarchy.Parse(axSelectorFormsTree) + if err != nil { + t.Fatal(err) + } + for _, test := range []struct { + value string + want string + }{ + {"BareThing", "bare"}, + {"AndroidThing", "android"}, + {"com.example.app:id/AndroidThing", "android"}, + {"IosThing", "ios"}, + } { + t.Run(test.value, func(t *testing.T) { + verifier := newVerifier(t) + mustLoad(t, verifier, ` + globalThis.fromObject = __sanderling__.extract( + state => state.ax.find({ id: `+strconv.Quote(test.value)+` })?.text, "fromObject"); + globalThis.fromString = __sanderling__.extract( + state => state.ax.find("id:" + `+strconv.Quote(test.value)+`)?.text, "fromString"); + globalThis.properties = {}; + `) + if err := verifier.PushSnapshot(SnapshotInput{Tree: tree}); err != nil { + t.Fatal(err) + } + fromObject := readCurrent(t, verifier, "fromObject") + fromString := readCurrent(t, verifier, "fromString") + if fromString != test.want { + t.Fatalf(`ax.find("id:%s") read %v, want %q`, test.value, fromString, test.want) + } + if fromObject != fromString { + t.Errorf( + `one selector, two answers: ax.find({id: %q}) read %v and ax.find("id:%s") read %v`, + test.value, fromObject, test.value, fromString, + ) + } + }) + } +} + +// readCurrent returns a named extractor's current value, or nil when the getter +// returned undefined, which is what an unresolved selector produces. +func readCurrent(t *testing.T, verifier *Verifier, name string) any { + t.Helper() + handle := verifier.runtime.GlobalObject().Get(name) + if handle == nil { + t.Fatalf("%s is not defined", name) + } + return handle.ToObject(verifier.runtime).Get("current").Export() +} diff --git a/internal/verifier/extractor_encoding_test.go b/internal/verifier/extractor_encoding_test.go new file mode 100644 index 0000000..40b9a32 --- /dev/null +++ b/internal/verifier/extractor_encoding_test.go @@ -0,0 +1,255 @@ +package verifier + +import ( + "bytes" + "encoding/json" + "testing" +) + +const elementTreeJSON = `{ + "attributes": {"resource-id": "root", "bounds": "[0,0,400,800]"}, + "enabled": true, + "children": [ + {"attributes": {"resource-id": "TxnAmountField", "text": "199", "bounds": "[0,100,400,160]"}, + "editable": true, "enabled": true, "children": []} + ] +}` + +const elementExtractorSpec = ` +const field = __sanderling__.extract(state => state.ax.find({ "resource-id": "TxnAmountField" }), "field"); +globalThis.properties = {}; +` + +// canonicalElement is the trace's record of one ax element, written out in the +// key order encoding/json emits. It is the contract both hosts owe the replay +// UI: an element the reader can read, with no host-function members and nothing +// dropped. Keys the two hosts disagree on (a DOM has no `checked`, a native +// tree has no `dataset`) are each host's own business; the ENCODING is not. +const canonicalElement = `{ + "__sanderlingSelector": "resource-id:TxnAmountField", + "attrs": { + "bounds": "[0,100,400,160]", + "editable": "true", + "enabled": "true", + "resource-id": "TxnAmountField", + "text": "199" + }, + "bounds": {"bottom": 160, "left": 0, "right": 400, "top": 100}, + "checked": false, + "class": "", + "clickable": false, + "desc": "", + "editable": true, + "enabled": true, + "focused": false, + "id": "TxnAmountField", + "selected": false, + "text": "199", + "x": 200, + "y": 130 +}` + +// TestExtractorEncoding_ElementIsIdenticalOnBothHosts holds the two extractor +// paths to one encoding of one element. The goja hosts (ios, android) run the +// getter in-process and encode the value it returned; the web host runs it in +// V8 and injects the page's reading through OverrideExtractorValues. A reader +// opening a trace does not know which host wrote it, so the same element has to +// land as the same bytes either way. +// +// The goja side used to write null here: an ax element carries find/findAll as +// host functions and json.Marshal refuses the whole object over them. +func TestExtractorEncoding_ElementIsIdenticalOnBothHosts(t *testing.T) { + want := compactJSON(t, canonicalElement) + + native := newVerifier(t) + mustLoad(t, native, elementExtractorSpec) + pushTree(t, native, elementTreeJSON) + fromGoja := string(native.extractors[0].curr) + if fromGoja != want { + t.Errorf("goja host encoded the element as\n %s\nwant\n %s", fromGoja, want) + } + + web := newVerifier(t) + mustLoad(t, web, elementExtractorSpec) + if err := web.PushSnapshot(SnapshotInput{}); err != nil { + t.Fatal(err) + } + if _, err := web.OverrideExtractorValues(map[int]json.RawMessage{0: json.RawMessage(want)}); err != nil { + t.Fatal(err) + } + fromWeb := string(web.extractors[0].curr) + if fromWeb != fromGoja { + t.Errorf("the same element reaches the trace as\n %s\non the web host and\n %s\non goja", + fromWeb, fromGoja) + } +} + +// TestExtractorEncoding_MirrorsTheWebSanitizeRule pins the goja host to the +// rule the web host applies before a reading leaves the page (sanitize in +// pkg/spec/src/web-runtime.ts, asserted there by the "sanitize ..." tests in +// pkg/spec/test/web-runtime.test.ts). Two hosts encoding one value two ways is +// the same defect as encoding it not at all: the reader cannot line the traces +// up. +func TestExtractorEncoding_MirrorsTheWebSanitizeRule(t *testing.T) { + for _, test := range []struct { + name string + expression string + want string + }{ + { + name: "function-valued properties are dropped", + expression: `({ keep: 1, fn: () => 7 })`, + want: `{"keep":1}`, + }, + { + name: "a top-level function is not a value", + expression: `(() => 7)`, + want: `null`, + }, + { + name: "a self-referential cycle breaks instead of overflowing", + expression: `(() => { const a = { name: "root" }; a.self = a; return a; })()`, + want: `{"name":"root","self":null}`, + }, + { + name: "arrays and nested plain values are preserved", + expression: `({ items: [1, "two", { ok: true }] })`, + want: `{"items":[1,"two",{"ok":true}]}`, + }, + { + name: "a non-finite number is not a value", + expression: `Number("nope")`, + want: `null`, + }, + } { + t.Run(test.name, func(t *testing.T) { + if got := encodeSpecValue(t, test.expression); got != test.want { + t.Errorf("encoded as %s, want %s", got, test.want) + } + }) + } +} + +// TestExtractorEncoding_BoundsRecursionPastTheDepthLimit mirrors the web host's +// depth cap. state.ax hands out no cyclic element, but a spec returning a value +// it built itself can nest without end, and a walk with no bound takes the run +// down with a stack overflow. +func TestExtractorEncoding_BoundsRecursionPastTheDepthLimit(t *testing.T) { + encoded := encodeSpecValue(t, `(() => { + let deep = { leaf: true }; + for (let i = 0; i < 40; i++) deep = { next: deep }; + return deep; + })()`) + + var node any + if err := json.Unmarshal([]byte(encoded), &node); err != nil { + t.Fatalf("decode %s: %v", encoded, err) + } + for depth := 0; depth < recordableMaxDepth; depth++ { + object, ok := node.(map[string]any) + if !ok { + t.Fatalf("depth %d: recursion stopped early at %v", depth, node) + } + node = object["next"] + } + if node != nil { + t.Errorf("depth %d is %v, want null", recordableMaxDepth, node) + } +} + +// encodeSpecValue returns what the trace records for an extractor whose getter +// returned the given expression. +func encodeSpecValue(t *testing.T, expression string) string { + t.Helper() + verifier := newVerifier(t) + mustLoad(t, verifier, "__sanderling__.extract(state => "+expression+", \"value\");\nglobalThis.properties = {};") + if err := verifier.PushSnapshot(SnapshotInput{}); err != nil { + t.Fatal(err) + } + return string(verifier.extractors[0].curr) +} + +func compactJSON(t *testing.T, source string) string { + t.Helper() + var compact bytes.Buffer + if err := json.Compact(&compact, []byte(source)); err != nil { + t.Fatal(err) + } + return compact.String() +} + +// TestExtractorEncoding_NestedUndefinedIsNotOnTheWire pins the one reading +// shape the two hosts do NOT encode alike, rather than hiding it. +// +// JSON has no undefined, so the page loses the whole key (asserted in +// pkg/spec/test/web-runtime.test.ts) while goja writes null. goja cannot mirror +// the drop: Export reports an undefined member and a null member identically as +// nil, so dropping those keys here would drop the genuine nulls the page keeps. +// Mirroring the other way, by writing null on the page, would break the one +// thing that does agree. Carrying the member across takes a wire format that +// can express undefined, which is a change to every layer that parses a reading +// and to the replay UI that renders one. +// +// So the guarantee is narrower than "the same object": both hosts answer +// undefined when a property READS the member. Key presence (`in`, Object.keys) +// is not part of it, and this test says so out loud, so closing the gap has to +// be a deliberate change to both hosts at once. +func TestExtractorEncoding_NestedUndefinedIsNotOnTheWire(t *testing.T) { + const reading = `({ absent: undefined, empty: null, present: 1 })` + const fromGoja = `{"absent":null,"empty":null,"present":1}` + // What the page sends for the same getter, with the key gone. + const fromWeb = `{"empty":null,"present":1}` + + if got := encodeSpecValue(t, reading); got != fromGoja { + t.Errorf("goja encoded the reading as %s, want %s", got, fromGoja) + } + + native := newVerifier(t) + mustLoad(t, native, "__sanderling__.extract(state => "+reading+", \"value\");\nglobalThis.properties = {};") + if err := native.PushSnapshot(SnapshotInput{}); err != nil { + t.Fatal(err) + } + + web := newVerifier(t) + mustLoad(t, web, "__sanderling__.extract(state => null, \"value\");\nglobalThis.properties = {};") + if err := web.PushSnapshot(SnapshotInput{}); err != nil { + t.Fatal(err) + } + if _, err := web.OverrideExtractorValues(map[int]json.RawMessage{0: json.RawMessage(fromWeb)}); err != nil { + t.Fatal(err) + } + + for _, probe := range []struct { + expression string + native bool + web bool + }{ + {"reading.absent === undefined", true, true}, + {"reading.empty === null", true, true}, + {"reading.present === 1", true, true}, + // The half that does not survive the wire. + {`"absent" in reading`, true, false}, + } { + if got := evaluateAgainstReading(t, native, probe.expression); got != probe.native { + t.Errorf("goja host: %s is %v, want %v", probe.expression, got, probe.native) + } + if got := evaluateAgainstReading(t, web, probe.expression); got != probe.web { + t.Errorf("web host: %s is %v, want %v", probe.expression, got, probe.web) + } + } +} + +// evaluateAgainstReading answers a boolean expression over the value a property +// would read out of the first extractor, which is where the two hosts have to +// agree. +func evaluateAgainstReading(t *testing.T, verifier *Verifier, expression string) bool { + t.Helper() + if err := verifier.runtime.GlobalObject().Set("reading", verifier.extractors[0].currentValue); err != nil { + t.Fatal(err) + } + value, err := verifier.runtime.RunString(expression) + if err != nil { + t.Fatalf("evaluate %s: %v", expression, err) + } + return value.ToBoolean() +} diff --git a/internal/verifier/marshal.go b/internal/verifier/marshal.go index f2ba923..abe863e 100644 --- a/internal/verifier/marshal.go +++ b/internal/verifier/marshal.go @@ -4,8 +4,6 @@ import ( "bytes" "encoding/json" "fmt" - "math" - "reflect" "slices" "strings" "time" @@ -29,7 +27,7 @@ type stateInput struct { } // stateObject builds the JS-side `state` object matching the State type from -// pkg/spec-api. Fields beyond snapshots/ax are included when the caller +// pkg/spec. Fields beyond snapshots/ax are included when the caller // populated them on stateInput. func stateObject(runtime *goja.Runtime, input stateInput) (*goja.Object, error) { state := runtime.NewObject() @@ -400,7 +398,30 @@ func lastActionFields(action *Action) []actionField { point := func(x, y int) []actionField { return []actionField{{key: "x", value: x}, {key: "y", value: y}} } - fields := []actionField{{key: "kind", value: string(action.Kind)}} + // An action whose apply call failed is not an action that did not happen: + // the dispatch may have landed before the error. That is unknown, and + // unknown is null here for the same reason every other absence in the spec + // surface is, so a property decides for itself instead of being handed a + // "nothing happened" the runner cannot vouch for. + var applied any + if action.Applied { + applied = true + } + // A relaunch between two readings is not "no action happened", which is + // what dropping the action reported instead: the app restarted after an + // action that did run. Only the positive report is a fact the runner can + // vouch for, so the other side is null rather than false: a target whose + // foreground the runner cannot read (web, iOS) never relaunches the app and + // still cannot promise it never restarted. + var relaunched any + if action.Relaunched { + relaunched = true + } + fields := []actionField{ + {key: "kind", value: string(action.Kind)}, + {key: "applied", value: applied}, + {key: "relaunched", value: relaunched}, + } if action.On != "" { fields = append(fields, actionField{key: "on", value: action.On}) } @@ -453,7 +474,7 @@ func objectFromFields(runtime *goja.Runtime, fields []actionField) *goja.Object // has no Go-side state object to read: the runner pushes this JSON into the // page before each extractor evaluation. A nil action encodes as JSON null, // the same value the goja host reports on the first step of a run and after a -// step whose action was never applied. +// step whose action was never dispatched. func EncodeLastAction(action *Action) json.RawMessage { if action == nil { return json.RawMessage("null") @@ -501,19 +522,44 @@ func runtimeMillis(stepTime, runStart time.Time) int64 { return stepTime.Sub(runStart).Milliseconds() } +// logFields is the ONE description of a state.logs entry, for the same reason +// lastActionFields is: the goja host turns it into a JS object (logsArray) and +// the web host receives the same fields as JSON (EncodeLogs), so a property +// counting error-level lines cannot read one shape on native and another on web. +func logFields(entry LogEntry) []actionField { + return []actionField{ + {key: "unixMillis", value: entry.UnixMillis}, + {key: "level", value: entry.Level}, + {key: "tag", value: entry.Tag}, + {key: "message", value: entry.Message}, + } +} + func logsArray(runtime *goja.Runtime, logs []LogEntry) *goja.Object { array := runtime.NewArray() for index, entry := range logs { - item := runtime.NewObject() - _ = item.Set("unixMillis", entry.UnixMillis) - _ = item.Set("level", entry.Level) - _ = item.Set("tag", entry.Tag) - _ = item.Set("message", entry.Message) - _ = array.Set(fmt.Sprintf("%d", index), item) + _ = array.Set(fmt.Sprintf("%d", index), objectFromFields(runtime, logFields(entry))) } return array } +// EncodeLogs renders this step's log entries for the web host, which has no +// Go-side state object to read: the runner pushes this JSON into the page +// before each extractor evaluation. No entries encodes as an empty array, the +// same value the goja host reports for a step whose log fetch found nothing. +func EncodeLogs(logs []LogEntry) json.RawMessage { + var buffer bytes.Buffer + buffer.WriteByte('[') + for index, entry := range logs { + if index > 0 { + buffer.WriteByte(',') + } + buffer.Write(encodeFields(logFields(entry))) + } + buffer.WriteByte(']') + return buffer.Bytes() +} + func exceptionsArray(runtime *goja.Runtime, exceptions []Exception) *goja.Object { array := runtime.NewArray() for index, exception := range exceptions { @@ -529,106 +575,6 @@ func exceptionsArray(runtime *goja.Runtime, exceptions []Exception) *goja.Object return array } -// traceValueMaxDepth bounds how far recordableValue walks. It mirrors -// SANITIZE_MAX_DEPTH in pkg/spec/src/web-runtime.ts, whose sanitize does this -// same job for the values the page reports, so both hosts record the same JSON -// for the same extractor. -const traceValueMaxDepth = 32 - -// recordableValue rewrites an exported goja value into one json.Marshal -// accepts. An accessibility element is a plain object carrying two host -// functions (find/findAll); marshalling it fails on those alone, so the whole -// element used to go unrecorded. Dropping them leaves the element's data (id, -// text, desc, class, the flags, bounds, attrs), which is what a trace reader -// wants and is a subset of the hierarchy the same step already records. -// -// ok is false for a value with no JSON form at all: callers drop that key from -// its object, matching the web host, where a function-valued property is -// skipped and a function inside an array stringifies to null. -func recordableValue(value any, depth int, seen map[uintptr]bool) (any, bool) { - if value == nil { - return nil, true - } - switch typed := value.(type) { - case float64: - return finiteOrNull(typed), true - case float32: - return finiteOrNull(float64(typed)), true - } - reflected := reflect.ValueOf(value) - switch reflected.Kind() { - case reflect.Func: - return nil, false - case reflect.Map: - if reflected.Type().Key().Kind() != reflect.String || !holdsAny(reflected.Type().Elem()) { - return value, true - } - if depth >= traceValueMaxDepth || !firstVisit(reflected, seen) { - return nil, true - } - out := make(map[string]any, reflected.Len()) - iterator := reflected.MapRange() - for iterator.Next() { - entry, ok := recordableValue(iterator.Value().Interface(), depth+1, seen) - if !ok { - continue - } - out[iterator.Key().String()] = entry - } - return out, true - case reflect.Slice, reflect.Array: - if !holdsAny(reflected.Type().Elem()) { - return value, true - } - if depth >= traceValueMaxDepth || !firstVisit(reflected, seen) { - return nil, true - } - out := make([]any, reflected.Len()) - for index := range out { - entry, ok := recordableValue(reflected.Index(index).Interface(), depth+1, seen) - if !ok { - entry = nil - } - out[index] = entry - } - return out, true - default: - return value, true - } -} - -// holdsAny reports whether a container's elements can hide a host function or -// a cycle. Concretely typed containers ([]string, []byte, map[string]string) -// can hold neither, and walking them would rewrite shapes json.Marshal already -// handles, such as []byte's base64 form. -func holdsAny(elem reflect.Type) bool { - return elem.Kind() == reflect.Interface -} - -// firstVisit reports whether a container has not been walked yet, so a cyclic -// value terminates. Empty containers are never recorded: they cannot close a -// cycle, and Go may hand every one of them the same address. -func firstVisit(container reflect.Value, seen map[uintptr]bool) bool { - if container.Kind() == reflect.Array || container.Len() == 0 { - return true - } - address := container.Pointer() - if seen[address] { - return false - } - seen[address] = true - return true -} - -// finiteOrNull maps NaN and the infinities to JSON null, which is what -// JSON.stringify does with them on the web host. -func finiteOrNull(value float64) any { - if math.IsNaN(value) || math.IsInf(value, 0) { - return nil - } - return value -} - func jsonToJSValue(runtime *goja.Runtime, raw json.RawMessage) (goja.Value, error) { if len(raw) == 0 { return goja.Undefined(), nil diff --git a/internal/verifier/marshal_test.go b/internal/verifier/marshal_test.go index 8e8654d..ee744e0 100644 --- a/internal/verifier/marshal_test.go +++ b/internal/verifier/marshal_test.go @@ -175,6 +175,8 @@ func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) { }{ {"nil", nil}, {"Tap", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", X: 12, Y: 34}}, + {"TapApplied", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}}, + {"TapRelaunched", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true, Relaunched: true}}, {"TapWithoutSelector", &Action{Kind: ActionKindTap, X: 12, Y: 34}}, {"DoubleTap", &Action{Kind: ActionKindDoubleTap, On: `desc:say "hi" `}}, {"InputText", &Action{Kind: ActionKindInputText, On: "id:field", Text: "50"}}, @@ -200,3 +202,137 @@ func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) { }) } } + +// A spec has to be able to tell three things apart: no action ran, an action +// ran, and an action was dispatched whose fate the runner cannot vouch for. +// The third used to be reported as the first, which is how a property that +// reasons "an effect landed with no action to cause it" convicts an app over +// an RPC deadline. +func TestLastAction_SeparatesNoActionFromAnActionOfUnknownFate(t *testing.T) { + verifier := newVerifier(t) + mustLoad(t, verifier, ` + globalThis.fate = __sanderling__.extract(state => + state.lastAction === null ? "no action" + : state.lastAction.applied === true ? "applied" + : state.lastAction.applied === null ? "unknown" + : "unreadable"); + `) + + for _, testCase := range []struct { + name string + action *Action + want string + }{ + {"nothing ran", nil, "no action"}, + {"dispatch confirmed", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}, "applied"}, + {"dispatch unconfirmed", &Action{Kind: ActionKindTap, On: "id:TxnSubmit"}, "unknown"}, + } { + t.Run(testCase.name, func(t *testing.T) { + if err := verifier.PushSnapshot(SnapshotInput{ + Snapshots: Snapshots{}, + LastAction: testCase.action, + }); err != nil { + t.Fatal(err) + } + handle := verifier.runtime.GlobalObject().Get("fate").ToObject(verifier.runtime) + if got := handle.Get("current").String(); got != testCase.want { + t.Errorf("the spec read %q, want %q", got, testCase.want) + } + }) + } +} + +// The runner relaunches the app when it leaves the foreground, which used to be +// reported to the spec as "no action ran between these two readings". The +// action did run; what a property cannot assume across it is that app state was +// continuous, so the relaunch is its own fact on an action that keeps its +// confirmed dispatch. +func TestLastAction_ReportsARelaunchSeparatelyFromTheDispatch(t *testing.T) { + verifier := newVerifier(t) + mustLoad(t, verifier, ` + globalThis.continuity = __sanderling__.extract(state => + state.lastAction === null ? "no action" + : state.lastAction.applied !== true ? "unconfirmed" + : state.lastAction.relaunched === true ? "applied across a relaunch" + : state.lastAction.relaunched === null ? "applied, no relaunch reported" + : "unreadable"); + `) + + for _, testCase := range []struct { + name string + action *Action + want string + }{ + {"nothing ran", nil, "no action"}, + { + "confirmed, app stayed", + &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}, + "applied, no relaunch reported", + }, + { + "confirmed, app relaunched after it", + &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true, Relaunched: true}, + "applied across a relaunch", + }, + } { + t.Run(testCase.name, func(t *testing.T) { + if err := verifier.PushSnapshot(SnapshotInput{ + Snapshots: Snapshots{}, + LastAction: testCase.action, + }); err != nil { + t.Fatal(err) + } + handle := verifier.runtime.GlobalObject().Get("continuity").ToObject(verifier.runtime) + if got := handle.Get("current").String(); got != testCase.want { + t.Errorf("the spec read %q, want %q", got, testCase.want) + } + }) + } +} + +// TestLogs_WebJSONMatchesTheGojaObject pins state.logs to ONE shape across the +// two hosts, for the same reason lastAction is pinned. On web the page's +// reading of every extractor replaces the host's, so state.logs is whatever +// EncodeLogs put in the page: a field this side renames or cases differently +// leaves the default noLogcatErrors counting nothing on web while it counts on +// native, with nothing reporting that it never saw an entry. +func TestLogs_WebJSONMatchesTheGojaObject(t *testing.T) { + verifier := newVerifier(t) + mustLoad(t, verifier, ` + globalThis.lines = __sanderling__.extract(state => JSON.stringify(state.logs)); + `) + + for _, testCase := range []struct { + name string + logs []LogEntry + }{ + {"none", nil}, + {"empty", []LogEntry{}}, + { + "one error", + []LogEntry{{UnixMillis: 1700000000123, Level: "E", Tag: "console", Message: "boom from the page"}}, + }, + { + "mixed levels", + []LogEntry{ + {UnixMillis: 1, Level: "E", Tag: "console", Message: `say "hi" & co`}, + {UnixMillis: 2, Level: "W", Tag: "AndroidRuntime", Message: "a warning"}, + }, + }, + } { + t.Run(testCase.name, func(t *testing.T) { + if err := verifier.PushSnapshot(SnapshotInput{ + Snapshots: Snapshots{}, + Logs: testCase.logs, + }); err != nil { + t.Fatal(err) + } + handle := verifier.runtime.GlobalObject().Get("lines").ToObject(verifier.runtime) + goja := handle.Get("current").String() + web := string(EncodeLogs(testCase.logs)) + if goja != web { + t.Errorf("the two hosts disagree on state.logs\n goja: %s\n web: %s", goja, web) + } + }) + } +} diff --git a/internal/verifier/types.go b/internal/verifier/types.go index 5b35eb7..09d2c5c 100644 --- a/internal/verifier/types.go +++ b/internal/verifier/types.go @@ -33,6 +33,17 @@ type Action struct { // Direction is the scroll direction for ActionKindScroll: one of "up", // "down", "left", "right". Empty for every other kind. Direction string + // Applied is meaningful only on the action a step reports to the spec as + // state.lastAction: true when the runner saw the dispatch succeed, false + // when the apply call failed and nothing can say whether the action + // reached the app. The spec is told which of the two it is. + Applied bool + // Relaunched, like Applied, is meaningful only on state.lastAction: the + // runner brought the app back to the foreground after this action, so the + // two readings the spec compares straddle a restart. The action still + // happened; what a property cannot assume across it is that app state ran + // continuously from one reading to the next. + Relaunched bool } // LogEntry mirrors a logcat line captured between steps. diff --git a/internal/verifier/worker.go b/internal/verifier/worker.go index f3cc48a..740df73 100644 --- a/internal/verifier/worker.go +++ b/internal/verifier/worker.go @@ -7,6 +7,8 @@ import ( "errors" "fmt" "maps" + "math" + "reflect" "sort" "time" @@ -383,17 +385,70 @@ func encodeExtractorValue(value goja.Value) ([]byte, error) { if value == nil || goja.IsUndefined(value) || goja.IsNull(value) { return []byte("null"), nil } - recordable, ok := recordableValue(value.Export(), 0, map[uintptr]bool{}) - if !ok { - return []byte("null"), nil - } - body, err := json.Marshal(recordable) + body, err := json.Marshal(recordableValue(value.Export(), 0, map[uintptr]bool{})) if err != nil { return nil, err } return body, nil } +// recordableMaxDepth mirrors SANITIZE_MAX_DEPTH in pkg/spec/src/web-runtime.ts. +const recordableMaxDepth = 32 + +// recordableValue applies the web host's sanitize rule (web-runtime.ts) to an +// exported goja value: function members are dropped, a cycle or a branch past +// the depth cap becomes null, and a non-finite number becomes null. One rule on +// both hosts is what lets the replay UI render a trace without the reader +// having to know which host produced it. An ax element carries its find and +// findAll host functions, and json.Marshal rejects the whole element over them, +// so without this an element-valued extractor reached the trace as null. +func recordableValue(value any, depth int, seen map[uintptr]bool) any { + switch typed := value.(type) { + case map[string]any: + address := reflect.ValueOf(typed).Pointer() + if depth >= recordableMaxDepth || seen[address] { + return nil + } + seen[address] = true + members := make(map[string]any, len(typed)) + for key, member := range typed { + if reflect.ValueOf(member).Kind() == reflect.Func { + continue + } + members[key] = recordableValue(member, depth+1, seen) + } + return members + case []any: + if depth >= recordableMaxDepth { + return nil + } + // Every zero-length allocation shares one address, so tracking an empty + // array would identify it as every other empty array. It cannot close a + // cycle either way. + if len(typed) > 0 { + address := reflect.ValueOf(typed).Pointer() + if seen[address] { + return nil + } + seen[address] = true + } + members := make([]any, len(typed)) + for index, member := range typed { + members[index] = recordableValue(member, depth+1, seen) + } + return members + case float64: + if math.IsNaN(typed) || math.IsInf(typed, 0) { + return nil + } + return typed + } + if reflect.ValueOf(value).Kind() == reflect.Func { + return nil + } + return value +} + // ChangedExtractors returns the named extractors whose value changed between // the prior PushSnapshot and the current one. The map is keyed by extractor // name; unnamed extractors (extractor_N fallback) are included so the replay diff --git a/pkg/spec/LICENSE b/pkg/spec/LICENSE new file mode 100644 index 0000000..8755b39 --- /dev/null +++ b/pkg/spec/LICENSE @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Priyanshu Jain + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/pkg/spec/README.md b/pkg/spec/README.md index c651ddf..abfc57b 100644 --- a/pkg/spec/README.md +++ b/pkg/spec/README.md @@ -1,67 +1,13 @@ # @sanderling/spec -TypeScript spec API for [sanderling](https://github.com/priyanshujain/sanderling), a property-based UI fuzzer for mobile and web apps. +TypeScript spec API for [sanderling](https://github.com/priyanshujain/sanderling), a property-based UI fuzzer for Android, iOS and web apps. -Spec authors write properties (what the app must always or eventually do), extractors (structured state from the UI), and action generators (what sanderling is allowed to do). The `sanderling` CLI evaluates the spec in a loop against a running app. - -## Install +A spec exports properties (what the app must always or eventually do), extractors (structured state read off the UI), and action generators (what sanderling is allowed to do). The `sanderling` CLI evaluates the spec against a running app once per step. ```sh npm install --save-dev @sanderling/spec ``` -## Usage +[Getting started](https://priyanshujain.github.io/sanderling/manual/getting-started/) installs the CLI and runs a first spec. The [spec language reference](https://priyanshujain.github.io/sanderling/manual/spec-language/) lists every primitive, and the [case study](https://priyanshujain.github.io/sanderling/manual/case-study/) walks a complete spec end to end. -```ts -import { extract, always, eventually, actions, weighted, taps, swipes, InputText, Tap } from "@sanderling/spec"; - -const loggedIn = extract((s) => !!s.ax.find("id:home-tab-bar")); -const balance = extract((s) => (s.snapshots.balance as number) ?? 0); -const emailField = extract((s) => s.ax.find("id:email-field")); -const submitButton = extract((s) => s.ax.find("id:sign-in-button")); - -export const properties = { - balanceNeverNegative: always(() => balance.current >= 0), - loginSucceeds: eventually(() => loggedIn.current).within(30, "seconds"), -}; - -const doLogin = actions(() => { - if (loggedIn.current) return []; - const email = emailField.current; - const submit = submitButton.current; - if (!email || !submit) return []; - return [InputText({ into: email, text: "test@example.com" }), Tap({ on: submit })]; -}); - -export const actionsRoot = weighted( - [50, doLogin], - [10, taps], - [2, swipes], -); -``` - -## Setup actions - -Some action generators are not fuzz targets but preconditions: they drive the -app from a fresh state into the surface you actually want to fuzz (login, -onboarding, permission grants, seed data). Export them as `setup` instead of -mixing them into `actionsRoot`. The runner tries `setup` first; if it yields -no action, it falls through to `actionsRoot`. State regressing back across the -precondition (e.g. logout under fuzz) automatically re-engages setup. - -```ts -const login = actions(() => { - if (loggedIn.current) return []; - return [InputText({ into: emailField.current!, text: "demo@app.test" }), Tap({ on: submitButton.current! })]; -}); - -export const setup = login; -export const actionsRoot = weighted([60, browse], [40, edit]); - -(globalThis as { setup?: unknown }).setup = setup; -``` - -Setup is just an `ActionGenerator`; compose with `actions`, `weighted`, or -`whenRoute` exactly like the main pool. - -Works identically across Android, iOS, and web targets. +The CLI bundles this package's TypeScript sources at run time, so keep the CLI and the package on the same release. diff --git a/pkg/spec/package.json b/pkg/spec/package.json index ffbfb84..05db2dd 100644 --- a/pkg/spec/package.json +++ b/pkg/spec/package.json @@ -21,6 +21,7 @@ }, "files": [ "dist", + "src", "README.md" ], "repository": { diff --git a/pkg/spec/src/index.ts b/pkg/spec/src/index.ts index 4ac8c31..b80170f 100644 --- a/pkg/spec/src/index.ts +++ b/pkg/spec/src/index.ts @@ -4,6 +4,7 @@ export type { Action, ActionGenerator, AttrSelector, + Direction, DoubleTapAction, EventuallyFormula, ExceptionRecord, @@ -12,11 +13,14 @@ export type { InputTextAction, Key, KnownAttrSelectors, + LastAction, LogEntry, + LongPressAction, Point, PressKeyAction, RawAttrs, Sampler, + ScrollAction, SelectorPath, Snapshots, State, diff --git a/pkg/spec/src/ltl.ts b/pkg/spec/src/ltl.ts index aa5f4e9..2dc8e71 100644 --- a/pkg/spec/src/ltl.ts +++ b/pkg/spec/src/ltl.ts @@ -12,11 +12,12 @@ export function next(predicate: () => boolean): Formula { return globalThis.__sanderling__.next(predicate); } -// An unbounded `eventually` never forces a violation within a finite run. -// Prefer `.within(n, unit)` when you want the verifier to fail a property -// that stalls. `"steps"` counts observed steps rather than wall-clock time, -// which is what keeps the window the same size across runs of different -// speeds. +// An unbounded `eventually` that never fires is violated when the run ends, +// with the reason "eventually never satisfied", so a goal the run does not +// reach is a violation every time. `.within(n, unit)` convicts at the step the +// window closes instead of at run end. `"steps"` counts observed steps rather +// than wall-clock time, which is what keeps the window the same size across +// runs of different speeds. export function eventually(predicate: () => boolean): EventuallyFormula { return globalThis.__sanderling__.eventually(predicate); } diff --git a/pkg/spec/src/types.ts b/pkg/spec/src/types.ts index d09c021..49bc7a7 100644 --- a/pkg/spec/src/types.ts +++ b/pkg/spec/src/types.ts @@ -115,10 +115,29 @@ export interface ExceptionRecord { unixMillis?: number; } +/** + * The previous step's action as the runner reports it. `applied` is true when + * the runner saw the dispatch succeed and null when the apply call failed with + * the gesture possibly already delivered: an RPC deadline can fire after the + * tap landed. Null is unknown, not "it did not happen" (`state.lastAction` is + * itself null for that), so a property attributing an effect to this action + * has to decline unless `applied` is true. + * + * `relaunched` is true when the runner had to bring the app back to the + * foreground after this action, so the previous reading and the current one + * straddle a restart. The action itself still happened; what a property cannot + * assume across it is that app state ran continuously between the two readings, + * and one demanding an effect of this action has to decline. Null is "not + * reported", which is weaker than "the app never restarted": a target whose + * foreground the runner cannot read never relaunches the app and cannot promise + * that either. + */ +export type LastAction = Action & { applied: true | null; relaunched: true | null }; + export interface State { snapshots: Snapshots; ax: AccessibilityTree; - lastAction: Action | null; + lastAction: LastAction | null; time: number; logs: readonly LogEntry[]; exceptions: readonly ExceptionRecord[]; diff --git a/pkg/spec/src/web-runtime.ts b/pkg/spec/src/web-runtime.ts index d9e5935..fe37e49 100644 --- a/pkg/spec/src/web-runtime.ts +++ b/pkg/spec/src/web-runtime.ts @@ -306,13 +306,18 @@ function selectorFromString(selector: string): { css?: string; xpath?: string } // (Compose for Web mounts its canvas and its whole accessibility tree inside a // shadow root on the mount element) keeps its entire UI on the far side of one: // without this a spec sees four nodes and can neither enumerate a target nor -// resolve a testTag. Light-DOM matches come first, then shadow content in walk -// order. XPath has no equivalent, so `text:` selectors stop at the boundary. +// resolve a testTag. Matches come back in the order expandShadowContent walks +// and buildTree (internal/driver/chrome/driver.go) emits: a host, then that +// host's shadow content, then the host's light children. Sweeping the light DOM +// first and descending afterwards put a shadow-hosted match behind a later +// light-DOM one, so find() answered with a different element on each host. +// XPath has no equivalent, so `text:` selectors stop at the boundary. function deepQueryAll(selector: string, root: ParentNode): Element[] { const found: Element[] = []; const visit = (scope: ParentNode): void => { - for (const element of Array.from(scope.querySelectorAll(selector))) found.push(element); + const matched = new Set(Array.from(scope.querySelectorAll(selector))); for (const element of Array.from(scope.querySelectorAll("*"))) { + if (matched.has(element)) found.push(element); if (element.shadowRoot) visit(element.shadowRoot); } }; @@ -615,6 +620,15 @@ if (typeof globalThis.addEventListener === "function") { // that reads state.lastAction vacuously true on web. let lastAction: unknown = null; +// logs is what the driver captured between the previous step and this one, +// pushed in by the Go runner (via __sanderlingSetLogs__) before each extractor +// evaluation, in the shape internal/verifier/marshal.go builds for goja. The +// page cannot derive it: console output reaches the runner over CDP and nothing +// in the page reads it back. Hardcoding [] here, as this file used to, makes +// every spec property that reads state.logs vacuously true on web, the default +// noLogcatErrors included, because the page's reading is the one that wins. +let logs: unknown[] = []; + function buildState(): unknown { return { snapshots: {}, @@ -623,7 +637,7 @@ function buildState(): unknown { window, lastAction, time: 0, - logs: [], + logs, exceptions: capturedExceptions.slice(), }; } @@ -672,6 +686,11 @@ defineLockedGlobal("__sanderlingSetLastAction__", (value: unknown) => { lastAction = value ?? null; }); +// The host calls this once per step too, alongside __sanderlingSetLastAction__. +defineLockedGlobal("__sanderlingSetLogs__", (value: unknown) => { + logs = Array.isArray(value) ? value : []; +}); + // The host reads the same buffer buildState puts behind state.exceptions, so // the goja-side state.exceptions is the page's list rather than the empty one // it held before, and the trace records an error surface an offline oracle can diff --git a/pkg/spec/test/api.test.ts b/pkg/spec/test/api.test.ts index e0239ad..0f72c4c 100644 --- a/pkg/spec/test/api.test.ts +++ b/pkg/spec/test/api.test.ts @@ -29,6 +29,12 @@ import { weighted, whenRoute, } from "../src/index.ts"; +import type { + Action, + Direction, + LongPressAction, + ScrollAction, +} from "../src/index.ts"; import { setSamplerRng } from "../src/actions.ts"; import { SAMPLER_REFUSAL_NAME, setEnumeratingCandidates } from "../src/sampler-rng.ts"; import { Pcg } from "../src/pcg.ts"; @@ -446,3 +452,18 @@ test("whenRoute body is skipped for a null route", () => { const node = whenRoute(route, ["home"], () => [Tap({ on: "id:x" })]); assert.deepEqual((node as { generate: () => unknown }).generate(), []); }); + +// The package entry is the only module a spec author can import from, so every +// member of the exported Action union, and the Direction needed to build a +// Scroll, has to be reachable there rather than only from src/types.ts. +test("index exports every action type a spec author annotates with", () => { + const direction: Direction = "down"; + const scroll: ScrollAction = Scroll({ direction, in: "id:list" }); + const longPress: LongPressAction = LongPress({ on: "id:row" }); + const built: Action[] = [scroll, longPress]; + + assert.deepEqual( + built.map(action => action.kind), + ["Scroll", "LongPress"], + ); +}); diff --git a/pkg/spec/test/folio-account-card-parse.test.ts b/pkg/spec/test/folio-account-card-parse.test.ts index 6b046b3..18230f3 100644 --- a/pkg/spec/test/folio-account-card-parse.test.ts +++ b/pkg/spec/test/folio-account-card-parse.test.ts @@ -65,6 +65,18 @@ test("name ending in digits does not leak into the balance", () => { assert.equal(balanceOf(card("20", "2024", 3, "-$1,234.56")), -123456); }); +// The limit of a key read off merged text, and the reason homeTxnCountsOf +// guards the ambiguity rather than resolving it: two DIFFERENT accounts render +// the same card, character for character. Names are unique (Accounts.name is +// UNIQUE, checked NOCASE) but the count runs straight into a name that ends in +// digits, so nothing computed from this string can say which account it is. +test("two accounts can render one card, so no key off it can be injective", () => { + const travel1 = card("TR", "Travel1", 25, "$120.00"); + const travel12 = card("TR", "Travel12", 5, "$120.00"); + assert.equal(travel1, travel12); + assert.equal(cardAccountName({ childText: undefined, cardText: travel1 }), "TRTravel"); +}); + // The account key only has to be stable and per-account. newAccountBalanceIsZero // reads it as a set member: a key that drifted as an account's transaction // count grew would make an existing account look brand new, and the property diff --git a/pkg/spec/test/folio-home-card-readings.test.ts b/pkg/spec/test/folio-home-card-readings.test.ts index c76babe..876ab67 100644 --- a/pkg/spec/test/folio-home-card-readings.test.ts +++ b/pkg/spec/test/folio-home-card-readings.test.ts @@ -107,6 +107,7 @@ function run(steps: { route: string | null; cards: CardReading[]; lastAction: un const idle = { kind: "Tap", on: "testTag:AccountCard" }; const doubleSubmit = { kind: "DoubleTap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" }; +const submit = { kind: "Tap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" }; test("an un-laid-out Home no longer kills the counting invariant", () => { const trace = run([ @@ -157,3 +158,45 @@ test("an un-laid-out Home does not close the counting window", () => { true, ); }); + +// What keeps a healthy submit from ever arriving as a rise nobody paid for. The +// reading banked here also resets the submit window, so a Home card list drawn +// before the store caught up with a commit would bank stale counts, start the +// next window empty, and leave the rise turning up with no budget to cover it. +// The app cannot put that frame in front of the spec: submit() pops one entry, +// so a commit lands back on the ledger it came from and the first Home reading +// is a whole action later, with the submit still in the window when the rise +// does show up. +test("a submit landing on the ledger is still in the window when Home reads it", () => { + const trace = run([ + { route: "home", cards: [card("Checking", 0, "3")], lastAction: idle }, + { route: "ledger", cards: [], lastAction: submit }, + { route: "home", cards: [card("Checking", 5000, "4")], lastAction: idle }, + ]); + assert.equal(trace[2]?.submits, 1); + assert.equal( + committedTransactionsExceedSubmits({ + countsBefore: trace[1]?.counts ?? null, + countsAfter: trace[2]?.counts ?? null, + submitsInWindow: trace[2]?.submits ?? 0, + }), + false, + ); +}); + +test("and a double submit down that same path still convicts", () => { + const trace = run([ + { route: "home", cards: [card("Checking", 0, "3")], lastAction: idle }, + { route: "ledger", cards: [], lastAction: doubleSubmit }, + { route: "home", cards: [card("Checking", 10000, "5")], lastAction: idle }, + ]); + assert.equal(trace[2]?.submits, 1); + assert.equal( + committedTransactionsExceedSubmits({ + countsBefore: trace[1]?.counts ?? null, + countsAfter: trace[2]?.counts ?? null, + submitsInWindow: trace[2]?.submits ?? 0, + }), + true, + ); +}); diff --git a/pkg/spec/test/folio-ledger-window.test.ts b/pkg/spec/test/folio-ledger-window.test.ts new file mode 100644 index 0000000..a0fa6ae --- /dev/null +++ b/pkg/spec/test/folio-ledger-window.test.ts @@ -0,0 +1,501 @@ +import assert from "node:assert/strict"; +import { test } from "node:test"; + +import { + committedAmountExceedsOneSubmit, + committedTransactionsExceedSubmits, + countSubmitsInWindow, + homeTxnCountsOf, + parseAccountBalance, + parseTypedAmount, + readAccountBalance, + readHomeCards, +} from "../../../examples/folio/sanderling/predicates.ts"; +import type { + ObservedAction, + TxnCount, +} from "../../../examples/folio/sanderling/predicates.ts"; + +// The per-account balance the ledger and the add-transaction screen both show. +// Its window closes on every frame of the transaction flow, where the Home +// total's closes only when the walk goes back to Home: the iOS run in #78 went +// 117 steps between two Home readings and accumulated 37 submits against a rise +// of 15, so the double tap at step 32 sat in a window far too wide to judge. + +const submit: ObservedAction = { + kind: "Tap", + on: "testTag:AddTransactionScreen > testTag:TxnSubmit", + applied: true, +}; +const doubleSubmit: ObservedAction = { ...submit, kind: "DoubleTap" }; +const openLedger: ObservedAction = { + kind: "Tap", + on: "testTag:HomeScreen > testTag:AccountCard", + applied: true, +}; +const openAddTxn: ObservedAction = { + kind: "Tap", + on: "testTag:LedgerScreen > testTag:AddTransactionButton", + applied: true, +}; +const typeAmount: ObservedAction = { + kind: "InputText", + on: "testTag:AddTransactionScreen > testTag:TxnAmountField", + applied: true, +}; +const goBack: ObservedAction = { kind: "Tap", on: "testTag:BackButton", applied: true }; + +test("the ledger writes the balance bare and the add-transaction header labels it", () => { + assert.equal(parseAccountBalance("$196.00"), 19600); + assert.equal(parseAccountBalance("Balance: $196.00"), 19600); + assert.equal(parseAccountBalance("-$1,234.56"), -123456); + assert.equal(parseAccountBalance("Balance: -$1,234.56"), -123456); + assert.equal(parseAccountBalance("$0.00"), 0); +}); + +test("a balance that is not a complete amount is unknown, not zero", () => { + assert.equal(parseAccountBalance(undefined), null); + assert.equal(parseAccountBalance(""), null); + assert.equal(parseAccountBalance("Balance:"), null); + assert.equal(parseAccountBalance("$1,23.00"), null); +}); + +// Which account these numbers belong to is never asked, because inside a run of +// these two routes it cannot change: Route.Ledger is pushed only by tapping a +// card on Home, Route.AddTransaction only by the ledger's own button for its +// own account, and an accepted submit pops back to that same ledger. Reaching +// another account means passing through Home, so every frame that is not one of +// the two routes drops the carrier. +test("a frame off the account's own screens drops the carrier", () => { + for (const route of ["home", "login", "add-account", null]) { + assert.deepEqual( + readAccountBalance({ route, balanceText: "$196.00", previousCarrier: 10000 }), + { value: null, carrier: null, fresh: false }, + `route ${route} kept a carrier that may belong to another account`, + ); + } +}); + +test("a readable balance on either of the two screens closes the window", () => { + assert.deepEqual( + readAccountBalance({ route: "ledger", balanceText: "$196.00", previousCarrier: 10000 }), + { value: 19600, carrier: 19600, fresh: true }, + ); + assert.deepEqual( + readAccountBalance({ + route: "add-transaction", + balanceText: "Balance: $196.00", + previousCarrier: 10000, + }), + { value: 19600, carrier: 19600, fresh: true }, + ); +}); + +// The balance node scrolled out of the viewport is unknown, not a new value. +// The account still cannot have changed, so the carrier survives and the window +// stays open across the frame. +test("an unreadable balance keeps the carrier and does not close the window", () => { + assert.deepEqual( + readAccountBalance({ route: "ledger", balanceText: undefined, previousCarrier: 10000 }), + { value: 10000, carrier: 10000, fresh: false }, + ); +}); + +test("a double submit moves the account balance by twice what was typed", () => { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: doubleSubmit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: 49200, + }), + true, + ); +}); + +test("a double-submitted debit is caught by the same bound", () => { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: doubleSubmit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: -29200, + }), + true, + ); +}); + +test("one submit moving the balance by exactly the typed amount is the app working", () => { + for (const after of [29600, -9600]) { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: after, + }), + false, + ); + } +}); + +// The frames this bound is actually driven down, replayed off the recorded iOS +// run at runs/folio-ios/20260815-102711 (seed 7, 240 steps). It judged 18 of +// them and fired on none: every one was a single Tap on TxnSubmit landing back +// on the account's own ledger with the balance moved by exactly what was typed, +// which is the app working. Three of those readings are below, with the same +// frame as it looks when the one action commits twice. +// +// A double tap is nowhere in that list, and the run took three of them: all +// three landed on Home, where this conjunct has no balance to read and +// committedTransactionsExceedSubmits convicted instead. What reaches here is +// the interleaving where the second commit's pop does not run. +test("the ledger landings a real run produces are judged, and a doubled one fires", () => { + for (const [prev, typed] of [ + [357900, 25100], + [455800, 7900], + [682500, 19300], + ]) { + const judge = (currAccountBalance: number) => + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: typed!, + prevAccountBalance: prev!, + currAccountBalance, + }); + assert.equal(judge(prev! + typed!), false, `the recorded ${prev} -> ${prev! + typed!} was convicted`); + assert.equal(judge(prev! + 2 * typed!), true, `a second commit on ${prev} went unjudged`); + } +}); + +// A balance that has not moved is a commit still in flight (createTransaction +// runs in a coroutine), a submit the app rejected, or a tap that never landed. +// None of those is evidence, and an equality would convict all three. +test("a balance that has not moved yet is not evidence", () => { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: 10000, + }), + false, + ); +}); + +test("a window holding anything other than one submit is not attributable", () => { + for (const submitsInWindow of [0, 2, 37]) { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: 49200, + }), + false, + ); + } +}); + +test("an amount this reading cannot represent is vacuous, not a violation", () => { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: parseTypedAmount("not an amount"), + prevAccountBalance: 10000, + currAccountBalance: 49200, + }), + false, + ); + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: Number.MAX_SAFE_INTEGER + 2, + prevAccountBalance: 10000, + currAccountBalance: 49200, + }), + false, + ); +}); + +test("a balance too large to hold exactly is not compared", () => { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: Number.MAX_SAFE_INTEGER + 2, + currAccountBalance: 0, + }), + false, + ); + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 0, + currAccountBalance: Number.MAX_SAFE_INTEGER + 2, + }), + false, + ); +}); + +test("an unknown balance on either side is not evidence", () => { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: null, + currAccountBalance: 49200, + }), + false, + ); + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction: submit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: null, + }), + false, + ); +}); + +test("a step whose action was not a submit attributes nothing", () => { + for (const lastAction of [openLedger, openAddTxn, typeAmount, goBack, null]) { + assert.equal( + committedAmountExceedsOneSubmit({ + route: "ledger", + lastAction, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: 49200, + }), + false, + ); + } +}); + +test("Home shows every account's money, so it is not this comparison's scale", () => { + for (const route of ["home", "login", "add-account", null]) { + assert.equal( + committedAmountExceedsOneSubmit({ + route, + lastAction: doubleSubmit, + submitsInWindow: 1, + typedAmount: 19600, + prevAccountBalance: 10000, + currAccountBalance: 49200, + }), + false, + ); + } +}); + +// A step of the walk: the frame it landed on, the balance node that frame +// carried, the amount field as that frame shows it, and the action that got +// there. Driven through the same carrier and window the spec holds, `typed` +// included: the spec hands the landing frame's field to countSubmitsInWindow +// and the previous frame's to the property, and a walk that skips the first +// half drives a composition the spec never runs. +interface Frame { + route: string | null; + balanceText?: string; + typed?: string; + lastAction: ObservedAction | null; +} + +function walk(frames: readonly Frame[]) { + let carrier: number | null = null; + let submits = 0; + const verdicts: { violated: boolean; balance: number | null; submits: number }[] = []; + let previous: number | null = null; + let typedBefore = ""; + for (const frame of frames) { + const reading = readAccountBalance({ + route: frame.route, + balanceText: frame.balanceText, + previousCarrier: carrier, + }); + carrier = reading.carrier; + const window = countSubmitsInWindow({ + previousCount: submits, + lastAction: frame.lastAction, + amountText: frame.route === "add-transaction" ? (frame.typed ?? "") : undefined, + fresh: reading.fresh, + }); + submits = window.next; + verdicts.push({ + violated: committedAmountExceedsOneSubmit({ + route: frame.route, + lastAction: frame.lastAction, + submitsInWindow: window.reported, + typedAmount: parseTypedAmount(typedBefore), + prevAccountBalance: previous, + currAccountBalance: reading.value, + }), + balance: reading.value, + submits: window.reported, + }); + previous = reading.value; + typedBefore = frame.typed ?? ""; + } + return verdicts; +} + +// The trajectory of #78: open an account, open the transaction form, type, +// double tap. Not one frame of it is Home, so the Home readings the counting +// invariant compares never advance and it has nothing to say about any of it. +// This is what a 117-step stretch of that run looked like, and it is why the +// double tap at step 32 went unconvicted. +test("the Home window cannot judge a walk that never goes Home", () => { + const cards = [{ name: "Checking", balance: 10000, count: 3 as TxnCount }]; + let carrier: Record | null = homeTxnCountsOf(cards); + let submits = 0; + for (const lastAction of [openLedger, openAddTxn, typeAmount, doubleSubmit]) { + const reading = readHomeCards({ route: "ledger", reading: null, previousCarrier: carrier }); + const previous = carrier; + carrier = reading.carrier; + const window = countSubmitsInWindow({ previousCount: submits, lastAction, fresh: reading.fresh }); + submits = window.next; + assert.equal( + committedTransactionsExceedSubmits({ + countsBefore: previous, + countsAfter: reading.value, + submitsInWindow: window.reported, + }), + false, + ); + } +}); + +// The trajectory the recorded iOS run took to the frames this property judges, +// with the taps that reach TxnSubmit over an empty field: 40 of its 61 submit +// taps landed back on the transaction screen, and the field they read is the +// one the landing frame shows. Counting those as submits is what the run +// measures as the difference between 4 convictions and 0. Here the balance node +// is off the viewport while they happen, so nothing resets the window and the +// slack survives to the frame that matters. +test("submits the app must have refused do not buy a double tap an alibi", () => { + const verdicts = walk([ + { route: "ledger", balanceText: "$100.00", lastAction: openLedger }, + { route: "add-transaction", lastAction: openAddTxn }, + { route: "add-transaction", lastAction: submit }, + { route: "add-transaction", lastAction: submit }, + { route: "add-transaction", typed: "196", lastAction: typeAmount }, + { route: "ledger", balanceText: "$492.00", lastAction: doubleSubmit }, + ]); + assert.equal(verdicts[5]?.submits, 1); + assert.equal(verdicts[5]?.violated, true); +}); + +// The same trajectory, judged where the app actually is. The double tap sends +// two Submit events, and the frame it lands on says which of the two shapes +// they took: two commits and two pops reach Home, where the counting invariant +// judges them, and two commits with the second pop cancelled by the first stop +// on the account's own ledger, which is this one. The recorded iOS run took +// three double taps and all three landed on Home, so this frame is reasoned +// from the app's code (AddTransactionViewModel.submit commits inside +// viewModelScope, then pops) rather than measured. +test("the double tap is convicted on the frame it lands on", () => { + const verdicts = walk([ + { route: "ledger", balanceText: "$100.00", lastAction: openLedger }, + { route: "add-transaction", balanceText: "Balance: $100.00", lastAction: openAddTxn }, + { route: "add-transaction", balanceText: "Balance: $100.00", typed: "196", lastAction: typeAmount }, + { route: "ledger", balanceText: "$492.00", lastAction: doubleSubmit }, + ]); + assert.deepEqual( + verdicts.map(v => v.violated), + [false, false, false, true], + ); + assert.equal(verdicts[3]?.submits, 1); +}); + +test("the same walk with one transaction committed is silent throughout", () => { + const verdicts = walk([ + { route: "ledger", balanceText: "$100.00", lastAction: openLedger }, + { route: "add-transaction", balanceText: "Balance: $100.00", lastAction: openAddTxn }, + { route: "add-transaction", balanceText: "Balance: $100.00", typed: "196", lastAction: typeAmount }, + { route: "ledger", balanceText: "$296.00", lastAction: submit }, + { route: "add-transaction", balanceText: "Balance: $296.00", lastAction: openAddTxn }, + { route: "add-transaction", balanceText: "Balance: $296.00", typed: "50", lastAction: typeAmount }, + { route: "ledger", balanceText: "$346.00", lastAction: submit }, + ]); + assert.deepEqual( + verdicts.map(v => v.violated), + [false, false, false, false, false, false, false], + ); +}); + +// The reading a healthy app must survive: transactions arriving between two +// readings that the window can no longer attribute to one action. The balance +// node is off the viewport for a stretch, so the two numbers the property would +// compare straddle two commits, and the balance moves by 296.00 against a typed +// 50.00. Two submits in the window is not one, so there is nothing to judge. +test("transactions arriving between two readings do not convict a healthy app", () => { + const verdicts = walk([ + { route: "ledger", balanceText: "$100.00", lastAction: openLedger }, + { route: "add-transaction", lastAction: openAddTxn }, + { route: "add-transaction", typed: "196", lastAction: typeAmount }, + { route: "ledger", lastAction: submit }, + { route: "add-transaction", lastAction: openAddTxn }, + { route: "add-transaction", typed: "100", lastAction: typeAmount }, + { route: "ledger", balanceText: "$396.00", lastAction: submit }, + ]); + assert.deepEqual( + verdicts.map(v => v.violated), + [false, false, false, false, false, false, false], + ); + assert.equal(verdicts[6]?.submits, 2); + assert.equal(verdicts[6]?.balance, 39600); +}); + +// Attribution across accounts, which is the whole reason the carrier is dropped +// rather than carried. A $500.00 account is left behind for an empty one whose +// screens have not drawn their balance yet, and the submit into the new account +// lands with exactly one submit in the window: every gate this property has is +// open, and only the dropped carrier keeps it quiet. Carrying $500.00 across +// that frame reads as 30400 committed against 19600 typed, on an app that did +// nothing wrong. +test("a ledger opened for another account never inherits the old balance", () => { + for (const between of ["home", null]) { + const verdicts = walk([ + { route: "ledger", balanceText: "$500.00", lastAction: openAddTxn }, + { route: between, lastAction: goBack }, + { route: "ledger", lastAction: openLedger }, + { route: "add-transaction", typed: "196", lastAction: openAddTxn }, + { route: "ledger", balanceText: "$196.00", lastAction: submit }, + ]); + assert.deepEqual( + verdicts.map(v => v.violated), + [false, false, false, false, false], + `an account switch through ${between} was compared across accounts`, + ); + assert.equal(verdicts[4]?.submits, 1); + assert.equal(verdicts[4]?.balance, 19600); + } +}); diff --git a/pkg/spec/test/folio-new-account.test.ts b/pkg/spec/test/folio-new-account.test.ts index def23bb..98845f8 100644 --- a/pkg/spec/test/folio-new-account.test.ts +++ b/pkg/spec/test/folio-new-account.test.ts @@ -1,10 +1,17 @@ import assert from "node:assert/strict"; import { test } from "node:test"; -import { createdAccountHasNonZeroBalance } from "../../../examples/folio/sanderling/predicates.ts"; +import { + createdAccountHasNonZeroBalance, + initialsOf, +} from "../../../examples/folio/sanderling/predicates.ts"; -const created = { kind: "Tap", on: "testTag:AddAccountScreen > testTag:AddAccountSubmit" }; -const idle = { kind: "Tap", on: "testTag:HomeScreen > testTag:AccountCard" }; +const created = { + kind: "Tap", + on: "testTag:AddAccountScreen > testTag:AddAccountSubmit", + applied: true as const, +}; +const idle = { kind: "Tap", on: "testTag:HomeScreen > testTag:AccountCard", applied: true as const }; const account = (name: string, balance: number | null) => ({ name, balance }); @@ -27,7 +34,7 @@ test("a double-tapped create is judged the same way", () => { assert.equal( createdAccountHasNonZeroBalance({ route: "home", - lastAction: { kind: "DoubleTap", on: "id:AddAccountSubmit" }, + lastAction: { kind: "DoubleTap", on: "id:AddAccountSubmit", applied: true }, typedName: "Travel", before: [account("Checking", 0)], after: [account("Checking", 0), account("Travel", 5000)], @@ -205,9 +212,90 @@ test("the merged web key still matches the name that was typed", () => { ); }); -// Two cards answering to one typed name leave the appearance unattributable: -// the fuzzer creates duplicates from a five-name list, and the tree has been -// seen exposing the same card twice on a transition frame. +// The avatar the merged web key opens with, hand-computed off Format.kt rather +// than off the mirror, because a mirror checked against itself checks nothing. +// A single word gives its first two characters, several give the first letter +// of the first and of the last, and an empty name gives "?". +test("the initials a merged key opens with are the app's", () => { + const named: [string, string][] = [ + ["CH", "Checking"], + ["SA", "Savings"], + ["TR", "Travel"], + ["EF", "Emergency Fund"], + ["IN", "Investments"], + ["FU", "Fund"], + ["T2", "Travel 2024"], + ["A", "a"], + ["X9", "x9"], + ["-1", "-1"], + ["?", ""], + ["?", " "], + ]; + for (const [initials, name] of named) { + assert.equal(initialsOf(name), initials, `initials for ${JSON.stringify(name)}`); + } +}); + +// The attribution used to be a suffix test, and a suffix test hands the verdict +// to whichever OTHER account happens to end with the typed name. Home lists +// what fits the viewport, so the card that was just created is clipped out of +// the reading exactly as easily as any other, and the older account left in it +// is then judged for money it has held all along. +test("an older account whose name ends with the typed one is not the created one", () => { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: created, + typedName: "Fund", + before: [account("Checking", 0)], + after: [account("Checking", 0), account("Emergency Fund", 461012300)], + }), + false, + ); +}); + +test("the merged web key is matched whole too, not by its ending", () => { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: created, + typedName: "Fund", + before: [account("CHChecking", 0)], + after: [account("CHChecking", 0), account("EFEmergency Fund", 461012300)], + }), + false, + ); +}); + +// The card that was actually asked for is still judged, standing next to the +// account that merely ends with its name. +test("the created card is judged beside an account whose name ends with it", () => { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: created, + typedName: "Fund", + before: [account("Emergency Fund", 461012300)], + after: [account("Emergency Fund", 461012300), account("Fund", 5000)], + }), + true, + ); + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: created, + typedName: "Fund", + before: [account("EFEmergency Fund", 461012300)], + after: [account("EFEmergency Fund", 461012300), account("FUFund", 5000)], + }), + true, + ); +}); + +// Two cards answering to one typed name leave the appearance unattributable. +// Accounts.name is UNIQUE and Repository.createAccount rejects a name already +// taken, so the pair is one card the tree exposed twice on a transition frame, +// or two names the merged web key cannot tell apart. test("two cards matching the typed name are not attributable to the creation", () => { assert.equal( createdAccountHasNonZeroBalance({ @@ -215,7 +303,7 @@ test("two cards matching the typed name are not attributable to the creation", ( lastAction: created, typedName: "Travel", before: [account("Checking", 0)], - after: [account("Checking", 0), account("Travel", 5000), account("MyTravel", 900)], + after: [account("Checking", 0), account("Travel", 5000), account("Travel", 900)], }), false, ); @@ -233,3 +321,54 @@ test("a card that was already there is not a card that was just created", () => false, ); }); + +// The runner's foreground guard restarted the app after the create. A fresh +// launch draws Home from the top, so the visible set is whatever the new layout +// fits rather than what was there a step ago, and "appeared in the reading" is +// even less like "was created" than usual. The process may also have died +// before the write landed, which makes the card that carries the typed name an +// older account of that name coming into view. +test("a create the runner relaunched across attributes nothing", () => { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: { ...created, relaunched: true }, + typedName: "Travel", + before: [account("Checking", 0)], + after: [account("Checking", 0), account("Travel", 5000)], + }), + false, + ); +}); + +test("no relaunch reported still judges the account that was created", () => { + for (const relaunched of [null, undefined]) { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: { ...created, relaunched }, + typedName: "Travel", + before: [account("Checking", 0)], + after: [account("Checking", 0), account("Travel", 5000)], + }), + true, + ); + } +}); + +// The apply call failed with the gesture possibly already delivered, so nobody +// knows whether that account was created. The card carrying the typed name may +// be an older one that scrolled into view, and attributing it to a creation +// that may never have happened is a conviction built on a guess. +test("a create the runner could not confirm attributes nothing", () => { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: { ...created, applied: null }, + typedName: "Travel", + before: [account("Checking", 0)], + after: [account("Checking", 0), account("Travel", 5000)], + }), + false, + ); +}); diff --git a/pkg/spec/test/folio-submit-balance-predicate.test.ts b/pkg/spec/test/folio-submit-balance-predicate.test.ts index 57a94ce..e2be9da 100644 --- a/pkg/spec/test/folio-submit-balance-predicate.test.ts +++ b/pkg/spec/test/folio-submit-balance-predicate.test.ts @@ -3,16 +3,16 @@ import { test } from "node:test"; import { parseTypedAmount, - submitChangesBalanceByTypedAmount, + submitChangesBalanceByAtMostTypedAmount, } from "../../../examples/folio/sanderling/predicates.ts"; const submitOn = "testTag:LedgerScreen > testTag:TxnSubmit"; test("single submit: delta matches typed amount", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -24,9 +24,9 @@ test("single submit: delta matches typed amount", () => { test("double submit: delta is twice the typed amount, fires", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -38,9 +38,9 @@ test("double submit: delta is twice the typed amount, fires", () => { test("DoubleTap kind also caught when delta exceeds typed amount", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 0, @@ -52,9 +52,9 @@ test("DoubleTap kind also caught when delta exceeds typed amount", () => { test("wrong action kind: vacuous true even with mismatch", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "InputText", on: submitOn }, + lastAction: { kind: "InputText", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -66,9 +66,9 @@ test("wrong action kind: vacuous true even with mismatch", () => { test("wrong target: vacuous true even with mismatch", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: "testTag:LoginScreen > testTag:LoginSubmit" }, + lastAction: { kind: "Tap", on: "testTag:LoginScreen > testTag:LoginSubmit", applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -80,7 +80,7 @@ test("wrong target: vacuous true even with mismatch", () => { test("null lastAction: vacuous true", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", lastAction: null, submitsInWindow: 1, @@ -94,9 +94,9 @@ test("null lastAction: vacuous true", () => { test("zero typedAmount: vacuous true", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 0, prevTotalBalance: 1000, @@ -108,9 +108,9 @@ test("zero typedAmount: vacuous true", () => { test("selector as object: coerced safely and TxnSubmit detected", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: { testTag: "TxnSubmit" } }, + lastAction: { kind: "Tap", on: { testTag: "TxnSubmit" }, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 0, @@ -122,9 +122,9 @@ test("selector as object: coerced safely and TxnSubmit detected", () => { test("selector as object without TxnSubmit: vacuous true", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: { testTag: "LoginSubmit" } }, + lastAction: { kind: "Tap", on: { testTag: "LoginSubmit" }, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 0, @@ -136,9 +136,9 @@ test("selector as object without TxnSubmit: vacuous true", () => { test("raw whole-dollar input: single submit clears", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("50"), prevTotalBalance: 5000, @@ -150,9 +150,9 @@ test("raw whole-dollar input: single submit clears", () => { test("raw whole-dollar input: double submit fires", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("50"), prevTotalBalance: 5000, @@ -164,9 +164,9 @@ test("raw whole-dollar input: double submit fires", () => { test("decimal input from empty prior balance clears", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("5.50"), prevTotalBalance: 0, @@ -178,9 +178,9 @@ test("decimal input from empty prior balance clears", () => { test("DoubleTap kind with raw whole-dollar input fires", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("100"), prevTotalBalance: 0, @@ -192,9 +192,9 @@ test("DoubleTap kind with raw whole-dollar input fires", () => { test("route gate: ledger landing with stale carrier is skipped", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "ledger", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -206,9 +206,9 @@ test("route gate: ledger landing with stale carrier is skipped", () => { test("route gate: add-transaction landing with double-submit delta is skipped", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "add-transaction", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -220,9 +220,9 @@ test("route gate: add-transaction landing with double-submit delta is skipped", test("route gate: null route is skipped", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: null, - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -234,9 +234,9 @@ test("route gate: null route is skipped", () => { test("route gate: home landing with matching delta passes", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -248,9 +248,9 @@ test("route gate: home landing with matching delta passes", () => { test("route gate: home landing with double-insert delta fires", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -273,9 +273,9 @@ test("above 2^53 the arithmetic itself is wrong, which is why the guard exists", test("above 2^53 a healthy single submit is not reported", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1600, prevTotalBalance: HUGE_BALANCE, @@ -287,9 +287,9 @@ test("above 2^53 a healthy single submit is not reported", () => { test("above 2^53 a double-submit delta is not reported either", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1600, prevTotalBalance: HUGE_BALANCE, @@ -301,9 +301,9 @@ test("above 2^53 a double-submit delta is not reported either", () => { test("an unreadable previous balance above 2^53 is not evidence", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1600, prevTotalBalance: HUGE_BALANCE, @@ -318,9 +318,9 @@ test("an unreadable previous balance above 2^53 is not evidence", () => { // and must not convict on one it cannot hold. test("typed amount above 2^53 is not evidence", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1e23, prevTotalBalance: 0, @@ -334,9 +334,9 @@ test("typed amount above 2^53 is not evidence", () => { // more is where counting stops being exact. test("boundary: a double submit landing exactly on MAX_SAFE_INTEGER still fires", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 4503599627370495, prevTotalBalance: 0, @@ -348,9 +348,9 @@ test("boundary: a double submit landing exactly on MAX_SAFE_INTEGER still fires" test("boundary: a single submit landing exactly on MAX_SAFE_INTEGER passes", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 9007199254740991, prevTotalBalance: 0, @@ -362,9 +362,9 @@ test("boundary: a single submit landing exactly on MAX_SAFE_INTEGER passes", () test("boundary: one cent past MAX_SAFE_INTEGER stops being evidence", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 4503599627370496, prevTotalBalance: 0, @@ -379,9 +379,9 @@ test("boundary: one cent past MAX_SAFE_INTEGER stops being evidence", () => { // typed amount, so a mismatch here is real and must still be reported. test("a large but exact difference between safe balances still fires", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: -9007199254740991, @@ -395,9 +395,9 @@ test("a large but exact difference between safe balances still fires", () => { // and the property must stay quiet rather than demand a 1e23-cent move. test("21-digit typed amount with an unmoved balance is not a violation", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("999999999999999999999"), prevTotalBalance: 220900, @@ -415,9 +415,9 @@ test("21-digit typed amount with an unmoved balance is not a violation", () => { // and an unrelated 26200 credit. test("freshness: two submits in the window is vacuous, not a conviction", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 2, typedAmount: 19600, prevTotalBalance: 0, @@ -429,9 +429,9 @@ test("freshness: two submits in the window is vacuous, not a conviction", () => test("freshness: two submits cannot convict even on a clean 2x delta", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 2, typedAmount: 500, prevTotalBalance: 1000, @@ -446,9 +446,9 @@ test("freshness: two submits cannot convict even on a clean 2x delta", () => { // (nothing to attribute the move to), and two or more means the move is shared. test("freshness boundary: exactly one submit is the window that convicts", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 19600, prevTotalBalance: 0, @@ -460,9 +460,9 @@ test("freshness boundary: exactly one submit is the window that convicts", () => test("freshness boundary: one submit with a healthy 1x delta still passes", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 19600, prevTotalBalance: 0, @@ -474,9 +474,9 @@ test("freshness boundary: one submit with a healthy 1x delta still passes", () = test("freshness boundary: three submits is vacuous", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 3, typedAmount: 500, prevTotalBalance: 0, @@ -491,9 +491,9 @@ test("freshness boundary: three submits is vacuous", () => { // submit in it explains no balance move. test("freshness boundary: a window with no submit in it is vacuous", () => { assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 0, typedAmount: 500, prevTotalBalance: 1000, @@ -502,3 +502,124 @@ test("freshness boundary: a window with no submit in it is vacuous", () => { true, ); }); + +// applied: null is the runner saying it dispatched the tap and never learned +// whether it landed. Under the bound that buys the app nothing it did not +// already have: a submit that committed nothing leaves the balance where it +// was, and a balance that has not moved is under any bound. What the window +// still promises is that no OTHER submit action could have moved it, because +// countSubmitsInWindow counts an unconfirmed tap exactly like a confirmed one. +// So a move of twice the typed amount is the same double commit either way. +test("a submit the runner could not confirm is still held to the bound", () => { + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { kind: "DoubleTap", on: submitOn, applied: null }, + submitsInWindow: 1, + typedAmount: 500, + prevTotalBalance: 1000, + currTotalBalance: 2000, + }), + false, + ); +}); + +// The write finishes before AddTransactionViewModel navigates, but nothing +// establishes that Home's total has re-rendered by the time the frame is read: +// the store's flow re-emits on its own schedule and the destination composes off +// whatever value it has. A total that has not caught up has not moved at all, +// and an equality reads that as the app having ignored the amount. +test("a commit the Home total has not caught up with is not a violation", () => { + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { kind: "Tap", on: submitOn, applied: true }, + submitsInWindow: 1, + typedAmount: 19600, + prevTotalBalance: 220900, + currTotalBalance: 220900, + }), + true, + ); +}); + +// What the bound gives up, and it is a real bug class: an app that moves the +// balance by LESS than the amount typed. Nothing in this spec judges that any +// more. It cannot be told apart from a total that has not caught up, and a check +// that fires on both is not evidence about either. +test("an under-move is no longer judged, which is the trade", () => { + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { kind: "Tap", on: submitOn, applied: true }, + submitsInWindow: 1, + typedAmount: 19600, + prevTotalBalance: 0, + currTotalBalance: 10000, + }), + true, + ); +}); + +// The witness measured on four recorded android runs, all four of which convict +// here and nowhere else: the double tap moved the total by 6400 against 3200 +// typed. The bound has to keep every one of them. +test("the measured double submit still fires under the bound", () => { + for (const [prev, curr] of [ + [17952800, 17959200], + [19796100, 19802500], + [200000032904800, 200000032911200], + ]) { + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, + submitsInWindow: 1, + typedAmount: 3200, + prevTotalBalance: prev ?? null, + currTotalBalance: curr ?? null, + }), + false, + `the double submit at ${prev} -> ${curr} stopped firing`, + ); + } +}); + +// relaunched: true is the runner saying its foreground guard restarted the app +// after this action, so the two totals being compared were read from two +// different processes. SqlLedgerStore starts every one of them on +// stateIn(Eagerly, emptyList()) and HomeScreen composes formatCents(total) off +// whatever the flow holds, so the restarted app draws $0.00 into TotalBalance +// until sqlite answers. That reading is not a total this tap moved, and it is +// as far from the last one as the account is rich. +test("a total drawn by a restarted process is not compared with the old one", () => { + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { kind: "Tap", on: submitOn, applied: true, relaunched: true }, + submitsInWindow: 1, + typedAmount: 500, + prevTotalBalance: 455800, + currTotalBalance: 0, + }), + true, + ); +}); + +// The guard must not become a way of switching the property off. No relaunch +// reported is the ordinary case, and web and iOS cannot report one at all. +test("no relaunch reported still convicts a double submit", () => { + for (const relaunched of [null, undefined]) { + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { kind: "DoubleTap", on: submitOn, applied: true, relaunched }, + submitsInWindow: 1, + typedAmount: 500, + prevTotalBalance: 1000, + currTotalBalance: 2000, + }), + false, + ); + } +}); diff --git a/pkg/spec/test/folio-submit-window.test.ts b/pkg/spec/test/folio-submit-window.test.ts index da422ed..6349113 100644 --- a/pkg/spec/test/folio-submit-window.test.ts +++ b/pkg/spec/test/folio-submit-window.test.ts @@ -2,6 +2,7 @@ import assert from "node:assert/strict"; import { test } from "node:test"; import { + committedTransactionsExceedSubmits, countSubmitsInWindow, isTxnSubmitTap, readHomeTotalBalance, @@ -66,6 +67,117 @@ test("a second submit with no Home reading between them counts two", () => { ); }); +// The window is a budget: an upper bound on the transactions the interval could +// hold. A tap the app's own parser must have refused spends none of it, and on +// android it does not even reach the parser, because TxnSubmit is +// clickable(enabled = amount.isNotBlank()). Measured over four recorded android +// runs, 19, 11, 25 and 25 of 35, 26, 42 and 42 submit taps landed on the +// transaction screen with the amount field empty, so more than half the budget +// was being spent on taps that cannot commit anything. +test("a submit the app must have refused does not spend the window's budget", () => { + for (const amountText of ["", " ", "0", "0.00", "00", "5.", "abc"]) { + assert.deepEqual( + countSubmitsInWindow({ + previousCount: 0, + lastAction: { kind: "Tap", on: submitOn }, + amountText, + fresh: false, + }), + { reported: 0, next: 0 }, + `amount ${JSON.stringify(amountText)} was counted as a possible commit`, + ); + } +}); + +// Folio caps a transaction at $1,000,000.00 (MAX_TRANSACTION_AMOUNT_CENTS, in +// core/data/Repository.kt), and AddTransactionViewModel.submit refuses anything +// over it before a coroutine starts. The fuzzer's corpus carries +// "999999999999999999999", AMOUNT_REGEX lets it into the field and it reaches +// the button, so this is a refusal the window used to pay for. +test("an amount over the app's cap cannot commit", () => { + for (const amountText of ["1000000.01", "1,000,001", "999999999999999999999"]) { + assert.deepEqual( + countSubmitsInWindow({ + previousCount: 0, + lastAction: { kind: "Tap", on: submitOn }, + amountText, + fresh: false, + }), + { reported: 0, next: 0 }, + `amount ${JSON.stringify(amountText)} was counted as a possible commit`, + ); + } +}); + +// The field as the landing frame shows it, which is the form state the tap read: +// nothing between the two changes it. Anywhere but the transaction screen there +// is no field to read, and unknown has to count. The cap itself is an amount the +// app takes, so it counts too. +test("an amount that could commit, or that nobody could read, spends the budget", () => { + for (const amountText of ["5", "0.01", "1,000", "1000000.00", "999999.99", undefined]) { + assert.deepEqual( + countSubmitsInWindow({ + previousCount: 0, + lastAction: { kind: "Tap", on: submitOn }, + amountText, + fresh: false, + }), + { reported: 1, next: 1 }, + `amount ${JSON.stringify(amountText)} was dropped from the budget`, + ); + } +}); + +// The one thing that can put a different form state on screen than the one the +// tap read: the runner restarting the app, which the tap survives and the typed +// amount does not. The field a fresh process draws is empty whatever was +// submitted, so it proves nothing and the submit keeps its place in the budget. +test("a submit across a relaunch spends the budget whatever the field shows", () => { + assert.deepEqual( + countSubmitsInWindow({ + previousCount: 0, + lastAction: { kind: "Tap", on: submitOn, applied: true, relaunched: true }, + amountText: "", + fresh: false, + }), + { reported: 1, next: 1 }, + ); +}); + +// What the budget costs the counting invariant, in the shape of the iOS run in +// #78: a stretch of the walk that never went Home, most of it taps on a submit +// button with nothing typed into the form, and one double tap that committed +// twice. Counting the refused taps hands the app five transactions of slack it +// never used, and two rows against six actions is no violation. +test("refused submits used to hide a double submit behind their own budget", () => { + const frames = [ + { amountText: "", lastAction: { kind: "Tap", on: submitOn } }, + { amountText: "", lastAction: { kind: "Tap", on: submitOn } }, + { amountText: "", lastAction: { kind: "Tap", on: submitOn } }, + { amountText: "", lastAction: { kind: "Tap", on: submitOn } }, + { amountText: "", lastAction: { kind: "Tap", on: submitOn } }, + { amountText: undefined, lastAction: { kind: "DoubleTap", on: submitOn } }, + ]; + let budget = 0; + for (const frame of frames) { + budget = countSubmitsInWindow({ + previousCount: budget, + lastAction: frame.lastAction, + amountText: frame.amountText, + fresh: false, + }).next; + } + assert.equal(budget, 1); + assert.equal( + committedTransactionsExceedSubmits({ + countsBefore: { Checking: 3 }, + countsAfter: { Checking: 5 }, + submitsInWindow: budget, + }), + true, + ); +}); + // The two traces the freshness rule exists to tell apart, driven step by step // through the same pair of carriers the spec holds. function run(steps: { route: string | null; totalText?: string; lastAction: unknown }[]) { @@ -149,3 +261,25 @@ test("an unreadable Home does not close the window", () => { assert.equal(trace[2]?.total, null); assert.equal(trace[3]?.submits, 2); }); + +// The window is an upper bound on the submits it holds, so a submit whose +// dispatch the runner could not confirm belongs in it: the tap may well have +// landed, and a bound that leaves it out is one the transaction it committed +// exceeds. That is the false conviction, a rise of one against a window of +// zero, on the property carrying most of the detection on android. +test("a submit the runner could not confirm still counts toward the window", () => { + const window = countSubmitsInWindow({ + previousCount: 0, + lastAction: { kind: "Tap", on: submitOn, applied: null }, + fresh: true, + }); + assert.equal(window.reported, 1); + assert.equal( + committedTransactionsExceedSubmits({ + countsBefore: { Travel: 3 }, + countsAfter: { Travel: 4 }, + submitsInWindow: window.reported, + }), + false, + ); +}); diff --git a/pkg/spec/test/folio-transition-frame.test.ts b/pkg/spec/test/folio-transition-frame.test.ts index f785642..08e3744 100644 --- a/pkg/spec/test/folio-transition-frame.test.ts +++ b/pkg/spec/test/folio-transition-frame.test.ts @@ -6,7 +6,7 @@ import { readHomeCards, readHomeTotalBalance, routeOfFrame, - submitChangesBalanceByTypedAmount, + submitChangesBalanceByAtMostTypedAmount, } from "../../../examples/folio/sanderling/predicates.ts"; // The spec's own screen table. A frame is the set of markers its accessibility @@ -94,7 +94,7 @@ test("the measured android transition chain no longer convicts at delta 0", () = const step = ( tags: string[], totalText: string | undefined, - lastAction: { kind: string; on: string } | null, + lastAction: { kind: string; on: string; applied: true } | null, ) => { const route = routeOfFrame(SCREENS, frame(...tags)); const reading = readHomeTotalBalance({ route, totalText, previousCarrier: carrier }); @@ -104,8 +104,12 @@ test("the measured android transition chain no longer convicts at delta 0", () = return { route, total: reading.value, submits: window.reported }; }; - const back = { kind: "DoubleTap", on: "id:BackButton" }; - const phantomSubmit = { kind: "Tap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" }; + const back = { kind: "DoubleTap", on: "id:BackButton", applied: true as const }; + const phantomSubmit = { + kind: "Tap", + on: "testTag:AddTransactionScreen > testTag:TxnSubmit", + applied: true as const, + }; const transition = step(["AddTransactionScreen", "HomeScreen"], "$86,911.00", back); assert.equal(transition.route, null); @@ -115,7 +119,7 @@ test("the measured android transition chain no longer convicts at delta 0", () = const landing = step(["HomeScreen"], "$86,911.00", phantomSubmit); assert.equal(landing.submits, 6); assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: landing.route, lastAction: phantomSubmit, submitsInWindow: landing.submits, @@ -127,9 +131,13 @@ test("the measured android transition chain no longer convicts at delta 0", () = ); // What the reset bought the old spec: the same landing, judged against a - // window of one and a total the transition frame had already banked. + // window of one and a total the transition frame had already banked. It + // convicted on a delta of zero, and that shape cannot convict any more even + // with the window reset back to one, because the property is a bound rather + // than an equality. A balance that did not move is under any typed amount, + // whether nothing was submitted or the total has not caught up yet. assert.equal( - submitChangesBalanceByTypedAmount({ + submitChangesBalanceByAtMostTypedAmount({ route: "home", lastAction: phantomSubmit, submitsInWindow: 1, @@ -137,6 +145,20 @@ test("the measured android transition chain no longer convicts at delta 0", () = prevTotalBalance: 8691100, currTotalBalance: 8691100, }), + true, + ); + + // The double tap it was always meant to catch is untouched by that: two + // 33900 debits against one action still exceed the amount typed for it. + assert.equal( + submitChangesBalanceByAtMostTypedAmount({ + route: "home", + lastAction: { ...phantomSubmit, kind: "DoubleTap" }, + submitsInWindow: 1, + typedAmount: 33900, + prevTotalBalance: 8691100, + currTotalBalance: 8691100 - 67800, + }), false, ); }); diff --git a/pkg/spec/test/web-dom-harness.ts b/pkg/spec/test/web-dom-harness.ts index cc0e671..80feb09 100644 --- a/pkg/spec/test/web-dom-harness.ts +++ b/pkg/spec/test/web-dom-harness.ts @@ -1,8 +1,21 @@ -// A minimal stand-in for the DOM surface the web host reads, shared by the web -// runtime's own tests and the cross-host eligibility test. The host asks the -// document for three things -- every element, the tappable set, the editable set -// -- and reads geometry, `disabled` and the scroll extents off each element, so -// that is all a fake has to answer. +// A small DOM the web runtime can be driven over, shared by the web runtime's +// own tests and the cross-host eligibility test. +// +// It is a fake, but the structure is real: elements nest, a host owns a shadow +// root, and querySelectorAll WALKS the tree and stops at a shadow boundary +// exactly as the browser's does. That is what makes the shadow descent in +// deepQueryAll and expandShadowContent (src/web-runtime.ts) observable here at +// all; the previous harness answered three fixed selectors from a flat list, so +// deleting either descent changed no test result. +// +// What it fabricates is layout: getBoundingClientRect, scrollHeight and +// clientHeight are handed over from the spec. No headless DOM computes those, +// and they are precisely the facts collectTargets reads, so a real DOM +// implementation would have to be stubbed for them anyway. +// +// An unsupported selector throws rather than matching nothing, so a test whose +// selector this cannot parse fails loudly instead of quietly asserting over an +// empty list. import { __testing__ } from "../src/web-runtime.ts"; @@ -23,23 +36,46 @@ export interface FakeElementSpec { label?: string; alt?: string; title?: string; - // clickable/editable place the element in the selector sets the host queries; - // the fake answers those queries directly rather than matching CSS. + // text is what an ax element handle reports as `text`, the same field the + // goja host reads off a hierarchy node, so a test can name WHICH of two + // same-id elements a lookup resolved to. + text?: string; + attrs?: Record; + // clickable/editable place the element in the two fact sets the host queries + // by selector. They are answered from these flags rather than by matching + // their CSS: the cross-host golden (fixtures/host-parity-golden.json, built + // row for row in internal/verifier/host_parity_test.go) pins fact + // combinations no CSS can produce, such as an that is editable and + // not clickable. A test states the facts there; this harness reports them. clickable?: boolean; editable?: boolean; disabled?: boolean; // overflows makes the element's content taller than its box, which is how the // host decides an element is scrollable. overflows?: boolean; + children?: FakeElementSpec[]; + shadow?: FakeElementSpec[]; } -export interface FakeElement extends FakeElementSpec { +export interface FakeRoot { + children: FakeElement[]; + querySelectorAll(selector: string): FakeElement[]; +} + +export interface FakeElement extends Omit { tagName: string; type: string; isContentEditable: boolean; id: string; + className: string; + textContent: string; dataset: Record; + parentElement: FakeElement | null; + children: FakeElement[]; + shadowRoot: FakeRoot | null; getAttribute(name: string): string | null; + matches(selector: string): boolean; + querySelectorAll(selector: string): FakeElement[]; scrollHeight: number; clientHeight: number; scrollWidth: number; @@ -56,15 +92,30 @@ export interface FakeElement extends FakeElementSpec { export function fakeElement(spec: FakeElementSpec): FakeElement { const editable = spec.editable ?? false; - return { + const attributes: Record = { ...spec.attrs }; + if (spec.id !== undefined) attributes.id = spec.id; + if (spec.testid !== undefined) attributes["data-testid"] = spec.testid; + if (spec.label !== undefined) attributes["aria-label"] = spec.label; + if (spec.alt !== undefined) attributes.alt = spec.alt; + if (spec.title !== undefined) attributes.title = spec.title; + const element: FakeElement = { ...spec, tagName: spec.tag.toUpperCase(), type: spec.tag === "input" ? "text" : "", isContentEditable: editable && spec.tag !== "input" && spec.tag !== "textarea", id: spec.id ?? "", + className: attributes.class ?? "", + textContent: spec.text ?? "", dataset: { testid: spec.testid }, - getAttribute: (name: string) => - ({ "aria-label": spec.label, alt: spec.alt, title: spec.title })[name] ?? null, + parentElement: null, + children: (spec.children ?? []).map(fakeElement), + shadowRoot: null, + getAttribute: (name: string) => attributes[name] ?? null, + // elementHandle asks an element about itself rather than sweeping the + // document for it, so a fake that only answers querySelectorAll reports + // every element as untappable. + matches: (selector: string) => matchesQuery(element, selector), + querySelectorAll: (selector: string) => queryScope(element, selector), scrollHeight: spec.overflows ? spec.height * 2 : spec.height, clientHeight: spec.height, scrollWidth: spec.width, @@ -78,27 +129,170 @@ export function fakeElement(spec: FakeElementSpec): FakeElement { bottom: spec.y + spec.height, }), }; + for (const child of element.children) child.parentElement = element; + if (spec.shadow) element.shadowRoot = fakeRoot(spec.shadow.map(fakeElement)); + return element; } -// withFakeDocument installs a document answering the host's three queries over -// `elements`, resets the host's per-tick cache, and restores the real document -// afterwards. +// A shadow root's children have no parentElement, as in a real DOM, so a +// descendant selector cannot reach across the boundary from either side. +function fakeRoot(children: FakeElement[]): FakeRoot { + const root: FakeRoot = { + children, + querySelectorAll: (selector: string) => queryScope(root, selector), + }; + return root; +} + +function queryScope(scope: { children: FakeElement[] }, selector: string): FakeElement[] { + const found: FakeElement[] = []; + const walk = (nodes: FakeElement[]): void => { + for (const node of nodes) { + if (matchesQuery(node, selector)) found.push(node); + walk(node.children); + } + }; + walk(scope.children); + return found; +} + +function matchesQuery(element: FakeElement, selector: string): boolean { + if (selector === TAPPABLE_SELECTOR) return element.clickable === true; + if (selector === EDITABLE_SELECTOR) return element.editable === true; + return matchesSelectorList(element, selector); +} + +function matchesSelectorList(element: FakeElement, selector: string): boolean { + return splitTopLevel(selector, ",").some((complex) => matchesComplex(element, complex)); +} + +function matchesComplex(element: FakeElement, complex: string): boolean { + const compounds = splitTopLevel(complex, " "); + const subject = compounds.pop(); + if (subject === undefined) return false; + if (!matchesCompound(element, subject)) return false; + let ancestor = element.parentElement; + for (const compound of compounds.reverse()) { + while (ancestor && !matchesCompound(ancestor, compound)) ancestor = ancestor.parentElement; + if (!ancestor) return false; + ancestor = ancestor.parentElement; + } + return true; +} + +const TAG_NAME = /^[a-zA-Z][a-zA-Z0-9-]*/; +const ATTRIBUTE = /^([a-zA-Z][\w-]*)(?:([~^]?)=(.+))?$/; + +function matchesCompound(element: FakeElement, compound: string): boolean { + let rest = compound; + while (rest.length > 0) { + if (rest.startsWith("*")) { + rest = rest.slice(1); + continue; + } + if (rest.startsWith("[")) { + const end = closingIndex(rest, "[", "]"); + if (!matchesAttribute(element, rest.slice(1, end))) return false; + rest = rest.slice(end + 1); + continue; + } + if (rest.startsWith(":is(") || rest.startsWith(":not(")) { + const end = closingIndex(rest, "(", ")"); + const inner = rest.slice(rest.indexOf("(") + 1, end); + const anyMatched = splitTopLevel(inner, ",").some((part) => + matchesSelectorList(element, part), + ); + if (rest.startsWith(":is(") ? !anyMatched : anyMatched) return false; + rest = rest.slice(end + 1); + continue; + } + const tag = TAG_NAME.exec(rest); + if (!tag) throw new Error(`web-dom-harness cannot parse selector ${JSON.stringify(compound)}`); + if (element.tagName !== tag[0].toUpperCase()) return false; + rest = rest.slice(tag[0].length); + } + return true; +} + +function matchesAttribute(element: FakeElement, body: string): boolean { + const parsed = ATTRIBUTE.exec(body); + if (!parsed) throw new Error(`web-dom-harness cannot parse attribute [${body}]`); + const [, name, operator, quoted] = parsed; + const actual = element.getAttribute(name!); + if (actual === null) return false; + if (quoted === undefined) return true; + const value = unescapeCss(quoted.replace(/^"(.*)"$/, "$1").replace(/^'(.*)'$/, "$1")); + if (operator === "~") return actual.split(/\s+/).includes(value); + if (operator === "^") return actual.startsWith(value); + return actual === value; +} + +// Selector values reach the harness escaped by CSS.escape, so `[id="1a"]` +// arrives as `[id="\31 a"]` and comparing it raw would never match. +function unescapeCss(value: string): string { + return value.replace(/\\([0-9a-fA-F]{1,6}) ?|\\(.)/g, (_, hex: string, literal: string) => + hex ? String.fromCodePoint(parseInt(hex, 16)) : literal, + ); +} + +function closingIndex(input: string, open: string, close: string): number { + let depth = 0; + let quote = ""; + for (let index = input.indexOf(open); index < input.length; index++) { + const character = input[index]!; + if (quote) { + if (character === quote) quote = ""; + continue; + } + if (character === '"' || character === "'") quote = character; + else if (character === open) depth++; + else if (character === close && --depth === 0) return index; + } + throw new Error(`web-dom-harness cannot parse selector ${JSON.stringify(input)}`); +} + +function splitTopLevel(input: string, separator: string): string[] { + const parts: string[] = []; + let current = ""; + let depth = 0; + let quote = ""; + for (const character of input) { + if (quote) { + current += character; + if (character === quote) quote = ""; + continue; + } + if (character === '"' || character === "'") quote = character; + else if (character === "(" || character === "[") depth++; + else if (character === ")" || character === "]") depth--; + else if (depth === 0 && (character === separator || (separator === " " && /\s/.test(character)))) { + parts.push(current); + current = ""; + continue; + } + current += character; + } + parts.push(current); + return parts.map((part) => part.trim()).filter((part) => part.length > 0); +} + +// withFakeDocument installs a document whose top-level children are `elements`, +// resets the host's per-tick cache, and restores the real globals afterwards. +// window goes in alongside document because buildState reads both, so an +// extractor reaching state.ax needs it. export function withFakeDocument(elements: FakeElement[], run: () => void): void { const global = globalThis as Record; - const original = global.document; - const answers: Record = { - "*": elements, - [TAPPABLE_SELECTOR]: elements.filter((element) => element.clickable), - [EDITABLE_SELECTOR]: elements.filter((element) => element.editable), - }; - global.document = { - querySelectorAll: (selector: string) => answers[selector] ?? [], - }; + const originalDocument = global.document; + const originalWindow = global.window; + const document: FakeRoot = fakeRoot(elements); + global.document = document; + global.window = {}; __testing__.resetTargetCache(); try { run(); } finally { __testing__.resetTargetCache(); - global.document = original; + global.document = originalDocument; + global.window = originalWindow; } } diff --git a/pkg/spec/test/web-runtime.test.ts b/pkg/spec/test/web-runtime.test.ts index a48f0d0..ea9bfa8 100644 --- a/pkg/spec/test/web-runtime.test.ts +++ b/pkg/spec/test/web-runtime.test.ts @@ -84,6 +84,7 @@ test("installRuntime defined the host-invoked globals", () => { }); const { fakeElement, withFakeDocument } = await import("./web-dom-harness.ts"); +type FakeElementSpec = Parameters[0]; // The host reports facts and never routes verbs: which of these a verb may act // on is decided by the shared rule in src/targets.ts, exercised across both @@ -175,6 +176,55 @@ test("queryTargets leaves duplicated identities unnamed", () => { }); }); +// The enumeration ORDER is the parity contract. buildTree in +// internal/driver/chrome/driver.go emits a host's shadow children before its +// light ones, and TestHierarchy_DerivesTheSameFactsAsTheWebRuntime compares the +// two enumerations position by position. +test("queryTargets splices shadow content in before the host's light children", () => { + const page = fakeElement({ + tag: "div", x: 0, y: 0, width: 400, height: 800, id: "page", + children: [ + { + tag: "div", x: 0, y: 0, width: 400, height: 100, id: "mount", + shadow: [ + { tag: "button", x: 0, y: 0, width: 40, height: 20, id: "shadow-save", clickable: true }, + ], + children: [{ tag: "div", x: 0, y: 20, width: 40, height: 20, id: "mount-light-child" }], + }, + { tag: "div", x: 0, y: 100, width: 400, height: 100, id: "after" }, + ], + }); + withFakeDocument([page], () => { + assert.deepEqual( + host.queryTargets().map((target) => target.selector), + ["id:page", "id:mount", "id:shadow-save", "id:mount-light-child", "id:after"], + ); + }); +}); + +// The tappable set is resolved by selector, and querySelectorAll stops dead at +// a shadow boundary, so a control inside a shadow root carries the clickable +// fact only if the selector sweep descends. A Compose for Web app keeps every +// control it has on the far side of one boundary. +test("queryTargets reports a shadow-hosted control as clickable", () => { + const mount = fakeElement({ + tag: "div", x: 0, y: 0, width: 400, height: 100, id: "mount", + shadow: [ + { tag: "button", x: 0, y: 0, width: 40, height: 20, id: "shadow-save", clickable: true }, + { tag: "input", x: 0, y: 20, width: 40, height: 20, id: "shadow-amount", editable: true }, + ], + }); + withFakeDocument([mount], () => { + const targets = host.queryTargets(); + assert.deepEqual( + targets.map((target) => target.selector), + ["id:mount", "id:shadow-save", "id:shadow-amount"], + ); + assert.equal(targets[1]!.clickable, true); + assert.equal(targets[2]!.editable, true); + }); +}); + test("queryTargets caches within a tick until reset", () => { const button = fakeElement({ tag: "button", x: 0, y: 0, width: 10, height: 10, clickable: true }); withFakeDocument([button], () => { @@ -288,7 +338,7 @@ test("an extractor that returned undefined keeps its index through JSON", () => // state.lastAction is the one piece of state the page cannot observe for // itself: only the runner knows which action it actually applied. While the web // runtime hardcoded null there, a spec property gated on the last action (e.g. -// folio's submitMovesBalanceByTypedAmount, which only looks at taps on +// folio's submitMovesBalanceByAtMostTypedAmount, which only looks at taps on // TxnSubmit) was vacuously true on web forever, and the run went green having // checked nothing. function lastActionSeenByASpec(pushed: unknown): unknown { @@ -315,6 +365,35 @@ test("state.lastAction is null when the host pushed nothing", () => { assert.equal(lastActionSeenByASpec(null), null); }); +// state.logs is the same kind of hole. Console output reaches the runner over +// CDP, so the page cannot read it back, and while the web runtime hardcoded [] +// there the default noLogcatErrors counted an empty array on every web run: a +// page whose console was full of errors went green having checked nothing. +function logsSeenByASpec(pushed: unknown): unknown { + const setLogs = (globalThis as Record).__sanderlingSetLogs__ as ( + value: unknown, + ) => void; + __testing__.extractors.length = 0; + __testing__.runtime.extract((state) => (state as { logs: unknown }).logs); + let out: Record = {}; + withState(() => { + setLogs(pushed); + out = __testing__.evaluateExtractors(); + }); + return readingOf(out, 0); +} + +test("state.logs carries the entries the host pushed", () => { + const entries = [ + { unixMillis: 1700000000123, level: "E", tag: "console", message: "boom from the page" }, + ]; + assert.deepEqual(logsSeenByASpec(entries), entries); +}); + +test("state.logs is empty when the host pushed no entries", () => { + assert.deepEqual(logsSeenByASpec([]), []); +}); + // sanitize runs over every extractor's return value before it leaves the // runtime. A user extractor that returns a page object reachable from // document/window can be self-referential, carry functions, or nest deeply; @@ -727,29 +806,23 @@ test("selectorTag renders the selector shapes the goja host renders", () => { // accounts/totalBalance extractors (findAll([{HomeScreen}, {AccountCard}])) // were empty on every web step and the properties over them checked nothing. test("ax.findAll resolves a selector path segment by segment", () => { - const rect = { left: 0, top: 0, right: 10, bottom: 10, width: 10, height: 10 }; - const node = (id: string, answers: Record = {}) => ({ - id, - tagName: "DIV", - className: "", - textContent: id, - dataset: {}, - getAttribute: () => null, - matches: (selector: string) => matchesAnyPart(selector, "div", {}), - getBoundingClientRect: () => rect, - querySelectorAll: (selector: string) => answers[selector] ?? [], + const card = (id: string, y: number): FakeElementSpec => ({ + tag: "div", x: 0, y, width: 10, height: 10, testid: "AccountCard", text: id, + }); + // The stray card is outside HomeScreen, so a document-wide sweep for the + // second segment picks it up and the scoping assertion below fails. + const page = fakeElement({ + tag: "div", x: 0, y: 0, width: 100, height: 100, + children: [ + { + tag: "div", x: 0, y: 0, width: 100, height: 50, testid: "HomeScreen", + children: [card("first", 0), card("second", 10)], + }, + card("stray", 60), + ], }); - const cardCss = `:is([data-testid="AccountCard"], [id="AccountCard"])`; - const screenCss = `:is([data-testid="HomeScreen"], [id="HomeScreen"])`; - const cards = [node("first"), node("second")]; - const home = node("HomeScreen", { [cardCss]: cards }); - const g = globalThis as Record; - const originalDocument = g.document; - const originalWindow = g.window; - g.document = { querySelectorAll: (selector: string) => (selector === screenCss ? [home] : []) }; - g.window = {}; - try { + withFakeDocument([page], () => { __testing__.extractors.length = 0; __testing__.runtime.extract((state) => { const ax = (state as { ax: { findAll(s: unknown): Record[] } }).ax; @@ -770,10 +843,7 @@ test("ax.findAll resolves a selector path segment by segment", () => { // Both cards answer to the same path, so neither may carry it: the runner // re-resolves a named target and would send both taps to the first card. assert.deepEqual(readingOf(values, 1), ["", ""]); - } finally { - g.document = originalDocument; - g.window = originalWindow; - } + }); }); // A selector is a name only while ONE element answers to it. The runner prefers @@ -782,33 +852,19 @@ test("ax.findAll resolves a selector path segment by segment", () => { // testTag sends every one of their taps to the first sibling: on folio's Home // screen no account but the first could ever be opened. test("ax.find and ax.findAll label the element with the selector only when it names that element alone", () => { - const rect = (top: number) => ({ left: 0, top, right: 10, bottom: top + 10, width: 10, height: 10 }); - const node = (id: string, top: number) => ({ - id, - tagName: "DIV", - className: "", - textContent: id, - dataset: {}, - getAttribute: () => null, - matches: (selector: string) => matchesAnyPart(selector, "div", {}), - getBoundingClientRect: () => rect(top), + const sibling = (text: string, y: number): FakeElementSpec => ({ + tag: "div", x: 0, y, width: 10, height: 10, testid: "AccountCard", text, }); - const submit = node("TxnSubmit", 0); - const cards = [node("Alpha", 20), node("Beta", 40), node("Gamma", 60)]; - const matches = `:is([data-testid="TxnSubmit"], [id="TxnSubmit"])`; - const cardMatches = `:is([data-testid="AccountCard"], [id="AccountCard"])`; - const g = globalThis as Record; - const originalDocument = g.document; - const originalWindow = g.window; - g.document = { - querySelectorAll: (selector: string) => { - if (selector === matches) return [submit]; - if (selector === cardMatches) return cards; - return []; - }, - }; - g.window = {}; - try { + const page = fakeElement({ + tag: "div", x: 0, y: 0, width: 100, height: 100, + children: [ + { tag: "div", x: 0, y: 0, width: 10, height: 10, id: "TxnSubmit", text: "Submit" }, + sibling("Alpha", 20), + sibling("Beta", 40), + sibling("Gamma", 60), + ], + }); + withFakeDocument([page], () => { __testing__.extractors.length = 0; __testing__.runtime.extract((state) => { const ax = (state as { ax: { find(s: unknown): Record | undefined } }).ax; @@ -838,10 +894,7 @@ test("ax.find and ax.findAll label the element with the selector only when it na siblings.map((card) => card.y), [25, 45, 65], ); - } finally { - g.document = originalDocument; - g.window = originalWindow; - } + }); }); // The same rule for a child lookup, which is the shape a spec reaches a row @@ -849,38 +902,15 @@ test("ax.find and ax.findAll label the element with the selector only when it na // the whole dump, not the parent's subtree, so scoping does not make a shared // name safe. test("element.find and element.findAll label a child only when the selector names it alone", () => { - const rect = { left: 0, top: 0, right: 10, bottom: 10, width: 10, height: 10 }; - const node = (id: string, answers: Record = {}) => ({ - id, - tagName: "DIV", - className: "", - textContent: id, - dataset: {}, - getAttribute: () => null, - matches: (selector: string) => matchesAnyPart(selector, "div", {}), - getBoundingClientRect: () => rect, - querySelectorAll: (selector: string) => answers[selector] ?? [], + const home = fakeElement({ + tag: "div", x: 0, y: 0, width: 100, height: 100, testid: "HomeScreen", + children: [ + { tag: "div", x: 0, y: 0, width: 10, height: 10, testid: "AccountCard", text: "first" }, + { tag: "div", x: 0, y: 20, width: 10, height: 10, testid: "AccountCard", text: "second" }, + { tag: "div", x: 0, y: 40, width: 10, height: 10, testid: "Total", text: "Total" }, + ], }); - const cardCss = `:is([data-testid="AccountCard"], [id="AccountCard"])`; - const totalCss = `:is([data-testid="Total"], [id="Total"])`; - const screenCss = `:is([data-testid="HomeScreen"], [id="HomeScreen"])`; - const cards = [node("first"), node("second")]; - const total = node("Total"); - const home = node("HomeScreen", { [cardCss]: cards, [totalCss]: [total] }); - - const g = globalThis as Record; - const originalDocument = g.document; - const originalWindow = g.window; - g.document = { - querySelectorAll: (selector: string) => { - if (selector === screenCss) return [home]; - if (selector === cardCss) return cards; - if (selector === totalCss) return [total]; - return []; - }, - }; - g.window = {}; - try { + withFakeDocument([home], () => { __testing__.extractors.length = 0; __testing__.runtime.extract((state) => { const ax = (state as { @@ -900,8 +930,63 @@ test("element.find and element.findAll label a child only when the selector name const values = __testing__.evaluateExtractors(); assert.deepEqual(readingOf(values, 0), ["", ""]); assert.equal(readingOf(values, 1), "testTag:Total"); - } finally { - g.document = originalDocument; - g.window = originalWindow; - } + }); +}); + +// One page, one selector, two hosts. The goja host resolves a selector against +// the hierarchy dump, whose buildTree (internal/driver/chrome/driver.go) emits +// a host's shadow children BEFORE its light ones, so a pre-order search there +// reaches a shadow-hosted match first. deepQueryAll swept the whole light DOM +// first and only then descended, so this page answered find({id:"x"}) with the +// light node in V8 and the shadow node in goja, and on web V8's answer is the +// one that reaches the properties. +test("ax.find resolves the shadow-hosted match the hierarchy dump reaches first", () => { + const page = fakeElement({ + tag: "div", x: 0, y: 0, width: 400, height: 800, id: "page", + children: [ + { + tag: "div", x: 0, y: 0, width: 400, height: 100, id: "mount", + shadow: [{ tag: "span", x: 0, y: 0, width: 40, height: 20, id: "x", text: "shadow" }], + }, + { tag: "span", x: 0, y: 100, width: 40, height: 20, id: "x", text: "light" }, + ], + }); + withFakeDocument([page], () => { + __testing__.extractors.length = 0; + __testing__.runtime.extract((state) => { + const ax = (state as { ax: { find(s: unknown): Record | undefined } }).ax; + return ax.find("id:x")?.text; + }); + __testing__.runtime.extract((state) => { + const ax = (state as { ax: { findAll(s: unknown): Record[] } }).ax; + return ax.findAll("id:x").map((element) => element.text); + }); + const values = __testing__.evaluateExtractors(); + assert.equal(readingOf(values, 0), "shadow"); + assert.deepEqual(readingOf(values, 1), ["shadow", "light"]); + }); +}); + +// A nested undefined is the one reading shape the two hosts do NOT encode +// alike, and this pins the split instead of hiding it. JSON has no undefined, +// so the key goes with the value here; goja marshals the same member as null, +// and it cannot do otherwise, because an exported goja object reports undefined +// and null identically, so dropping those keys there would drop the genuine +// nulls this host keeps. Carrying the member across would take a wire format +// that can express undefined. +// +// What both hosts DO agree on is the member's value: reading it answers +// undefined either way, and that is the guarantee a property may rely on. Key +// presence (`in`, Object.keys) is not. +// TestExtractorEncoding_NestedUndefinedIsNotOnTheWire in +// internal/verifier/extractor_encoding_test.go pins the other half. +test("a nested undefined leaves the page as a dropped key, a nested null does not", () => { + __testing__.extractors.length = 0; + __testing__.runtime.extract(() => ({ absent: undefined, empty: null, present: 1 })); + let wire = ""; + withState(() => { + // Exactly what extractorScript in internal/driver/chrome/driver.go sends. + wire = JSON.stringify(__testing__.evaluateExtractors()); + }); + assert.equal(wire, `{"0":{"value":{"empty":null,"present":1}}}`); }); diff --git a/replay-ui/sanderling/spec.ts b/replay-ui/sanderling/spec.ts index 0044eaa..75d19d6 100644 --- a/replay-ui/sanderling/spec.ts +++ b/replay-ui/sanderling/spec.ts @@ -170,11 +170,44 @@ const switchATab = actions(() => { return tabs.length === 0 ? [] : [Tap({ on: from(tabs).generate() })]; }); +// badgeCountMatchesThePanel needs two readings on ONE step: the badge, which a +// tab strip renders only for a step that HAS a violation, and a violations +// panel to compare it against, which exists while the properties or violations +// tab is selected. Undirected actions put both on the same step 0 times in the +// 80 of the first dogfood run: the property was reachable in principle and +// judged nothing in practice. +// +// Both halves have to be aimed at. Aiming at the step alone just moved the +// misses to the other side, 0 judged either way. So this selects a step the +// list marks as violating, and once standing on one, opens a panel if none is +// up. It opens the AFTER panel's, because the before panel's screenshot is what +// screenshotShowsTheSelectedStep reads and covering that up trades one +// property's evidence for another's. +const violatingRows = extract("violatingRows", (s) => + s.ax.findAll({ "data-testid": "step-row" }).filter((row) => dataOf(row, "violations") === "true"), +); +const afterPropertiesTabs = extract("afterPropertiesTabs", (s) => + s.ax + .findAll([{ "data-testid": "state-after" }, { "data-testid": "tab" }]) + .filter((tab) => dataOf(tab, "tabId") === "properties"), +); + +const showAViolatingStepWithItsPanel = actions(() => { + const rows = violatingRows.current; + if (!rows.some((row) => dataOf(row, "active") === "true")) { + return rows.length === 0 ? [] : [Tap({ on: from(rows).generate() })]; + } + if (violationPanelCounts.current.length > 0) return []; + const tabs = afterPropertiesTabs.current; + return tabs.length === 0 ? [] : [Tap({ on: from(tabs).generate() })]; +}); + // defaultActions carries the rest: the jump-to-violation button, the theme // toggle, the link back to the run list, and the scrolling. export const actionsRoot = weighted( [30, selectAStep], [20, navigateByKeyboard], [25, switchATab], + [20, showAViolatingStepWithItsPanel], [25, defaultActions], ); diff --git a/replay-ui/src/components/Tabs.css b/replay-ui/src/components/Tabs.css index fea0f76..744d9b9 100644 --- a/replay-ui/src/components/Tabs.css +++ b/replay-ui/src/components/Tabs.css @@ -6,8 +6,15 @@ font-family: var(--font-mono); } +/* Wrapping is what keeps the last tabs reachable. In a narrow column the five + tabs are wider than the panel, and the overflow scrolls .detail-panel-body, + which carries that tab's own panel out of the column with it. In a 756px + viewport (what a headless run gets) Properties and Violations then sit under + the neighbouring panel, where neither a person nor the fuzzer can click + them. */ .tabs-header { display: flex; + flex-wrap: wrap; gap: 0; border-bottom: 1px solid var(--border); flex: 0 0 auto; diff --git a/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt b/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt index bc39fe3..6a6c535 100644 --- a/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt +++ b/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverBackend.kt @@ -30,11 +30,21 @@ interface DriverBackend { fun healthy(): Boolean fun metrics(bundleId: String): MetricsSample + // snapshotTree is the tree a snapshot reads, without the screenshot. It is + // what the Hierarchy RPC serves, so the runner's two reads of a step come + // off one pipeline: a backend that waits out a transition or closes a + // keyboard before reading has to do the same on both, or the two trees + // differ over what the backend did between them rather than over what the + // app did. Measured on an API 34 emulator, an IME standing open is a + // 489-node bare read against the snapshot's 134. + fun snapshotTree(): String = hierarchy() + // snapshot captures hierarchy then screenshot back-to-back. The service // layer holds a mutex around the call so concurrent callers observe a // serialized pair from the same on-device frame. Backends may override // to fuse the two reads more tightly when their native API allows. - fun snapshot(): SnapshotSample = SnapshotSample(hierarchy(), screenshot()) + fun snapshot(): SnapshotSample = + SnapshotSample(snapshotTree(), screenshot()) // close releases device-side resources on shutdown. The iOS backend must // stop its XCTest runner here: an orphaned runner session auto-restarts @@ -185,7 +195,8 @@ private val ROUTE_TAG_KEYS = setOf( "accessibilityIdentifier", ) -private val jsonMapper = com.fasterxml.jackson.module.kotlin.jacksonObjectMapper() +private val jsonMapper = + com.fasterxml.jackson.module.kotlin.jacksonObjectMapper() // countRouteScreens counts DISTINCT route-level destination tags, not the nodes // carrying them. A screen that nests a node repeating its own route id puts two @@ -328,11 +339,12 @@ internal fun readLogcat( arguments.add(since) } return try { - val process = ProcessBuilder( - adbCmd(serial) + arguments, - ).redirectErrorStream(false).start() - val output = process.inputStream.bufferedReader().readText() - process.waitFor() + val command = adbCmd(serial) + arguments + val output = readProcessOutput( + ProcessBuilder(command).redirectErrorStream(false).start(), + ADB_OUTPUT_TIMEOUT_MILLIS, + describe = { command.joinToString(" ") }, + ) StubDriverBackend.parseLogcatOutput(output) } catch (cause: Exception) { println("adb logcat failed: $cause") @@ -358,13 +370,76 @@ internal fun readProcMetrics(serial: String?, bundleId: String): MetricsSample { private fun adbCmd(serial: String?): List = if (serial == null) listOf("adb") else listOf("adb", "-s", serial) +// ADB_OUTPUT_TIMEOUT_MILLIS bounds the diagnostic adb reads: dumpsys, logcat, +// `settings get`, /proc stats. None of them is the driver's data path, so the +// bound wants to be generous enough that it cannot fire on a link that works, +// and it is: a hierarchy fetch, far heavier than any of these, measures at a +// 76ms median and a 168ms p90 over the same remote adb link. What it caps is +// the other end, where a wedged adb once held a step ~100s. +internal const val ADB_OUTPUT_TIMEOUT_MILLIS = 10_000L + +private val adbReaders: java.util.concurrent.ExecutorService = + java.util.concurrent.Executors.newCachedThreadPool { runnable -> + Thread(runnable, "adb-output").apply { isDaemon = true } + } + +// readProcessOutput returns a process's stdout, or "" when it does not arrive +// inside timeoutMillis. +// +// The bound belongs on the READ, not on waitFor. readText ends at EOF, and a +// wedged adb neither writes nor exits, so EOF never comes and a waitFor with a +// timeout after it is a line that never runs. Waiting first and reading after +// is worse still: a process with more to say than a pipe buffer holds, which +// logcat and dumpsys both are, blocks writing while the waiter waits for it to +// finish, and neither ever moves. +// +// So the read runs on a daemon thread and killing the process is what releases +// it: destroy closes the pipe, the reader sees EOF, the thread ends. Returning +// "" hands every caller the answer it already treats as "adb said nothing", +// which is the safe direction for all of them. +internal fun readProcessOutput( + process: Process, + timeoutMillis: Long, + describe: () -> String, + log: (String) -> Unit = { System.err.println(it) }, +): String { + val reader = adbReaders.submit { + process.inputStream.bufferedReader().readText() + } + return try { + val output = reader.get( + timeoutMillis, + java.util.concurrent.TimeUnit.MILLISECONDS, + ) + if (!process.waitFor( + timeoutMillis, + java.util.concurrent.TimeUnit.MILLISECONDS, + ) + ) { + process.destroyForcibly() + } + output + } catch (cause: java.util.concurrent.TimeoutException) { + process.destroyForcibly() + reader.cancel(true) + log( + "warn: ${describe()} gave nothing in ${timeoutMillis}ms; " + + "killed it and read no answer", + ) + "" + } catch (cause: Exception) { + process.destroyForcibly() + "" + } +} + private fun adbOutput(serial: String?, arguments: List): String = try { - val process = ProcessBuilder( - adbCmd(serial) + arguments, - ).redirectErrorStream(false).start() - val output = process.inputStream.bufferedReader().readText() - process.waitFor() - output + val command = adbCmd(serial) + arguments + readProcessOutput( + ProcessBuilder(command).redirectErrorStream(false).start(), + ADB_OUTPUT_TIMEOUT_MILLIS, + describe = { command.joinToString(" ") }, + ) } catch (cause: Exception) { "" } @@ -473,8 +548,14 @@ class StubDriverBackend( companion object { private const val IDLE_POLL_INTERVAL_MILLIS = 50L + // A count we could not read is not a count of zero. Defaulting it to + // zero made an unreadable dumpsys mean "nothing is animating, go + // ahead", which is the one answer the caller cannot check: it breaks + // out of the settle and snapshots whatever frame is on screen. Unknown + // keeps it waiting instead, inside the deadline waitForIdle already + // holds, and it agrees with the probe's own exception path. internal fun isAnimationCountIdle(grepOutput: String): Boolean = - (grepOutput.trim().toIntOrNull() ?: 0) == 0 + grepOutput.trim().toIntOrNull() == 0 internal fun parseResolvedActivity( bundleId: String, @@ -496,8 +577,8 @@ class StubDriverBackend( when (ch) { ' ' -> sb.append("%s") - '\\', '"', '\'', '&', '|', ';', '<', '>', '(', ')', '*', '?', - '$', '`', '[', ']', '{', '}', '~', '#', + '\\', '"', '\'', '&', '|', ';', '<', '>', '(', ')', '*', + '?', '$', '`', '[', ']', '{', '}', '~', '#', -> sb.append( '\\', ).append(ch) @@ -780,6 +861,210 @@ internal fun typeChunks( return typed } +// dismissSoftKeyboard closes an open IME, and issues nothing when none is open. +// +// The keyboard is its own window over the bottom of the app, and the hierarchy +// carries only what is visible to the user, so every app node under it is +// absent from the tree the picker enumerates targets from. Typing raises it, so +// an IME left open hides a form's submit control for as long as the fuzzer +// keeps typing into that form, which is a state it cannot type its way out of. +// +// The mInputShown guard is load-bearing rather than an optimisation: BACK is +// what closes an open IME, and BACK with no IME open navigates out of the +// screen, so an unguarded dismissal would make every InputText a back press. +// +// The flag trails the BACK it answers for, by about 0.6s on API 36, and a +// second dismissal inside that window reads the stale true and back-presses an +// IME that has already gone. What keeps that unreachable is the caller: one +// dismissal per inputText, and the runner focuses the field with a tap before +// every InputText, which raises the IME again long before this probe runs. A +// caller that dismissed twice in a row, or typed without focusing first, would +// lose that margin. +// +// treeWithoutKeyboard closes a keyboard too, on the snapshot path, and does not +// cost this one its margin: it reads no flag, and the two are a waitForIdle and +// a hierarchy fetch apart, several times the window in which this one is stale. +internal fun dismissSoftKeyboard(shell: (String) -> String) { + if (!shell("dumpsys input_method").contains("mInputShown=true")) return + shell("input keyevent 4") +} + +// KEYBOARD_DISMISS_READS bounds the re-reads a snapshot spends waiting for the +// IME window to leave the tree after BACK. The window leaves over an +// animation, so the first read back can still carry it. A hierarchy read +// measures at a 76ms median and a 168ms p90 on the API 34 emulator, so with +// the interval these four reads watch most of a second: several retractions +// over, without turning a keyboard the app keeps re-raising into a wait with +// no end. +internal const val KEYBOARD_DISMISS_READS = 4 +internal const val KEYBOARD_DISMISS_INTERVAL_MILLIS = 100L + +// imePackageOf takes the package half of an input-method component id +// ("pkg/.Service"), the form both `settings get secure default_input_method` +// and dumpsys' mCurMethodId use. Anything that is not a package name reads as +// "no IME known", which disables the dismissal rather than guessing. +internal fun imePackageOf(component: String): String? = + component.trim().substringBefore('/') + .takeIf { it.isNotEmpty() && it.contains('.') } + +// treeShowsIme reports whether the keyboard window is in the tree, by the view +// ids the IME's own resources give it ("pkg:id/name"). +internal fun treeShowsIme(treeJson: String, imePackage: String): Boolean = + treeJson.contains("$imePackage:id/") + +// treeWithoutKeyboard closes a keyboard standing in the snapshot and returns a +// tree read after it has gone, or the tree it was given when none is open. +// +// It belongs here, before the read the picker chooses from, rather than after +// the tap that raised the keyboard. Two reasons. The picker only ever sees +// snapshots, so a dismissal anywhere later leaves this step choosing between +// the handful of targets an open keyboard left in the tree, which is the +// budget the fuzzer was losing. And the state it has to judge is settled here: +// the action landed a waitForIdle ago, where straight after the tap the +// keyboard is still on its way up and nothing it could read would say so yet. +// +// The tree is also a better guard than mInputShown. BACK closes an open +// keyboard and navigates when none is open, so pressing it is only safe on a +// true reading; mInputShown trails the keyboard by up to 0.6s, while a tree +// carrying the IME's own view ids is the keyboard being on screen, read a +// moment ago. Once dismissed, the re-reads confirm it went rather than pressing +// BACK again, so a keyboard the app puts straight back costs re-reads and never +// a second back press. +internal fun treeWithoutKeyboard( + tree: String, + imePackage: String?, + dismiss: () -> Unit, + reread: () -> String, + sleep: (Long) -> Unit = { Thread.sleep(it) }, +): String { + if (imePackage == null || !treeShowsIme(tree, imePackage)) return tree + dismiss() + var current = tree + repeat(KEYBOARD_DISMISS_READS) { + sleep(KEYBOARD_DISMISS_INTERVAL_MILLIS) + current = reread() + if (!treeShowsIme(current, imePackage)) return current + } + return current +} + +// SELECT_ALL_COMMAND selects the focused field's whole content with +// CTRL+A (keycodes 113 and 29) and DELETE_KEY_COMMAND then deletes the +// selection (keycode 67). Two key events, whatever the field holds. +internal const val SELECT_ALL_COMMAND = "input keycombination 113 29" +internal const val DELETE_KEY_COMMAND = "input keyevent 67" + +// DELETE_BATCH_KEYS bounds how many deletes ride in one `input keyevent` +// invocation on the fallback path. `input` takes a list of keycodes, so the +// round trip is paid per batch rather than per character: measured 2.3 ms/char +// against the 29.6 ms/char of one round trip each. +internal const val DELETE_BATCH_KEYS = 200 + +internal fun deleteKeyCommands(count: Int, batch: Int): List { + if (count <= 0) return emptyList() + val size = batch.coerceAtLeast(1) + return (0 until count).chunked(size).map { chunk -> + chunk.joinToString(" ", prefix = "input keyevent ") { "67" } + } +} + +// focusedEditableTextLength reports how much text the focused text field +// holds, or null when the tree names no focused text field. Null is "cannot +// tell", which is not the same as empty and must not be read as it. +// +// The field is found by class, not by an "editable" attribute: maestro's tree +// carries no such attribute. Class also settles the trap an open keyboard +// sets, which is that the IME contributes a focused node of its own. That node +// holds no text, so taking the first focused node would read a field still +// holding 4096 characters as empty, and empty is the answer that stops the +// erase. +internal fun focusedEditableTextLength(treeJson: String): Int? { + if (treeJson.isBlank()) return null + return try { + focusedFieldLength(jsonMapper.readTree(treeJson)) + } catch (_: Exception) { + null + } +} + +private fun focusedFieldLength( + node: com.fasterxml.jackson.databind.JsonNode, +): Int? { + val attributes = node.get("attributes") + if (attributes != null && attributes.isObject && + attributes.get("focused")?.asText() == "true" && + attributes.get("class")?.asText().orEmpty().endsWith("EditText") + ) { + return attributes.get("text")?.asText().orEmpty().length + } + val children = node.get("children") ?: return null + if (!children.isArray) return null + for (child in children) focusedFieldLength(child)?.let { return it } + return null +} + +// eraseFocusedField clears the field the runner just tapped. +// +// maestro's eraseText sends one delete per character through its own +// instrumentation, which measured 29.6 ms/char on the API 34 emulator: the +// 4096-character string the corpus types cost ~121s to clear, a fifth of a 20 +// minute budget spent on one step. Selecting the content and deleting the +// selection costs the same two key events at any length, measured 0.15s to +// 1.16s for 4096 characters across API 34, 35 and 36. +// +// A fast erase that leaves characters behind would be far worse than a slow +// one, because the next InputText appends to the residue and nothing +// downstream detects it. So the result is read back off the tree, and a field +// that is not empty, or that the tree cannot report on at all, is finished off +// per character. Those deletes ride in batches, so even that path costs one +// round trip per batch rather than the one per character this replaces. +internal fun eraseFocusedField( + characterCount: Int, + shell: (String) -> Unit, + focusedTextLength: () -> Int?, +) { + if (characterCount <= 0) return + shell(SELECT_ALL_COMMAND) + shell(DELETE_KEY_COMMAND) + if (focusedTextLength() == 0) return + for (command in deleteKeyCommands(characterCount, DELETE_BATCH_KEYS)) { + shell(command) + } +} + +// typingOwner picks what the mid-type foreground guard holds later reads +// against. A dumpsys it could read names the resumed package, and that is the +// answer. +// +// A read that failed is the interesting case, and neither obvious answer is +// right. Passing null hands typeChunks "no owner", which switches the guard off +// altogether and lets the rest of the string spray into whatever holds the +// foreground: an unreadable probe must never read as focus being fine. But +// refusing to type is worse in practice. The failure is a degraded link, which +// lasts, so every InputText in the run becomes a no-op, the budget goes on +// typing nothing, and the run ends green having tested nothing. +// +// So it falls back to the bundle the run launched, which leaves the guard armed +// against the app the keystrokes were meant for. That reference is better than +// the resumed package anyway: a foreground already stolen before typing began +// reads as its own owner, and the guard then matches it happily chunk after +// chunk. +internal fun typingOwner( + dumpsys: String, + launchedBundleId: String?, + warn: (String) -> Unit, +): String? { + parseResumedPackage(dumpsys)?.let { return it } + warn( + "warn: could not read the foreground app; guarding typing with " + + ( + launchedBundleId?.let { "the launched bundle $it" } + ?: "nothing, no launch was recorded" + ), + ) + return launchedBundleId +} + // resumedActivityPackage matches a "package/activity" component, mirroring the // Go scope guard's regex so both read the same dumpsys wording. private val resumedActivityPackage = @@ -832,6 +1117,14 @@ internal fun retryOpen( class MaestroDriverBackend(private val serial: String?) : DriverBackend { private val dadb: dadb.Dadb = buildDadb(serial) + private val imePackage: String? by lazy { + imePackageOf( + runCatching { + dadb.shell("settings get secure default_input_method").allOutput + }.getOrDefault(""), + ) + } + // A fresh AndroidDriver per open attempt. Its gRPC channel is built once in // the constructor and permanently shut down by close(), so reopening a // closed instance would reuse a dead channel; rebuild it each try instead. @@ -848,6 +1141,9 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { } } + @Volatile + private var launchedBundleId: String? = null + override fun launch( bundleId: String, clearState: Boolean, @@ -855,6 +1151,7 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { ) { if (clearState) driver.clearAppState(bundleId) driver.launchApp(bundleId, env) + launchedBundleId = bundleId } override fun terminate(bundleId: String) = driver.stopApp(bundleId) @@ -909,6 +1206,11 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { } else { driver.inputText(text) } + // A probe that fails reads as "no IME open", which is the safe way to be + // wrong: it skips the dismissal rather than sending a stray BACK. + dismissSoftKeyboard { + runCatching { dadb.shell(it).allOutput }.getOrDefault("") + } } // typeShellSafe types shell-safe ASCII through adb `input text` in chunks, @@ -916,8 +1218,14 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { // started in has lost the foreground, the remaining keystrokes would spray // into whatever window stole it (the launcher search box, in practice), so // typing stops instead of leaking out of the app under test. + // + // typingOwner decides what "the app the type started in" means when the + // read that would name it fails: the launched bundle, so a link that cannot + // answer degrades the guard rather than switching it off. private fun typeShellSafe(text: String) { - val owner = foregroundPackage() + val owner = typingOwner(foregroundDumpsys(), launchedBundleId) { + System.err.println(it) + } val typed = typeChunks(chunkForInput(text, INPUT_CHUNK_CHARS), owner, { foregroundPackage() @@ -931,17 +1239,21 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { } } - // foregroundPackage returns the package of the top resumed activity, or null - // if it cannot be read. Used to detect mid-type focus escapes. - private fun foregroundPackage(): String? = parseResumedPackage( - adbOutput( - serial, - listOf("shell", "dumpsys", "activity", "activities"), - ), + private fun foregroundDumpsys(): String = adbOutput( + serial, + listOf("shell", "dumpsys", "activity", "activities"), ) - override fun eraseText(characterCount: Int) = - driver.eraseText(characterCount) + // foregroundPackage returns the package of the top resumed activity, or null + // if it cannot be read. Used to detect mid-type focus escapes. + private fun foregroundPackage(): String? = + parseResumedPackage(foregroundDumpsys()) + + override fun eraseText(characterCount: Int) = eraseFocusedField( + characterCount, + shell = { dadb.shell(it) }, + focusedTextLength = { focusedEditableTextLength(hierarchy()) }, + ) override fun swipe( fromX: Int, @@ -975,8 +1287,8 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { override fun recentLogs(sinceUnixMillis: Long, minLevel: String) = readLogcat(serial, sinceUnixMillis, minLevel) - // snapshot waits out a NavHost cross-fade before it reads, so the runner is - // never handed a tree holding two routes at once. It belongs here rather + // snapshotTree waits out a NavHost cross-fade before it reads, so the runner + // is never handed a tree holding two routes at once. It belongs here rather // than in waitForIdle: the runner gives waitForIdle a one-second deadline // and abandons the RPC when it expires, which is not enough room for a // 700ms fade that began before the settle did, and a wait that outlives the @@ -987,9 +1299,15 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend { // The predicate costs nothing on a settled frame: the read it needs is the // read the snapshot was going to do anyway. That is what makes this // affordable, where the structural poll that used to run in waitForIdle was - // not: it fetched the hierarchy ~4 more times on every mutating step. - override fun snapshot(): SnapshotSample = - SnapshotSample(awaitSettledTree { hierarchy() }, screenshot()) + // not: it fetched the hierarchy ~4 more times on every mutating step. The + // keyboard leg costs nothing either when no IME is standing in the tree, + // which is what lets the Hierarchy RPC serve this too. + override fun snapshotTree(): String = treeWithoutKeyboard( + awaitSettledTree { hierarchy() }, + imePackage, + dismiss = { runCatching { dadb.shell("input keyevent 4") } }, + reread = { awaitSettledTree { hierarchy() } }, + ) override fun waitForIdle(durationMillis: Long) { // waitForAppToSettle blocks on the View-system animation and maestro's @@ -1041,15 +1359,75 @@ internal fun dadbTargetFor(serial: String?): DadbTarget { } } +internal data class AdbServerEndpoint(val host: String, val port: Int) + +private const val ADB_SERVER_HOST = "localhost" +private const val ADB_SERVER_PORT = 5037 + +// adbServerEndpoint reads where the adb server listens the way the adb CLI +// reads it: ADB_SERVER_SOCKET ("tcp:host:port", or "tcp:port" for a server on +// this machine) outranks the older ANDROID_ADB_SERVER_ADDRESS / +// ANDROID_ADB_SERVER_PORT pair, and unset means the loopback default. +// +// A value it cannot read throws instead of falling back to loopback. The +// fallback is the dangerous answer: emulator serials are numbered per server, +// so a run aimed at a remote emulator-5554 would quietly drive whatever this +// machine calls emulator-5554 and report the results as the remote device's. +internal fun adbServerEndpoint( + env: (String) -> String? = System::getenv, +): AdbServerEndpoint { + val socket = env("ADB_SERVER_SOCKET")?.trim().orEmpty() + if (socket.isNotEmpty()) return parseAdbServerSocket(socket) + val host = env("ANDROID_ADB_SERVER_ADDRESS")?.trim().orEmpty() + val port = env("ANDROID_ADB_SERVER_PORT")?.trim().orEmpty() + return AdbServerEndpoint( + host.ifEmpty { ADB_SERVER_HOST }, + if (port.isEmpty()) { + ADB_SERVER_PORT + } else { + adbServerPort(port, "ANDROID_ADB_SERVER_PORT=\"$port\"") + }, + ) +} + +private fun parseAdbServerSocket(value: String): AdbServerEndpoint { + val named = "ADB_SERVER_SOCKET=\"$value\"" + val address = value.removePrefix("tcp:") + if (address == value) rejectAdbServerSocket(named) + val colon = address.lastIndexOf(':') + if (colon < 0) { + return AdbServerEndpoint( + ADB_SERVER_HOST, + adbServerPort(address, named), + ) + } + val host = address.substring(0, colon) + if (host.isEmpty()) rejectAdbServerSocket(named) + return AdbServerEndpoint( + host, + adbServerPort(address.substring(colon + 1), named), + ) +} + +private fun rejectAdbServerSocket(named: String): Nothing = + throw IllegalArgumentException("$named is not tcp:host:port") + +private fun adbServerPort(text: String, named: String): Int = + text.toIntOrNull()?.takeIf { it in 1..65535 } + ?: throw IllegalArgumentException("$named has no usable port") + private fun buildDadb(serial: String?): dadb.Dadb = when (val target = dadbTargetFor(serial)) { is DadbTarget.Tcp -> dadb.Dadb.create(target.host, target.port) - is DadbTarget.Server -> dadb.adbserver.AdbServer.createDadb( - "localhost", - 5037, - "host:transport:${target.serial}", - ) + is DadbTarget.Server -> { + val server = adbServerEndpoint() + dadb.adbserver.AdbServer.createDadb( + server.host, + server.port, + "host:transport:${target.serial}", + ) + } } // requireBoundsBySelector refuses a selector that names nothing on the current @@ -1141,7 +1519,8 @@ internal fun pngHeight(bytes: ByteArray): Int { (bytes[22].toInt() and 0xFF shl 8) or (bytes[23].toInt() and 0xFF) } -internal const val IOS_XCTEST_RUNNER_BUNDLE_ID = "dev.mobile.maestro-driver-iosUITests.xctrunner" +internal const val IOS_XCTEST_RUNNER_BUNDLE_ID = + "dev.mobile.maestro-driver-iosUITests.xctrunner" // reapOrphanIosRunners kills XCTest runner sessions left over from a prior // run. A sidecar that died without its shutdown hook leaves its xcodebuild diff --git a/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverService.kt b/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverService.kt index cfe11d0..ea3b137 100644 --- a/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverService.kt +++ b/sidecar/src/main/kotlin/dev/sanderling/sidecar/DriverService.kt @@ -31,7 +31,10 @@ class DriverService( private val launchedBundleId = AtomicReference(null) private val snapshotLock = Any() - override fun launch(request: LaunchRequest, responseObserver: StreamObserver) { + override fun launch( + request: LaunchRequest, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.launch(request.bundleId, request.clearState, request.envMap) launchedBundleId.set(request.bundleId) @@ -39,7 +42,10 @@ class DriverService( } } - override fun terminate(request: Empty, responseObserver: StreamObserver) { + override fun terminate( + request: Empty, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { launchedBundleId.get()?.let { backend.terminate(it) } launchedBundleId.set(null) @@ -54,42 +60,60 @@ class DriverService( } } - override fun doubleTap(request: Point, responseObserver: StreamObserver) { + override fun doubleTap( + request: Point, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.doubleTap(request.x, request.y) Empty.getDefaultInstance() } } - override fun longPress(request: Point, responseObserver: StreamObserver) { + override fun longPress( + request: Point, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.longPress(request.x, request.y) Empty.getDefaultInstance() } } - override fun tapSelector(request: Selector, responseObserver: StreamObserver) { + override fun tapSelector( + request: Selector, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.tapSelector(request.value) Empty.getDefaultInstance() } } - override fun inputText(request: Text, responseObserver: StreamObserver) { + override fun inputText( + request: Text, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.inputText(request.value) Empty.getDefaultInstance() } } - override fun eraseText(request: EraseTextRequest, responseObserver: StreamObserver) { + override fun eraseText( + request: EraseTextRequest, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.eraseText(request.characterCount) Empty.getDefaultInstance() } } - override fun swipe(request: SwipeRequest, responseObserver: StreamObserver) { + override fun swipe( + request: SwipeRequest, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { val from = request.from val to = request.to @@ -98,16 +122,25 @@ class DriverService( } } - override fun pressKey(request: PressKeyRequest, responseObserver: StreamObserver) { + override fun pressKey( + request: PressKeyRequest, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.pressKey(request.key) Empty.getDefaultInstance() } } - override fun recentLogs(request: RecentLogsRequest, responseObserver: StreamObserver) { + override fun recentLogs( + request: RecentLogsRequest, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { - val entries = backend.recentLogs(request.sinceUnixMillis, request.levelAtLeast) + val entries = backend.recentLogs( + request.sinceUnixMillis, + request.levelAtLeast, + ) val builder = LogEntries.newBuilder() for (entry in entries) { builder.addEntries( @@ -123,7 +156,10 @@ class DriverService( } } - override fun screenshot(request: Empty, responseObserver: StreamObserver) { + override fun screenshot( + request: Empty, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { val (png, width, height) = backend.screenshot() Image.newBuilder() @@ -134,18 +170,34 @@ class DriverService( } } - override fun hierarchy(request: Empty, responseObserver: StreamObserver) { + // The runner reads this a second time per step to see whether the screen + // changed while it was looking, so it has to describe the same thing the + // snapshot's tree describes: same settle, same keyboard handling, same + // lock. Served off the bare backend read, the pair differed over what the + // backend did between them rather than over what the app did. + override fun hierarchy( + request: Empty, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { - HierarchyJSON.newBuilder().setJson(backend.hierarchy()).build() + val tree = synchronized(snapshotLock) { backend.snapshotTree() } + HierarchyJSON.newBuilder().setJson(tree).build() } } - override fun snapshot(request: Empty, responseObserver: StreamObserver) { + override fun snapshot( + request: Empty, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { val sample = synchronized(snapshotLock) { backend.snapshot() } val (png, width, height) = sample.screenshot SnapshotResponse.newBuilder() - .setHierarchy(HierarchyJSON.newBuilder().setJson(sample.hierarchyJson).build()) + .setHierarchy( + HierarchyJSON.newBuilder() + .setJson(sample.hierarchyJson) + .build(), + ) .setScreenshot( Image.newBuilder() .setPng(ByteString.copyFrom(png)) @@ -157,14 +209,20 @@ class DriverService( } } - override fun waitForIdle(request: Duration, responseObserver: StreamObserver) { + override fun waitForIdle( + request: Duration, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { backend.waitForIdle(request.millis) Empty.getDefaultInstance() } } - override fun health(request: Empty, responseObserver: StreamObserver) { + override fun health( + request: Empty, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { HealthStatus.newBuilder() .setReady(backend.healthy()) @@ -174,9 +232,16 @@ class DriverService( } } - override fun metrics(request: MetricsRequest, responseObserver: StreamObserver) { + override fun metrics( + request: MetricsRequest, + responseObserver: StreamObserver, + ) { runRpc(responseObserver) { - val bundleId = if (request.bundleId.isNotEmpty()) request.bundleId else launchedBundleId.get().orEmpty() + val bundleId = if (request.bundleId.isNotEmpty()) { + request.bundleId + } else { + launchedBundleId.get().orEmpty() + } val sample = backend.metrics(bundleId) MetricsResponse.newBuilder() .setCpuPercent(sample.cpuPercent) @@ -191,7 +256,9 @@ class DriverService( // stale session, then closes the backend so the iOS XCTest runner process // dies with us instead of being orphaned. fun shutdown() { - runCatching { launchedBundleId.getAndSet(null)?.let { backend.terminate(it) } } + runCatching { + launchedBundleId.getAndSet(null)?.let { backend.terminate(it) } + } runCatching { backend.close() } } @@ -209,8 +276,10 @@ class DriverService( // failures that do not extend Exception, and an uncaught one // kills the RPC as a channel-level Unknown instead of a status // the runner can classify. - observer.onError(io.grpc.Status.INTERNAL.withDescription(cause.toString()) - .withCause(cause).asRuntimeException()) + observer.onError( + io.grpc.Status.INTERNAL.withDescription(cause.toString()) + .withCause(cause).asRuntimeException(), + ) } } diff --git a/sidecar/src/main/kotlin/dev/sanderling/sidecar/Main.kt b/sidecar/src/main/kotlin/dev/sanderling/sidecar/Main.kt index 39f9a5e..5692d29 100644 --- a/sidecar/src/main/kotlin/dev/sanderling/sidecar/Main.kt +++ b/sidecar/src/main/kotlin/dev/sanderling/sidecar/Main.kt @@ -13,14 +13,17 @@ class SidecarServer( private val shutdownLatch = CountDownLatch(1) fun start(): Int { - val server = NettyServerBuilder.forAddress(InetSocketAddress("127.0.0.1", port)) + val server = NettyServerBuilder + .forAddress(InetSocketAddress("127.0.0.1", port)) .addService(service) .build() server.start() grpcServer = server - Runtime.getRuntime().addShutdownHook(Thread { - stop() - }) + Runtime.getRuntime().addShutdownHook( + Thread { + stop() + }, + ) return server.port } @@ -48,23 +51,41 @@ class SidecarServer( // lost from run output. private fun quietExpectedDriverNoise() { org.apache.logging.log4j.core.config.Configurator.setLevel( - "util.CommandLineUtils", org.apache.logging.log4j.Level.OFF) + "util.CommandLineUtils", + org.apache.logging.log4j.Level.OFF, + ) org.apache.logging.log4j.core.config.Configurator.setLevel( - "xcuitest.XCTestDriverClient", org.apache.logging.log4j.Level.OFF) + "xcuitest.XCTestDriverClient", + org.apache.logging.log4j.Level.OFF, + ) org.apache.logging.log4j.core.config.Configurator.setLevel( - "maestro.drivers.AndroidDriver", org.apache.logging.log4j.Level.OFF) + "maestro.drivers.AndroidDriver", + org.apache.logging.log4j.Level.OFF, + ) } fun main(arguments: Array) { quietExpectedDriverNoise() val port = arguments.indexOf("--port").let { index -> - if (index >= 0 && index + 1 < arguments.size) arguments[index + 1].toInt() else 0 + if (index >= 0 && index + 1 < arguments.size) { + arguments[index + 1].toInt() + } else { + 0 + } } val platform = arguments.indexOf("--platform").let { index -> - if (index >= 0 && index + 1 < arguments.size) arguments[index + 1] else "android" + if (index >= 0 && index + 1 < arguments.size) { + arguments[index + 1] + } else { + "android" + } } val serial = arguments.indexOf("--serial").let { index -> - if (index >= 0 && index + 1 < arguments.size) arguments[index + 1] else null + if (index >= 0 && index + 1 < arguments.size) { + arguments[index + 1] + } else { + null + } } val backend: DriverBackend = when (platform) { @@ -74,7 +95,9 @@ fun main(arguments: Array) { val service = DriverService(platform = platform, backend = backend) val server = SidecarServer(port, service) val boundPort = server.start() - println("sanderling-sidecar listening on 127.0.0.1:$boundPort platform=$platform") + println( + "sanderling-sidecar listening on 127.0.0.1:$boundPort platform=$platform", + ) System.out.flush() server.awaitTermination() } diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/AdbOutputTimeoutTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/AdbOutputTimeoutTest.kt new file mode 100644 index 0000000..6a741b9 --- /dev/null +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/AdbOutputTimeoutTest.kt @@ -0,0 +1,120 @@ +package dev.sanderling.sidecar + +import org.junit.Test +import java.io.InputStream +import java.io.OutputStream +import java.util.concurrent.CountDownLatch +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +class AdbOutputTimeoutTest { + + // An adb wedged on the link neither writes nor exits, so the read never + // reaches EOF. Without a bound on the READ the step waits for as long as + // adb feels like it: one such stall measured ~100s against a remote adb + // server. The bound has to release the reader as well as return, which is + // what killing the process does. + @Test(timeout = 20_000L) + fun aWedgedReadIsAbandonedAtTheBoundInsteadOfHangingForever() { + val process = FakeProcess(BlockingStream()) + val logged = mutableListOf() + + val started = System.currentTimeMillis() + val output = readProcessOutput(process, 200L, { "adb shell pidof" }) { + logged.add(it) + } + val elapsed = System.currentTimeMillis() - started + + assertEquals("", output, "a read that never lands is no answer") + assertTrue(process.destroyed, "the wedged adb must be killed, not left") + assertTrue( + elapsed < 10_000L, + "returned in ${elapsed}ms, not at a bound", + ) + assertEquals(1, logged.size, "a silent timeout hides a degrading link") + } + + @Test(timeout = 20_000L) + fun theAbandonedReadNamesTheCommandAndTheBound() { + val logged = mutableListOf() + readProcessOutput( + FakeProcess(BlockingStream()), + 200L, + { "adb -s emulator-5556 shell cat /proc/6103/stat" }, + ) { logged.add(it) } + + val line = logged.single() + assertTrue( + line.contains("adb -s emulator-5556 shell cat /proc/6103/stat"), + line, + ) + assertTrue(line.contains("200"), line) + } + + // The bound must cost the healthy path nothing: output that arrives comes + // back whole, and the process is left to exit on its own. + @Test(timeout = 20_000L) + fun outputThatArrivesComesBackWholeAndTheProcessSurvives() { + val text = "VmRSS:\t 123456 kB\nVmSize:\t 654321 kB\n" + val process = FakeProcess(text.byteInputStream()) + val logged = mutableListOf() + + val output = readProcessOutput(process, 5_000L, { "adb shell cat" }) { + logged.add(it) + } + + assertEquals(text, output) + assertTrue(!process.destroyed, "a process that answered is not killed") + assertTrue(logged.isEmpty(), "nothing to report on the healthy path") + } + + // Output larger than a pipe buffer is why the read cannot be deferred + // until after the process exits: a process with more to say than the + // buffer holds blocks writing while a waiter waits for it to finish. + @Test(timeout = 20_000L) + fun outputLargerThanAPipeBufferComesBackWhole() { + val text = "x".repeat(512 * 1024) + val output = readProcessOutput( + FakeProcess(text.byteInputStream()), + 5_000L, + { "adb logcat -d" }, + ) {} + assertEquals(text.length, output.length) + } +} + +// BlockingStream models a wedged adb: no bytes, and no EOF either, until the +// process is killed and the pipe closes under the reader. +private class BlockingStream : InputStream() { + private val released = CountDownLatch(1) + + override fun read(): Int { + released.await() + return -1 + } + + override fun close() { + released.countDown() + } +} + +private class FakeProcess(private val stream: InputStream) : Process() { + @Volatile var destroyed = false + private set + + override fun getOutputStream(): OutputStream = + OutputStream.nullOutputStream() + + override fun getInputStream(): InputStream = stream + + override fun getErrorStream(): InputStream = InputStream.nullInputStream() + + override fun waitFor(): Int = 0 + + override fun exitValue(): Int = 0 + + override fun destroy() { + destroyed = true + stream.close() + } +} diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/DadbTargetTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/DadbTargetTest.kt index 7db71f8..c51310c 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/DadbTargetTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/DadbTargetTest.kt @@ -2,6 +2,8 @@ package dev.sanderling.sidecar import org.junit.Test import kotlin.test.assertEquals +import kotlin.test.assertFailsWith +import kotlin.test.assertTrue class DadbTargetTest { @@ -10,7 +12,10 @@ class DadbTargetTest { } @Test fun hostPortSerialConnectsDirectly() { - assertEquals(DadbTarget.Tcp("192.168.1.243", 5555), dadbTargetFor("192.168.1.243:5555")) + assertEquals( + DadbTarget.Tcp("192.168.1.243", 5555), + dadbTargetFor("192.168.1.243:5555"), + ) } @Test fun usbSerialRoutesThroughAdbServer() { @@ -20,6 +25,108 @@ class DadbTargetTest { // A colon with a non-numeric port is a USB serial that merely contains a // colon, not a host:port, so it must route through the adb server. @Test fun colonWithNonNumericPortIsAServerSerial() { - assertEquals(DadbTarget.Server("emulator:5554x"), dadbTargetFor("emulator:5554x")) + assertEquals( + DadbTarget.Server("emulator:5554x"), + dadbTargetFor("emulator:5554x"), + ) + } + + // A serial-addressed device is reached through whichever adb server the + // environment names. Ignoring it sends the run to this machine's own + // server, where the serial either is missing or, worse, names a different + // device that happens to share the emulator numbering. + @Test fun adbServerSocketNamesARemoteServer() { + assertEquals( + AdbServerEndpoint("100.68.126.75", 5037), + adbServerEndpoint( + env("ADB_SERVER_SOCKET" to "tcp:100.68.126.75:5037"), + ), + ) + } + + @Test fun adbServerSocketWithOnlyAPortStaysLocal() { + assertEquals( + AdbServerEndpoint("localhost", 5038), + adbServerEndpoint(env("ADB_SERVER_SOCKET" to "tcp:5038")), + ) + } + + @Test fun androidAdbServerAddressAndPortPairIsHonoured() { + assertEquals( + AdbServerEndpoint("10.0.0.4", 5040), + adbServerEndpoint( + env( + "ANDROID_ADB_SERVER_ADDRESS" to "10.0.0.4", + "ANDROID_ADB_SERVER_PORT" to "5040", + ), + ), + ) + } + + @Test fun adbServerSocketOutranksTheOlderPair() { + assertEquals( + AdbServerEndpoint("100.68.126.75", 5037), + adbServerEndpoint( + env( + "ADB_SERVER_SOCKET" to "tcp:100.68.126.75:5037", + "ANDROID_ADB_SERVER_ADDRESS" to "10.0.0.4", + "ANDROID_ADB_SERVER_PORT" to "5040", + ), + ), + ) + } + + @Test fun unsetEnvironmentKeepsTheLoopbackDefault() { + assertEquals( + AdbServerEndpoint("localhost", 5037), + adbServerEndpoint(env()), + ) + assertEquals( + AdbServerEndpoint("localhost", 5037), + adbServerEndpoint(env("ADB_SERVER_SOCKET" to "")), + ) + } + + @Test fun eitherHalfOfTheOlderPairAloneKeepsTheOtherDefault() { + assertEquals( + AdbServerEndpoint("10.0.0.4", 5037), + adbServerEndpoint(env("ANDROID_ADB_SERVER_ADDRESS" to "10.0.0.4")), + ) + assertEquals( + AdbServerEndpoint("localhost", 5040), + adbServerEndpoint(env("ANDROID_ADB_SERVER_PORT" to "5040")), + ) + } + + // A value that cannot be read must stop the run and say which variable + // held what. Falling back to loopback would drive this machine's devices + // while the operator believes the run is on the remote ones. + @Test fun malformedValuesFailNamingTheVariableAndItsContents() { + val cases = mapOf( + "ADB_SERVER_SOCKET" to listOf( + "100.68.126.75:5037", + "tcp:100.68.126.75:pear", + "tcp:", + "tcp::5037", + "unix:/tmp/adb", + "tcp:100.68.126.75:70000", + ), + "ANDROID_ADB_SERVER_PORT" to listOf("pear", "0", "-1"), + ) + for ((variable, values) in cases) { + for (value in values) { + val failure = assertFailsWith(value) { + adbServerEndpoint(env(variable to value)) + } + val message = failure.message.orEmpty() + assertTrue(message.contains(variable), message) + assertTrue(message.contains(value), message) + } + } } } + +private fun env(vararg entries: Pair): (String) -> String? { + val values = entries.toMap() + return { values[it] } +} diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/DeviceOutputParserTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/DeviceOutputParserTest.kt index 8cec21e..994168c 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/DeviceOutputParserTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/DeviceOutputParserTest.kt @@ -25,7 +25,8 @@ class DeviceOutputParserTest { assertEquals("FATAL EXCEPTION: main", lines[0].message) val year = java.util.Calendar.getInstance().get(java.util.Calendar.YEAR) - val cal = java.util.Calendar.getInstance().apply { timeInMillis = lines[0].unixMillis } + val cal = java.util.Calendar.getInstance() + .apply { timeInMillis = lines[0].unixMillis } assertEquals(year, cal.get(java.util.Calendar.YEAR)) assertEquals(56, cal.get(java.util.Calendar.SECOND)) assertEquals(789, cal.get(java.util.Calendar.MILLISECOND)) @@ -52,7 +53,9 @@ class DeviceOutputParserTest { @Test fun parseCpuTicksReturnsNullOnTruncatedOrNonNumericStat() { assertNull(parseCpuTicks("1234 (app) S 1 2 3")) - assertNull(parseCpuTicks("1234 (app) S " + (1..12).joinToString(" ") { "x" })) + assertNull( + parseCpuTicks("1234 (app) S " + (1..12).joinToString(" ") { "x" }), + ) assertNull(parseCpuTicks("")) } @@ -80,8 +83,14 @@ class DeviceOutputParserTest { } @Test fun parseBoundsAcceptsWellFormedAndRejectsMalformed() { - assertEquals(listOf(0, 0, 1080, 2340), parseBounds("[0,0,1080,2340]")?.toList()) - assertEquals(listOf(-5, -10, 20, 30), parseBounds("[-5,-10,20,30]")?.toList()) + assertEquals( + listOf(0, 0, 1080, 2340), + parseBounds("[0,0,1080,2340]")?.toList(), + ) + assertEquals( + listOf(-5, -10, 20, 30), + parseBounds("[-5,-10,20,30]")?.toList(), + ) assertNull(parseBounds("[0,0,1080]")) assertNull(parseBounds("0,0,1,1")) assertNull(parseBounds("[0, 0, 1, 1]")) @@ -93,10 +102,14 @@ class DeviceOutputParserTest { "resource-id" to "com.example:id/loginButton", "bounds" to "[10,20,110,80]", ) - assertEquals(listOf(10, 20, 110, 80), findBoundsBySelector(tree, "id:loginButton")?.toList()) assertEquals( listOf(10, 20, 110, 80), - findBoundsBySelector(tree, "id:com.example:id/loginButton")?.toList(), + findBoundsBySelector(tree, "id:loginButton")?.toList(), + ) + assertEquals( + listOf(10, 20, 110, 80), + findBoundsBySelector(tree, "id:com.example:id/loginButton") + ?.toList(), ) } @@ -105,11 +118,20 @@ class DeviceOutputParserTest { "resource-id" to "root", children = listOf( node("text" to "Sign in", "bounds" to "[1,2,3,4]"), - node("content-desc" to "AccountCardRow-7", "bounds" to "[5,6,7,8]"), + node( + "content-desc" to "AccountCardRow-7", + "bounds" to "[5,6,7,8]", + ), ), ) - assertEquals(listOf(1, 2, 3, 4), findBoundsBySelector(tree, "text:Sign in")?.toList()) - assertEquals(listOf(5, 6, 7, 8), findBoundsBySelector(tree, "descPrefix:AccountCard")?.toList()) + assertEquals( + listOf(1, 2, 3, 4), + findBoundsBySelector(tree, "text:Sign in")?.toList(), + ) + assertEquals( + listOf(5, 6, 7, 8), + findBoundsBySelector(tree, "descPrefix:AccountCard")?.toList(), + ) } @Test fun findBoundsBySelectorMatchesIdPrefixWithoutThePackage() { @@ -132,7 +154,10 @@ class DeviceOutputParserTest { } @Test fun findBoundsBySelectorReturnsNullForBadSelectorOrNoMatch() { - val tree = node("resource-id" to "com.example:id/x", "bounds" to "[0,0,1,1]") + val tree = node( + "resource-id" to "com.example:id/x", + "bounds" to "[0,0,1,1]", + ) assertNull(findBoundsBySelector(tree, "id")) assertNull(findBoundsBySelector(tree, "id:missing")) } @@ -152,7 +177,10 @@ class DeviceOutputParserTest { } @Test fun findBoundsBySelectorReturnsNullWhenMatchHasMalformedBounds() { - val tree = node("resource-id" to "com.example:id/x", "bounds" to "not-bounds") + val tree = node( + "resource-id" to "com.example:id/x", + "bounds" to "not-bounds", + ) assertNull(findBoundsBySelector(tree, "id:x")) } @@ -166,17 +194,26 @@ class DeviceOutputParserTest { private fun ihdr(width: Int, height: Int): ByteArray { val b = ByteArray(33) for (i in 0 until 8) b[8 + i] = 0 - b[12] = 'I'.code.toByte(); b[13] = 'H'.code.toByte() - b[14] = 'D'.code.toByte(); b[15] = 'R'.code.toByte() - b[16] = (width ushr 24).toByte(); b[17] = (width ushr 16).toByte() - b[18] = (width ushr 8).toByte(); b[19] = width.toByte() - b[20] = (height ushr 24).toByte(); b[21] = (height ushr 16).toByte() - b[22] = (height ushr 8).toByte(); b[23] = height.toByte() + b[12] = 'I'.code.toByte() + b[13] = 'H'.code.toByte() + b[14] = 'D'.code.toByte() + b[15] = 'R'.code.toByte() + b[16] = (width ushr 24).toByte() + b[17] = (width ushr 16).toByte() + b[18] = (width ushr 8).toByte() + b[19] = width.toByte() + b[20] = (height ushr 24).toByte() + b[21] = (height ushr 16).toByte() + b[22] = (height ushr 8).toByte() + b[23] = height.toByte() return b } private fun node( vararg attrs: Pair, children: List = emptyList(), - ): maestro.TreeNode = maestro.TreeNode(attributes = attrs.toMap().toMutableMap(), children = children) + ): maestro.TreeNode = maestro.TreeNode( + attributes = attrs.toMap().toMutableMap(), + children = children, + ) } diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/DriverServiceTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/DriverServiceTest.kt index 65d046b..29cd6a9 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/DriverServiceTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/DriverServiceTest.kt @@ -20,20 +20,34 @@ import org.junit.Test import kotlin.test.assertEquals import kotlin.test.assertTrue -private data class Quintuple(val a: A, val b: B, val c: C, val d: D, val e: E) +private data class Quintuple( + val a: A, + val b: B, + val c: C, + val d: D, + val e: E, +) class DriverServiceTest { @get:Rule val grpcCleanup: GrpcCleanupRule = GrpcCleanupRule() - private fun newClient(backend: DriverBackend): DriverGrpc.DriverBlockingStub { + private fun newClient( + backend: DriverBackend, + ): DriverGrpc.DriverBlockingStub { val serverName = InProcessServerBuilder.generateName() val service = DriverService(platform = "android", backend = backend) grpcCleanup.register( - InProcessServerBuilder.forName(serverName).directExecutor().addService(service).build().start() + InProcessServerBuilder.forName(serverName) + .directExecutor() + .addService(service) + .build() + .start(), ) val channel: ManagedChannel = grpcCleanup.register( - InProcessChannelBuilder.forName(serverName).directExecutor().build() + InProcessChannelBuilder.forName(serverName) + .directExecutor() + .build(), ) return DriverGrpc.newBlockingStub(channel) } @@ -59,20 +73,32 @@ class DriverServiceTest { var terminated: String? = null var closed = false val backend = object : DriverBackend by StubDriverBackend("android") { - override fun terminate(bundleId: String) { terminated = bundleId } - override fun close() { closed = true } + override fun terminate(bundleId: String) { + terminated = bundleId + } + override fun close() { + closed = true + } } val serverName = InProcessServerBuilder.generateName() val service = DriverService(platform = "android", backend = backend) grpcCleanup.register( - InProcessServerBuilder.forName(serverName).directExecutor().addService(service).build().start() + InProcessServerBuilder.forName(serverName) + .directExecutor() + .addService(service) + .build() + .start(), ) val channel: ManagedChannel = grpcCleanup.register( - InProcessChannelBuilder.forName(serverName).directExecutor().build() + InProcessChannelBuilder.forName(serverName) + .directExecutor() + .build(), ) val client = DriverGrpc.newBlockingStub(channel) - client.launch(LaunchRequest.newBuilder().setBundleId("com.example").build()) + client.launch( + LaunchRequest.newBuilder().setBundleId("com.example").build(), + ) service.shutdown() assertEquals("com.example", terminated) @@ -83,8 +109,12 @@ class DriverServiceTest { var terminated: String? = null var closed = false val backend = object : DriverBackend by StubDriverBackend("android") { - override fun terminate(bundleId: String) { terminated = bundleId } - override fun close() { closed = true } + override fun terminate(bundleId: String) { + terminated = bundleId + } + override fun close() { + closed = true + } } val service = DriverService(platform = "android", backend = backend) @@ -128,17 +158,17 @@ class DriverServiceTest { // the runner can tell transient failures from fatal ones. @Test fun backendStatusCodePassesThrough() { val backend = object : DriverBackend by StubDriverBackend("android") { - override fun inputText(text: String) { + override fun inputText(text: String): Unit = throw io.grpc.Status.UNAVAILABLE .withDescription("connection dropped mid-action") .asRuntimeException() - } } val client = newClient(backend) - val thrown = kotlin.test.assertFailsWith { - client.inputText(Text.newBuilder().setValue("hello").build()) - } + val thrown = + kotlin.test.assertFailsWith { + client.inputText(Text.newBuilder().setValue("hello").build()) + } assertEquals(io.grpc.Status.Code.UNAVAILABLE, thrown.status.code) } @@ -147,17 +177,19 @@ class DriverServiceTest { // channel-level Unknown the runner cannot classify. @Test fun nonExceptionThrowableMapsToInternal() { val backend = object : DriverBackend by StubDriverBackend("android") { - override fun inputText(text: String) { + override fun inputText(text: String): Unit = throw Throwable("only one gesture can be performed at a time") - } } val client = newClient(backend) - val thrown = kotlin.test.assertFailsWith { - client.inputText(Text.newBuilder().setValue("hello").build()) - } + val thrown = + kotlin.test.assertFailsWith { + client.inputText(Text.newBuilder().setValue("hello").build()) + } assertEquals(io.grpc.Status.Code.INTERNAL, thrown.status.code) - assertTrue(thrown.status.description.orEmpty().contains("only one gesture")) + assertTrue( + thrown.status.description.orEmpty().contains("only one gesture"), + ) } @Test fun reapOrphanIosRunnersKillsStrayXcodebuildAndRunnerApp() { @@ -171,7 +203,16 @@ class DriverServiceTest { assertEquals("pkill", commands[0][0]) assertTrue(commands[0][2].contains("test-without-building")) assertTrue(commands[0][2].contains("UDID-1234")) - assertEquals(listOf("xcrun", "simctl", "terminate", "UDID-1234", IOS_XCTEST_RUNNER_BUNDLE_ID), commands[1]) + assertEquals( + listOf( + "xcrun", + "simctl", + "terminate", + "UDID-1234", + IOS_XCTEST_RUNNER_BUNDLE_ID, + ), + commands[1], + ) } @Test fun reapOrphanIosRunnersReportsNothingFound() { @@ -192,7 +233,9 @@ class DriverServiceTest { // still executing fails instead of queuing. val tapAction = { if (!inFlight.compareAndSet(false, true)) { - throw IllegalStateException("only one gesture can be performed at a time") + throw IllegalStateException( + "only one gesture can be performed at a time", + ) } invocations.incrementAndGet() Thread.sleep(150) @@ -211,7 +254,9 @@ class DriverServiceTest { val tapAction = { if (failedFirst.compareAndSet(false, true)) { Thread.sleep(60) - throw IllegalStateException("only one gesture can be performed at a time") + throw IllegalStateException( + "only one gesture can be performed at a time", + ) } landed.incrementAndGet() Unit @@ -226,21 +271,38 @@ class DriverServiceTest { // directly. val taps = mutableListOf>() val backend = object : DriverBackend { - override fun launch(bundleId: String, clearState: Boolean, env: Map) {} + override fun launch( + bundleId: String, + clearState: Boolean, + env: Map, + ) {} override fun terminate(bundleId: String) {} - override fun tap(x: Int, y: Int) { taps.add(x to y) } + override fun tap(x: Int, y: Int) { + taps.add(x to y) + } override fun tapSelector(selector: String) {} override fun inputText(text: String) {} override fun eraseText(characterCount: Int) {} - override fun swipe(fromX: Int, fromY: Int, toX: Int, toY: Int, durationMillis: Long) {} + override fun swipe( + fromX: Int, + fromY: Int, + toX: Int, + toY: Int, + durationMillis: Long, + ) {} override fun pressKey(key: String) {} override fun longPress(x: Int, y: Int) {} - override fun screenshot(): Triple = Triple(byteArrayOf(), 0, 0) + override fun screenshot(): Triple = + Triple(byteArrayOf(), 0, 0) override fun hierarchy(): String = "{}" - override fun recentLogs(sinceUnixMillis: Long, minLevel: String): List = emptyList() + override fun recentLogs( + sinceUnixMillis: Long, + minLevel: String, + ): List = emptyList() override fun waitForIdle(durationMillis: Long) {} override fun healthy(): Boolean = true - override fun metrics(bundleId: String): MetricsSample = MetricsSample(0.0, 0L, 0L) + override fun metrics(bundleId: String): MetricsSample = + MetricsSample(0.0, 0L, 0L) } val client = newClient(backend) @@ -252,13 +314,16 @@ class DriverServiceTest { val backend = StubDriverBackend("android") val client = newClient(backend) - client.eraseText(EraseTextRequest.newBuilder().setCharacterCount(11).build()) + client.eraseText( + EraseTextRequest.newBuilder().setCharacterCount(11).build(), + ) assertEquals(11, backend.lastEraseCharacterCount) } @Test fun screenshotReturnsBackendBytes() { val backend = object : DriverBackend by StubDriverBackend("android") { - override fun screenshot(): Triple = Triple(byteArrayOf(1, 2, 3), 1080, 2340) + override fun screenshot(): Triple = + Triple(byteArrayOf(1, 2, 3), 1080, 2340) } val client = newClient(backend) @@ -268,9 +333,17 @@ class DriverServiceTest { assertEquals(3, image.png.size()) } - @Test fun hierarchyReturnsBackendJson() { + // The runner reads the hierarchy a second time per step to see whether the + // screen changed while it was looking, so this has to answer with the tree + // the snapshot's read produces: same settle, same keyboard handling. Served + // off the bare backend read, the pair differs over what the backend did + // between the two reads rather than over what the app did. Measured on an + // API 34 emulator with an IME standing open, that is a 489-node tree + // against the snapshot's 134. + @Test fun hierarchyServesTheTreeTheSnapshotReads() { val backend = object : DriverBackend by StubDriverBackend("android") { - override fun hierarchy(): String = "{\"x\":1}" + override fun hierarchy(): String = "{\"bare\":1}" + override fun snapshotTree(): String = "{\"x\":1}" } val client = newClient(backend) @@ -294,7 +367,13 @@ class DriverServiceTest { @Test fun swipeForwardsEndpointsAndDuration() { var observed: Quintuple? = null val backend = object : DriverBackend by StubDriverBackend("android") { - override fun swipe(fromX: Int, fromY: Int, toX: Int, toY: Int, durationMillis: Long) { + override fun swipe( + fromX: Int, + fromY: Int, + toX: Int, + toY: Int, + durationMillis: Long, + ) { observed = Quintuple(fromX, fromY, toX, toY, durationMillis) } } @@ -325,14 +404,18 @@ class DriverServiceTest { @Test fun recentLogsReturnsBackendEntries() { val backend = object : DriverBackend by StubDriverBackend("android") { - override fun recentLogs(sinceUnixMillis: Long, minLevel: String): List { - return listOf(LogLine(1, "E", "AndroidRuntime", "boom")) - } + override fun recentLogs( + sinceUnixMillis: Long, + minLevel: String, + ): List = listOf(LogLine(1, "E", "AndroidRuntime", "boom")) } val client = newClient(backend) val response = client.recentLogs( - RecentLogsRequest.newBuilder().setSinceUnixMillis(0).setLevelAtLeast("E").build(), + RecentLogsRequest.newBuilder() + .setSinceUnixMillis(0) + .setLevelAtLeast("E") + .build(), ) assertEquals(1, response.entriesCount) assertEquals("AndroidRuntime", response.getEntries(0).tag) @@ -351,7 +434,9 @@ class DriverServiceTest { } val client = newClient(backend) - client.launch(LaunchRequest.newBuilder().setBundleId("com.launched").build()) + client.launch( + LaunchRequest.newBuilder().setBundleId("com.launched").build(), + ) client.metrics(MetricsRequest.getDefaultInstance()) assertEquals("com.launched", sampled) @@ -367,8 +452,12 @@ class DriverServiceTest { } val client = newClient(backend) - client.launch(LaunchRequest.newBuilder().setBundleId("com.launched").build()) - client.metrics(MetricsRequest.newBuilder().setBundleId("com.other").build()) + client.launch( + LaunchRequest.newBuilder().setBundleId("com.launched").build(), + ) + client.metrics( + MetricsRequest.newBuilder().setBundleId("com.other").build(), + ) assertEquals("com.other", sampled) } diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/EraseTextTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/EraseTextTest.kt new file mode 100644 index 0000000..cfa1131 --- /dev/null +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/EraseTextTest.kt @@ -0,0 +1,128 @@ +package dev.sanderling.sidecar + +import org.junit.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +class EraseTextTest { + + // maestro's eraseText sends one delete per character through its + // instrumentation, measured at 29.6 ms/char on the API 34 emulator: the + // 4096-character string the corpus types cost ~121s to clear, a fifth of a + // 20 minute run for one step. Selecting the field and deleting the + // selection is the same two key events whatever the field holds. + @Test fun aClearedFieldCostsTwoKeyEventsWhateverItsLength() { + for (length in listOf(1, 21, 512, 4096)) { + val sent = mutableListOf() + eraseFocusedField(length, { sent.add(it) }) { 0 } + assertEquals( + listOf(SELECT_ALL_COMMAND, DELETE_KEY_COMMAND), + sent, + "length $length must not scale the erase", + ) + } + } + + // The dangerous failure is a fast erase that leaves characters behind: the + // next InputText appends to the residue and every reading downstream is + // wrong with nothing to catch it. A field the select-all did not clear is + // finished off per character rather than assumed empty. + @Test fun aFieldTheSelectAllMissedIsFinishedOffPerCharacter() { + val sent = mutableListOf() + eraseFocusedField(4096, { sent.add(it) }) { 4096 } + + assertEquals(SELECT_ALL_COMMAND, sent.first()) + assertEquals(DELETE_KEY_COMMAND, sent[1]) + assertEquals( + 4096, + sent.drop(2).sumOf { command -> + command.removePrefix("input keyevent ").split(" ").size + }, + "every character must still be deleted", + ) + } + + // Unknown is not empty. A tree that cannot name the focused field is no + // evidence the erase worked, and the safe way to be wrong is the delete + // that costs time rather than the one that leaves residue. + @Test fun aFieldThatCannotBeReadIsFinishedOffRatherThanAssumedEmpty() { + val sent = mutableListOf() + eraseFocusedField(8, { sent.add(it) }) { null } + assertTrue(sent.size > 2, "an unverified erase must not stop at two") + } + + @Test fun nothingToEraseIssuesNoKeysAtAll() { + val sent = mutableListOf() + eraseFocusedField(0, { sent.add(it) }) { 0 } + eraseFocusedField(-1, { sent.add(it) }) { 0 } + assertEquals(emptyList(), sent) + } + + // Batching is what keeps the fallback affordable: one round trip per batch + // rather than one per character, measured 2.3 ms/char against maestro's + // 29.6. The count must survive the batching exactly. + @Test fun deleteKeyCommandsBatchesWithoutLosingACharacter() { + for (count in listOf(1, 199, 200, 201, 4096)) { + val commands = deleteKeyCommands(count, DELETE_BATCH_KEYS) + val keys = commands.flatMap { + it.removePrefix("input keyevent ").split(" ") + } + assertEquals(count, keys.size, "count $count") + assertTrue(keys.all { it == "67" }, "only KEYCODE_DEL") + assertTrue( + commands.size <= (count + DELETE_BATCH_KEYS - 1) / + DELETE_BATCH_KEYS, + "count $count used ${commands.size} round trips", + ) + } + } + + // The erase targets the field the runner just tapped, so the length that + // decides whether it worked is that field's, not some other field that + // legitimately still holds text. + // + // The tree these fixtures copy is the one the device really returns, and + // it holds the trap: an open keyboard puts a SECOND focused node in the + // tree, one of the IME's own keys, and it carries no text. Reading the + // first focused node would call a field that still holds 4096 characters + // empty, which is the one wrong answer that matters here. maestro's tree + // also carries no "editable" attribute at all, so the text field has to be + // recognised by its class. + @Test fun theFocusedFieldIsReadPastTheKeyboardsOwnFocusedKey() { + assertEquals(4096, focusedEditableTextLength(TREE_WITH_FULL_FIELD)) + assertEquals(0, focusedEditableTextLength(TREE_WITH_EMPTY_FIELD)) + } + + @Test fun aTreeWithNoFocusedFieldReadsAsUnknown() { + assertEquals(null, focusedEditableTextLength(TREE_WITH_NO_FOCUS)) + assertEquals(null, focusedEditableTextLength("")) + assertEquals(null, focusedEditableTextLength("not json")) + } +} + +// The keyboard's own focused key, exactly as the device reports it: focused, +// no text, and not a text field. +private val IME_FOCUSED_KEY = + """ + {"attributes":{"text":"","resource-id": + "com.google.android.inputmethod.latin:id/key_pos_header_access", + "focused":"true","class":"android.widget.FrameLayout"},"children":[]} + """.trimIndent() + +private fun tree(focusedText: String?, otherText: String): String { + val field = focusedText?.let { + """,{"attributes":{"resource-id":"AccountNameField","focused":"true", + "class":"android.widget.EditText","text":"$it"},"children":[]}""" + } ?: "" + return """ + {"attributes":{"resource-id":"AddAccountScreen"},"children":[ + $IME_FOCUSED_KEY $field, + {"attributes":{"resource-id":"OtherField","focused":"false", + "class":"android.widget.EditText","text":"$otherText"}, + "children":[]}]} + """.trimIndent() +} + +private val TREE_WITH_FULL_FIELD = tree("a".repeat(4096), "keep me") +private val TREE_WITH_EMPTY_FIELD = tree("", "keep me") +private val TREE_WITH_NO_FOCUS = tree(null, "keep me") diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/IdleDetectionTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/IdleDetectionTest.kt index 82d48bd..201b3fe 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/IdleDetectionTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/IdleDetectionTest.kt @@ -21,11 +21,17 @@ class IdleDetectionTest { assertFalse(StubDriverBackend.isAnimationCountIdle("3\n")) } - @Test fun idleWhenOutputEmpty() { - assertTrue(StubDriverBackend.isAnimationCountIdle("")) + // A dumpsys that said nothing does not say the device is still. Reading + // absence as idle is how a settle returns instantly on a degraded link and + // hands the runner a screen caught mid-animation; the caller bounds its own + // wait, so the cost of being wrong the other way is a wait it already + // budgeted for. The exception path of the same probe already answers false. + @Test fun unreadableOutputIsNotIdle() { + assertFalse(StubDriverBackend.isAnimationCountIdle("")) + assertFalse(StubDriverBackend.isAnimationCountIdle(" \n")) } - @Test fun idleWhenOutputIsNotANumber() { - assertTrue(StubDriverBackend.isAnimationCountIdle("error: no service")) + @Test fun unparseableOutputIsNotIdle() { + assertFalse(StubDriverBackend.isAnimationCountIdle("error: no service")) } } diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/InputTextTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/InputTextTest.kt index 41813c8..f9d3ad7 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/InputTextTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/InputTextTest.kt @@ -32,7 +32,8 @@ class InputTextTest { val fallback = listOf( "Emergency Fund", "🙂🔥💸", " ", "\t\n", "'; DROP TABLE--", "", "../../etc/passwd", "%s%n", "", - "-1", "-rf", // a leading dash could be read as an option by `input text` + // a leading dash could be read as an option by `input text` + "-1", "-rf", ) for (text in fallback) { assertTrue( @@ -190,12 +191,12 @@ class InputTextTest { @Test fun parseResumedPackageReadsEachResumedActivityWording() { val cases = mapOf( - " topResumedActivity=ActivityRecord{8b u0 app.folio/.MainActivity t42}" to - "app.folio", - " mResumedActivity: ActivityRecord{1c u0 com.example.app/.Home t9}" to - "com.example.app", - " ResumedActivity: ActivityRecord{2d u0 app.folio/com.folio.Detail t9}" to - "app.folio", + " topResumedActivity=ActivityRecord{8b u0 " + + "app.folio/.MainActivity t42}" to "app.folio", + " mResumedActivity: ActivityRecord{1c u0 " + + "com.example.app/.Home t9}" to "com.example.app", + " ResumedActivity: ActivityRecord{2d u0 " + + "app.folio/com.folio.Detail t9}" to "app.folio", ) for ((line, want) in cases) { assertEquals(want, parseResumedPackage(line), line) @@ -206,6 +207,77 @@ class InputTextTest { ) } + @Test fun aReadableDumpsysNamesTheResumedPackage() { + val warnings = mutableListOf() + assertEquals( + "app.folio", + typingOwner(RESUMED_DUMPSYS, "app.folio") { warnings.add(it) }, + ) + assertTrue(warnings.isEmpty(), "nothing to report when the read worked") + } + + // A dumpsys that said nothing is not evidence that focus is fine. Handing + // typeChunks a null owner turns the guard off outright, and the keystrokes + // then go wherever the foreground happens to be. + @Test fun anUnreadableDumpsysGuardsWithTheLaunchedBundle() { + val warnings = mutableListOf() + assertEquals( + "app.folio", + typingOwner("", "app.folio") { warnings.add(it) }, + ) + assertEquals(1, warnings.size, "a degraded guard must not be silent") + assertTrue(warnings.single().contains("app.folio"), warnings.single()) + } + + // Wording no marker matches is the same "we do not know" as an empty read. + @Test fun dumpsysWithNoResumedMarkerGuardsWithTheLaunchedBundle() { + assertEquals( + "app.folio", + typingOwner(" mFocusedApp=null\n nothing here\n", "app.folio") {}, + ) + } + + // The whole point of the fallback: on a link that cannot answer, typing is + // still guarded, so a foreground that was stolen stops it after the first + // chunk instead of spraying the rest into whatever took focus. + @Test fun unreadableLinkStillStopsTypingWhenFocusWasStolen() { + val owner = typingOwner("", "app.folio") {} + val sent = mutableListOf() + + typeChunks(listOf("aaa", "bbb", "ccc"), owner, { + "com.android.launcher" + }) { sent.add(it) } + + assertEquals( + listOf("aaa"), + sent, + "an unguarded type would have sent every chunk to the launcher", + ) + } + + // And the other half of the trade: the fallback must not turn a degraded + // link into a run that types nothing. A no-op InputText on every step is a + // green run that tested nothing, which is worse than the spray it avoids. + @Test fun unreadableLinkStillTypesWhenTheAppKeepsFocus() { + val owner = typingOwner("", "app.folio") {} + val sent = mutableListOf() + + val typed = typeChunks(listOf("aaa", "bbb", "cc"), owner, { + "app.folio" + }) { sent.add(it) } + + assertEquals(listOf("aaa", "bbb", "cc"), sent) + assertEquals(8, typed) + } + + // With no launch recorded there is nothing to guard against, and the honest + // answer is to say the guard is off rather than imply it ran. + @Test fun noLaunchedBundleLeavesTheGuardOffAndSaysSo() { + val warnings = mutableListOf() + assertEquals(null, typingOwner("", null) { warnings.add(it) }) + assertEquals(1, warnings.size) + } + @Test fun maestroKeyForResolvesAndRejects() { assertEquals(maestro.KeyCode.BACK, maestroKeyFor("back")) assertEquals(maestro.KeyCode.BACK, maestroKeyFor("BACK")) @@ -225,4 +297,172 @@ class InputTextTest { ) assertEquals(maestro.KeyCode.ESCAPE, maestroKeyFor("escape")) } + + // An IME left open hides every app node beneath it from the hierarchy, so a + // form whose submit button sits under the keyboard becomes unreachable for + // as long as the fuzzer keeps typing into it. Typing must close the + // keyboard it raised. + @Test fun dismissSoftKeyboardClosesAnOpenIme() { + val commands = mutableListOf() + dismissSoftKeyboard { + commands.add(it) + IME_OPEN_DUMPSYS + } + assertEquals( + listOf("dumpsys input_method", "input keyevent 4"), + commands, + ) + } + + // The guard is the dangerous half: BACK is only swallowed by an open IME, + // so dismissing unconditionally would turn every InputText into a back + // press and walk the fuzzer straight out of the screen it was filling in. + @Test fun dismissSoftKeyboardSendsNoBackWhenNoImeIsOpen() { + val commands = mutableListOf() + dismissSoftKeyboard { + commands.add(it) + IME_CLOSED_DUMPSYS + } + assertEquals(listOf("dumpsys input_method"), commands) + } + + // Typing is not the only thing that raises the keyboard: tapping a field + // raises it too, and nothing was closing that one. The snapshot the picker + // chooses from is missing every app node the keyboard covers, so the step + // spends its budget choosing between the few targets left. Closing it + // before the tree is read is what gives the step its targets back. + @Test fun aKeyboardInTheTreeIsClosedBeforeTheTreeIsReturned() { + var backs = 0 + val reads = mutableListOf() + val settled = treeWithoutKeyboard( + IME_TREE, + IME_PACKAGE, + dismiss = { backs++ }, + reread = { APP_TREE.also { reads.add(it) } }, + sleep = {}, + ) + assertEquals(APP_TREE, settled) + assertEquals(1, backs, "one BACK closes the keyboard") + assertEquals(1, reads.size, "the tree is re-read once it is gone") + } + + // The guard has to be the tree itself. BACK with no keyboard open + // navigates out of the screen, so a snapshot that pressed it on every read + // would walk the fuzzer backwards out of the app a step at a time. + @Test fun aTreeWithNoKeyboardIsReturnedUntouched() { + var backs = 0 + var reads = 0 + val settled = treeWithoutKeyboard( + APP_TREE, + IME_PACKAGE, + dismiss = { backs++ }, + reread = { + reads++ + APP_TREE + }, + sleep = {}, + ) + assertEquals(APP_TREE, settled) + assertEquals(0, backs, "no keyboard in the tree means no BACK") + assertEquals(0, reads, "and no second hierarchy read to pay for") + } + + // An unknown IME package is the honest "cannot tell", and the safe way to + // be wrong is to leave the keyboard up rather than press BACK blind. + @Test fun anUnknownImePackageSendsNoBack() { + var backs = 0 + assertEquals( + IME_TREE, + treeWithoutKeyboard( + IME_TREE, + null, + dismiss = { backs++ }, + reread = { APP_TREE }, + sleep = {}, + ), + ) + assertEquals(0, backs) + } + + // A keyboard the app puts straight back gets ONE back press, not one per + // re-read. The flag behind the older dismissal lags a BACK by up to 0.6s, + // and a burst of them inside that window is how a dismissal turns into + // navigation. + @Test fun aKeyboardThatStaysUpIsNotBackPressedRepeatedly() { + var backs = 0 + var reads = 0 + val settled = treeWithoutKeyboard( + IME_TREE, + IME_PACKAGE, + dismiss = { backs++ }, + reread = { + reads++ + IME_TREE + }, + sleep = {}, + ) + assertEquals(IME_TREE, settled, "the caller still gets a tree") + assertEquals(1, backs) + assertTrue( + reads in 1..KEYBOARD_DISMISS_READS, + "bounded re-reads, got $reads", + ) + } + + @Test fun imePackageOfReadsTheComponentAndRejectsNonsense() { + assertEquals( + "com.google.android.inputmethod.latin", + imePackageOf( + "com.google.android.inputmethod.latin/.LatinIME\n", + ), + ) + assertEquals(null, imePackageOf("null")) + assertEquals(null, imePackageOf("")) + assertEquals(null, imePackageOf(" \n")) + } + + @Test fun treeShowsImeMatchesTheImesOwnViewIdsOnly() { + assertTrue(treeShowsIme(IME_TREE, IME_PACKAGE)) + assertTrue(!treeShowsIme(APP_TREE, IME_PACKAGE)) + } } + +private val RESUMED_DUMPSYS = + """ + mFocusedApp=ActivityRecord{1a u0 app.folio/.MainActivity t14} + topResumedActivity=ActivityRecord{f3 u0 app.folio/.MainActivity t14} + """.trimIndent() + +private const val IME_PACKAGE = "com.google.android.inputmethod.latin" + +private val APP_TREE = + """ + {"attributes":{"resource-id":"AddAccountScreen"},"children":[ + {"attributes":{"resource-id":"AccountNameField"},"children":[]}, + {"attributes":{"resource-id":"AddAccountSubmit"},"children":[]}]} + """.trimIndent() + +// The submit control is gone: an open keyboard does not merely cover the node, +// it takes it out of the tree the picker enumerates. +private val IME_TREE = + """ + {"attributes":{"resource-id":"AddAccountScreen"},"children":[ + {"attributes":{"resource-id":"AccountNameField"},"children":[]}, + {"attributes":{ + "resource-id":"com.google.android.inputmethod.latin:id/keyboard_holder" + },"children":[]}]} + """.trimIndent() + +private val IME_OPEN_DUMPSYS = + """ + mCurMethodId=com.google.android.inputmethod.latin/.LatinIME + mInputShown=true + mSystemReady=true mInteractive=true + """.trimIndent() + +private val IME_CLOSED_DUMPSYS = + """ + mCurMethodId=com.google.android.inputmethod.latin/.LatinIME + mInputShown=false + mSystemReady=true mInteractive=true + """.trimIndent() diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/ResolveActivityTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/ResolveActivityTest.kt index 5ffaab8..77b25ab 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/ResolveActivityTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/ResolveActivityTest.kt @@ -12,28 +12,40 @@ class ResolveActivityTest { com.example.app/.MainActivity """.trimIndent() - val activity = StubDriverBackend.parseResolvedActivity("com.example.app", output) + val activity = StubDriverBackend.parseResolvedActivity( + "com.example.app", + output, + ) assertEquals(".MainActivity", activity) } @Test fun extractsFullyQualifiedActivity() { val output = "com.example.app/com.example.app.ui.LaunchActivity" - val activity = StubDriverBackend.parseResolvedActivity("com.example.app", output) + val activity = StubDriverBackend.parseResolvedActivity( + "com.example.app", + output, + ) assertEquals("com.example.app.ui.LaunchActivity", activity) } @Test fun returnsNullWhenPackageNotFound() { val output = "No activity found" - val activity = StubDriverBackend.parseResolvedActivity("com.example.app", output) + val activity = StubDriverBackend.parseResolvedActivity( + "com.example.app", + output, + ) assertNull(activity) } @Test fun doesNotMatchDifferentPackagePrefix() { val output = "other.pkg/.MainActivity" - val activity = StubDriverBackend.parseResolvedActivity("com.example.app", output) + val activity = StubDriverBackend.parseResolvedActivity( + "com.example.app", + output, + ) assertNull(activity) } } diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt index 3c7c776..7836f66 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/RouteTransitionTest.kt @@ -13,10 +13,13 @@ class RouteTransitionTest { private fun screen(id: String, child: String = "") = """{"attributes":{"resource-id":"$id"},"children":[$child]}""" - private fun tree(vararg children: String) = - """{"attributes":{"resource-id":"root"},"children":[${children.joinToString(",")}]}""" + private fun tree(vararg children: String): String { + val joined = children.joinToString(",") + return """{"attributes":{"resource-id":"root"},"children":[$joined]}""" + } - private val crossFade = tree(screen("LedgerScreen"), screen("AddTransactionScreen")) + private val crossFade = + tree(screen("LedgerScreen"), screen("AddTransactionScreen")) private val landed = tree(screen("AddTransactionScreen")) @Test fun waitsForTheCrossFadeToLandAndReturnsTheLandedTree() { @@ -29,8 +32,15 @@ class RouteTransitionTest { reads++ if (reads <= 3) crossFade else landed } - assertTrue(reads > 3, "must keep reading until the fade lands, reads=$reads") - assertEquals(1, countRouteScreens(settled), "must return a tree with one route") + assertTrue( + reads > 3, + "must keep reading until the fade lands, reads=$reads", + ) + assertEquals( + 1, + countRouteScreens(settled), + "must return a tree with one route", + ) } @Test fun settledFrameCostsExactlyOneRead() { @@ -54,13 +64,21 @@ class RouteTransitionTest { // would burn the whole poll budget and still hand over a frame the // runner refuses to act on. val nested = tree(screen("HomeScreen", screen("HomeScreen"))) - assertEquals(1, countRouteScreens(nested), "the same id twice is one route") + assertEquals( + 1, + countRouteScreens(nested), + "the same id twice is one route", + ) var reads = 0 awaitSettledTree { reads++ nested } - assertEquals(1, reads, "a repeated route id must not be treated as a transition") + assertEquals( + 1, + reads, + "a repeated route id must not be treated as a transition", + ) } @Test fun aLayoutThatKeepsTwoRoutesIsBoundedByTheCap() { @@ -78,7 +96,11 @@ class RouteTransitionTest { elapsed < TRANSITION_POLL_CAP_MILLIS + 1000L, "must stop at the cap, elapsed=${elapsed}ms", ) - assertEquals(crossFade, settled, "the caller still gets a tree to record") + assertEquals( + crossFade, + settled, + "the caller still gets a tree to record", + ) } @Test fun capCoversTheNavHostFadePlusTheStreak() { @@ -89,20 +111,29 @@ class RouteTransitionTest { val fadeMillis = 700L val start = System.currentTimeMillis() val settled = awaitSettledTree { - if (System.currentTimeMillis() - start < fadeMillis) crossFade else landed + if (System.currentTimeMillis() - start < fadeMillis) { + crossFade + } else { + landed + } } val elapsed = System.currentTimeMillis() - start - assertEquals(landed, settled, "must hand back the landed tree, not the fade") + assertEquals( + landed, + settled, + "must hand back the landed tree, not the fade", + ) assertTrue( elapsed >= fadeMillis, "cannot have settled before the fade ended, elapsed=${elapsed}ms", ) assertTrue( elapsed < TRANSITION_POLL_CAP_MILLIS, - "the ${TRANSITION_POLL_CAP_MILLIS}ms cap has to leave room for a ${fadeMillis}ms " + - "fade and the ${TRANSITION_STABLE_STREAK_MILLIS}ms streak after it, but the " + - "wait ran to the cap instead, elapsed=${elapsed}ms", + "the ${TRANSITION_POLL_CAP_MILLIS}ms cap has to leave room for " + + "a ${fadeMillis}ms fade and the " + + "${TRANSITION_STABLE_STREAK_MILLIS}ms streak after it, but " + + "the wait ran to the cap instead, elapsed=${elapsed}ms", ) } } diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/SidecarServerTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/SidecarServerTest.kt index f449c42..05e24a8 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/SidecarServerTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/SidecarServerTest.kt @@ -6,7 +6,10 @@ import kotlin.test.assertTrue class SidecarServerTest { @Test fun startBindsEphemeralPortAndStopReleasesIt() { - val server = SidecarServer(port = 0, service = DriverService(backend = StubDriverBackend("android"))) + val server = SidecarServer( + port = 0, + service = DriverService(backend = StubDriverBackend("android")), + ) val boundPort = server.start() try { assertTrue(boundPort > 0, "expected ephemeral port, got $boundPort") diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/SnapshotHandlerTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/SnapshotHandlerTest.kt index 13c6e45..2252e4c 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/SnapshotHandlerTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/SnapshotHandlerTest.kt @@ -19,14 +19,22 @@ class SnapshotHandlerTest { @get:Rule val grpcCleanup: GrpcCleanupRule = GrpcCleanupRule() - private fun newClient(backend: DriverBackend): DriverGrpc.DriverBlockingStub { + private fun newClient( + backend: DriverBackend, + ): DriverGrpc.DriverBlockingStub { val serverName = InProcessServerBuilder.generateName() val service = DriverService(platform = "android", backend = backend) grpcCleanup.register( - InProcessServerBuilder.forName(serverName).directExecutor().addService(service).build().start(), + InProcessServerBuilder.forName(serverName) + .directExecutor() + .addService(service) + .build() + .start(), ) val channel: ManagedChannel = grpcCleanup.register( - InProcessChannelBuilder.forName(serverName).directExecutor().build(), + InProcessChannelBuilder.forName(serverName) + .directExecutor() + .build(), ) return DriverGrpc.newBlockingStub(channel) } @@ -37,8 +45,10 @@ class SnapshotHandlerTest { // forward those calls to the delegate, not these overrides. Override // snapshot() directly so the test exercises the wire path end-to-end. val backend = object : DriverBackend by StubDriverBackend("android") { - override fun snapshot(): SnapshotSample = - SnapshotSample("{\"x\":1}", Triple(byteArrayOf(7, 8, 9), 1080, 2340)) + override fun snapshot(): SnapshotSample = SnapshotSample( + "{\"x\":1}", + Triple(byteArrayOf(7, 8, 9), 1080, 2340), + ) } val client = newClient(backend) @@ -55,13 +65,23 @@ class SnapshotHandlerTest { // aligned with the final hierarchy snapshot the runner accepts. val callOrder = mutableListOf() val backend = object : DriverBackend { - override fun launch(bundleId: String, clearState: Boolean, env: Map) {} + override fun launch( + bundleId: String, + clearState: Boolean, + env: Map, + ) {} override fun terminate(bundleId: String) {} override fun tap(x: Int, y: Int) {} override fun tapSelector(selector: String) {} override fun inputText(text: String) {} override fun eraseText(characterCount: Int) {} - override fun swipe(fromX: Int, fromY: Int, toX: Int, toY: Int, durationMillis: Long) {} + override fun swipe( + fromX: Int, + fromY: Int, + toX: Int, + toY: Int, + durationMillis: Long, + ) {} override fun pressKey(key: String) {} override fun longPress(x: Int, y: Int) {} override fun screenshot(): Triple { @@ -72,10 +92,14 @@ class SnapshotHandlerTest { callOrder.add("hierarchy") return "{}" } - override fun recentLogs(sinceUnixMillis: Long, minLevel: String): List = emptyList() + override fun recentLogs( + sinceUnixMillis: Long, + minLevel: String, + ): List = emptyList() override fun waitForIdle(durationMillis: Long) {} override fun healthy(): Boolean = true - override fun metrics(bundleId: String): MetricsSample = MetricsSample(0.0, 0L, 0L) + override fun metrics(bundleId: String): MetricsSample = + MetricsSample(0.0, 0L, 0L) } backend.snapshot() assertEquals(listOf("hierarchy", "screenshot"), callOrder) @@ -88,7 +112,8 @@ class SnapshotHandlerTest { val maxObserved = AtomicInteger(0) val callCount = AtomicInteger(0) val lock = ReentrantLock() - val recordingBackend = object : DriverBackend by StubDriverBackend("android") { + val delegate = StubDriverBackend("android") + val recordingBackend = object : DriverBackend by delegate { override fun snapshot(): SnapshotSample { val now = inFlight.incrementAndGet() try { @@ -109,9 +134,13 @@ class SnapshotHandlerTest { // Use a real (multi-threaded) executor on the server side so the service // is not artificially serialized by directExecutor. val serverName = InProcessServerBuilder.generateName() - val service = DriverService(platform = "android", backend = recordingBackend) + val service = + DriverService(platform = "android", backend = recordingBackend) grpcCleanup.register( - InProcessServerBuilder.forName(serverName).addService(service).build().start(), + InProcessServerBuilder.forName(serverName) + .addService(service) + .build() + .start(), ) val channel: ManagedChannel = grpcCleanup.register( InProcessChannelBuilder.forName(serverName).build(), diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt index c8da826..29d8c79 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/StabilityPollTest.kt @@ -13,7 +13,10 @@ class StabilityPollTest { elapsed >= MIN_STABLE_STREAK_MILLIS, "must observe a stable streak of at least ${MIN_STABLE_STREAK_MILLIS}ms, elapsed=${elapsed}ms", ) - assertTrue(elapsed < 3000L, "should not run to cap when stable, elapsed=${elapsed}ms") + assertTrue( + elapsed < 3000L, + "should not run to cap when stable, elapsed=${elapsed}ms", + ) } @Test fun slowSnapshotReadsDoNotEatTheStreak() { @@ -38,8 +41,9 @@ class StabilityPollTest { val observedQuiet = sampleStarts.last() - sampleEnds.first() assertTrue( observedQuiet >= MIN_STABLE_STREAK_MILLIS, - "the poll returned having observed only ${observedQuiet}ms of quiet, not " + - "${MIN_STABLE_STREAK_MILLIS}ms; starts=$sampleStarts ends=$sampleEnds", + "the poll returned having observed only ${observedQuiet}ms of " + + "quiet, not ${MIN_STABLE_STREAK_MILLIS}ms; " + + "starts=$sampleStarts ends=$sampleEnds", ) assertTrue( sampleStarts.size >= 3, @@ -60,23 +64,27 @@ class StabilityPollTest { calls++ when { calls <= 2 -> "calm" + calls == 3 -> { transientAt = System.currentTimeMillis() "transient" } + else -> "stable" } } val sinceTransition = System.currentTimeMillis() - transientAt assertTrue( calls >= 8, - "after the transition the poll needs a fresh matching pair and then a full " + - "${MIN_STABLE_STREAK_MILLIS}ms of quiet, which is 8 samples, got $calls", + "after the transition the poll needs a fresh matching pair and " + + "then a full ${MIN_STABLE_STREAK_MILLIS}ms of quiet, which " + + "is 8 samples, got $calls", ) assertTrue( sinceTransition >= MIN_STABLE_STREAK_MILLIS, - "the calm prefix must not count: a full ${MIN_STABLE_STREAK_MILLIS}ms streak has to " + - "start over after the transition, returned ${sinceTransition}ms after it", + "the calm prefix must not count: a full " + + "${MIN_STABLE_STREAK_MILLIS}ms streak has to start over " + + "after the transition, returned ${sinceTransition}ms after it", ) } @@ -105,7 +113,10 @@ class StabilityPollTest { "frame-$calls" } val elapsed = System.currentTimeMillis() - start - assertTrue(elapsed in budget..(budget + 1000L), "expected to hit cap, elapsed=$elapsed") + assertTrue( + elapsed in budget..(budget + 1000L), + "expected to hit cap, elapsed=$elapsed", + ) } @Test fun zeroBudgetReturnsImmediately() { @@ -117,7 +128,8 @@ class StabilityPollTest { assertEquals(0, calls) } - @Test fun structuralHashIgnoresBoundsAndIdenticalForSemanticallyEqualTrees() { + @Test + fun structuralHashIgnoresBoundsAndIdenticalForSemanticallyEqualTrees() { val a = """ {"attributes":{"resource-id":"LoginScreen","bounds":"[0,0,1080,2340]"}, "children":[ @@ -130,13 +142,24 @@ class StabilityPollTest { {"attributes":{"resource-id":"LoginEmail","bounds":"[10,11,1070,101]","text":"a@b"},"children":[]} ]} """.trimIndent() - assertEquals(structuralHash(a), structuralHash(b), "bounds-only flicker must not change hash") + assertEquals( + structuralHash(a), + structuralHash(b), + "bounds-only flicker must not change hash", + ) } @Test fun structuralHashDiffersWhenContentChanges() { - val a = """{"attributes":{"resource-id":"LoginEmail","text":"a@b"},"children":[]}""" - val b = """{"attributes":{"resource-id":"LoginEmail","text":"c@d"},"children":[]}""" - assertTrue(structuralHash(a) != structuralHash(b), "text change must alter hash") + val a = """ + {"attributes":{"resource-id":"LoginEmail","text":"a@b"},"children":[]} + """.trimIndent() + val b = """ + {"attributes":{"resource-id":"LoginEmail","text":"c@d"},"children":[]} + """.trimIndent() + assertTrue( + structuralHash(a) != structuralHash(b), + "text change must alter hash", + ) } @Test fun stabilitySnapshotReturnsNullDuringNavHostCrossFade() { @@ -160,7 +183,10 @@ class StabilityPollTest { ]} """.trimIndent() val hash = stabilitySnapshot(singleScreen) - assertTrue(hash != null && hash.isNotBlank(), "single-screen tree must yield a hash, got $hash") + assertTrue( + hash != null && hash.isNotBlank(), + "single-screen tree must yield a hash, got $hash", + ) } @Test fun stabilitySnapshotIgnoresNonRouteAttributeValues() { @@ -173,7 +199,10 @@ class StabilityPollTest { {"attributes":{"text":"Welcome to MyScreen"},"children":[]} ]} """.trimIndent() - assertTrue(stabilitySnapshot(tree) != null, "non-route attribute must not be counted as a screen") + assertTrue( + stabilitySnapshot(tree) != null, + "non-route attribute must not be counted as a screen", + ) } @Test fun countRouteScreensCountsTestTagAndIdentifier() { diff --git a/sidecar/src/test/kotlin/dev/sanderling/sidecar/WdaRecoveryTest.kt b/sidecar/src/test/kotlin/dev/sanderling/sidecar/WdaRecoveryTest.kt index e3eb3d8..6062e41 100644 --- a/sidecar/src/test/kotlin/dev/sanderling/sidecar/WdaRecoveryTest.kt +++ b/sidecar/src/test/kotlin/dev/sanderling/sidecar/WdaRecoveryTest.kt @@ -12,14 +12,15 @@ import kotlin.test.assertTrue class WdaRecoveryTest { - private fun recovery( - isAlive: () -> Boolean, - restart: () -> Unit, - ) = WdaRecovery(isAlive = isAlive, restart = restart, log = {}) + private fun recovery(isAlive: () -> Boolean, restart: () -> Unit) = + WdaRecovery(isAlive = isAlive, restart = restart, log = {}) @Test fun aliveChannelSkipsRestartAndRetriesReads() { val restarts = AtomicInteger(0) - val recovery = recovery(isAlive = { true }, restart = { restarts.incrementAndGet() }) + val recovery = recovery( + isAlive = { true }, + restart = { restarts.incrementAndGet() }, + ) var calls = 0 val result = recovery.run(replay = true) { @@ -35,10 +36,15 @@ class WdaRecoveryTest { @Test fun aliveChannelSurfacesUnavailableForActions() { val restarts = AtomicInteger(0) - val recovery = recovery(isAlive = { true }, restart = { restarts.incrementAndGet() }) + val recovery = recovery( + isAlive = { true }, + restart = { restarts.incrementAndGet() }, + ) val thrown = assertFailsWith { - recovery.run(replay = false) { throw IOException("connection reset") } + recovery.run(replay = false) { + throw IOException("connection reset") + } } assertEquals(io.grpc.Status.Code.UNAVAILABLE, thrown.status.code) @@ -101,7 +107,9 @@ class WdaRecoveryTest { ) val thrown = assertFailsWith { - recovery.run(replay = true) { throw IOException("connection refused") } + recovery.run(replay = true) { + throw IOException("connection refused") + } } assertTrue(thrown.message.orEmpty().contains("WDA reconnect failed")) @@ -116,7 +124,9 @@ class WdaRecoveryTest { ) assertFailsWith { - recovery.run(replay = true) { throw IllegalArgumentException("bad selector") } + recovery.run(replay = true) { + throw IllegalArgumentException("bad selector") + } } assertEquals(0, restarts.get()) @@ -127,7 +137,9 @@ class WdaRecoveryTest { val recovery = recovery(isAlive = { true }, restart = {}) val thrown = assertFailsWith { - recovery.run(replay = true) { throw IOException("connection reset") } + recovery.run(replay = true) { + throw IOException("connection reset") + } } assertEquals(io.grpc.Status.Code.UNAVAILABLE, thrown.status.code) diff --git a/skills/README.md b/skills/README.md new file mode 100644 index 0000000..50c7887 --- /dev/null +++ b/skills/README.md @@ -0,0 +1,15 @@ +# Skills for writing sanderling specs + +Agent skills for adopting sanderling: getting it running against your app, writing +property specifications, reviewing them for the failure modes that make a spec +look like it works when it does not, and reading a run honestly. + +Copy the ones you want into your agent's skills directory (`.claude/skills/` for +Claude Code) or point your agent at this directory directly. + +Start with `sanderling-setup`, then `sanderling-spec-authoring`. Run +`sanderling-spec-review` over anything before you trust it. + +The reasoning behind the rules these encode is in +[docs/development/design-principles.md](../docs/development/design-principles.md), +section 8 in particular. diff --git a/skills/sanderling-property-patterns/SKILL.md b/skills/sanderling-property-patterns/SKILL.md new file mode 100644 index 0000000..0c7c142 --- /dev/null +++ b/skills/sanderling-property-patterns/SKILL.md @@ -0,0 +1,442 @@ +--- +name: sanderling-property-patterns +description: Decide what a sanderling spec should assert. A catalogue of property shapes that are sound (cross-panel agreement, bounds on an effect, counting actions against effects, input and navigation invariants), each with the tempting unsound version beside it. Use when starting a spec, when adding a property to one, or when a property keeps convicting an app that behaved. +--- + +# Choosing what to assert + +You have sanderling driving your app and now you have to say what must be true. +This is the hard part, and it fails in two directions: you freeze, or you write +six properties none of which can ever be false. + +One rule orders everything below. **Soundness outranks detection.** A property +that convicts more often and is sometimes wrong is strictly worse than one that +convicts less and is never wrong, because a false conviction costs someone a day +and then costs the whole suite its credibility. When a property cannot establish +what it needs, it declines. + +Each shape below gives the sound form and the tempting form next to it, because +the tempting one is usually what gets written first. The examples are from the +two specs in this repo: `replay-ui/sanderling/spec.ts` (sanderling fuzzing its +own trace browser) and `examples/folio/sanderling/spec.ts` with +`examples/folio/sanderling/predicates.ts` (a KMP finance app). + +Once you have written properties, run `sanderling-spec-review` over them. It +audits what this file helps you build. + +## 1. Two parts of the UI derive the same fact and must agree + +Reach for this first, always. If your app shows the same number in two places, +or shows a thing and a count of that thing, or renders a list and a selection +into that list, you have a property and you do not have to think about windows, +calibration, or attribution to write it. + +It is the strongest shape available. It holds on any run against any data, so +nothing needs recalibrating when a fixture changes; it needs no reasoning about +which action caused what; and an app that drifted on one of the two paths cannot +satisfy it. It is the backbone of `replay-ui/sanderling/spec.ts`, which states it +three times over: the toolbar's step count against the number of rows the list +renders, the toolbar's step against the step the screenshot panel built its URL +from, and the tab badge's violation count against the number of rows the +violations panel shows. + +```ts +const stepCountMatchesTheList = always(() => { + const current = toolbar.current; + const rows = stepRows.current; + if (!current || current.stepCount === null || rows.length === 0) return true; + return current.stepCount === rows.length; +}); +``` + +**What goes wrong: reading the second value off the wrong element.** Scope each +reading to the panel you mean, by name, not by position in the tree. + +```ts +// tempting: the first screenshot on the page +s.ax.find({ "data-testid": "screenshot" }) +// sound: the before panel's screenshot +s.ax.find([{ "data-testid": "state-before" }, { "data-testid": "screenshot" }]) +``` + +Both versions pass most of the time. The fuzzer put the before panel on another +tab, which left the after panel's image first on the page, and the first version +fired against a UI that was behaving correctly. + +**What else goes wrong: never getting both readings onto one step.** This +shape's failure mode is vacuity, not false conviction, which makes it quiet. An +undirected run over replay-ui went 40 steps without switching a single tab out +of roughly 15 clickable elements, leaving both tab-facing properties vacuously +true. The fix is in the action tree, not the property: give the action that +brings the second reading into view its own weight. + +```ts +const switchATab = actions(() => { + const tabs = tabElements.current; + return tabs.length === 0 ? [] : [Tap({ on: from(tabs).generate() })]; +}); + +export const actionsRoot = weighted( + [25, switchATab], + [20, showAViolatingStepWithItsPanel], + [25, defaultActions], +); +``` + +Weighting one half is usually not enough, and this is the part that surprises +people. `badgeCountMatchesThePanel` needs a badge, which a tab strip renders +only for a step that has a violation, and a panel to compare it against, which +exists only while a particular tab is selected. Undirected actions put both on +the same step 0 times in the 80 steps of replay-ui's first dogfood run. Aiming +at the violating step alone just moved the misses to the other side: still 0 +judged. `showAViolatingStepWithItsPanel` in that spec aims at both halves in +sequence, selecting a violating row and then opening a panel if none is up. + +It also opens the *after* panel deliberately, because the before panel's +screenshot is what `screenshotShowsTheSelectedStep` reads, and covering it up +would buy one property's evidence with another's. When two properties read the +same screen, an action tree can starve one to feed the other, and nothing in the +run output will say so. + +## 2. An effect must not exceed what the actions could have caused + +When the app has an effect you can measure (money moved, rows added, a counter +climbed), state a bound on it rather than a prediction of it. + +**Prefer an upper bound to an equality.** This is the single most valuable +sentence in this file. + +Folio shipped one of these both ways and the equality lost, so the two are worth +reading side by side. Each line is the last line of a predicate in +`examples/folio/sanderling/predicates.ts`, after the guards, at a step where +exactly one submit sits in the window: + +```ts +// what folio's total-balance property demanded, until 6e8e6d5 +Math.abs(currTotalBalance - prevTotalBalance) === typedAmount +// what it demands now +Math.abs(currTotalBalance - prevTotalBalance) <= typedAmount +// and the same bound stated as the violation, over the account's own balance +Math.abs(currAccountBalance - prevAccountBalance) > typedAmount +``` + +All three catch the bug, because a double submit moves the balance by twice the +typed amount and twice x exceeds x. Only the equality also convicts an app that +behaved. A balance that has not moved is a commit still in flight (folio's +`createTransaction` runs in a coroutine, and Home's total re-renders on the +store's own schedule), a submit the app rejected, or a tap that never landed, +and none of those is evidence of anything. + +The asymmetry is the point. Moving by more than one submit's worth is not +something a correct app can do, so the bound needs no case for any of the three. +The equality needs a case for each, and every one you forget is a false +conviction. Two of those cases are facts the runner cannot promise you (see +below): an action it could not confirm was applied, and an action it had to +relaunch the app after. Both leave the balance under the bound and both break an +equality, so a bound counts them and an equality has to decline on them. + +You do give something up, so make the trade deliberately. A bound cannot see a +balance that moved by *less* than the typed amount, and for a ledger that is a +real bug. The question to settle before giving it up is whether your readings are +tight enough to tell "moved by less" from "has not finished moving yet". If they +are not, the equality was never detecting that bug either; it was reporting it at +random. + +The bound has one precondition, and it is the same one as shape 3: it bounds the +effect by what the actions in the window could have caused, so the window has to +count every action that could cause the effect. Miss one and the bound is not a +bound. + +## 3. Count the actions, not the amounts + +The same bound, stated in counts. One action must not produce two effects. + +```ts +!committedTransactionsExceedSubmits({ + countsBefore: homeTxnCounts.previous ?? null, + countsAfter: homeTxnCounts.current, + submitsInWindow: submitsSinceCounts.current, +}) +``` + +Reach for this whenever the effect is countable. No arithmetic on values the UI +formatted and you parsed back, no float precision to reason about, and it stays +sound however wide the window between two readings gets, since both sides +accumulate over the same window. + +It has exactly two failure modes and both are about the window. Neither makes it +unsound. Both make it useless, quietly. + +**The window has to close often enough to attribute anything.** The window opens +when you last read the fact and closes when you read it again, so a run that +wanders away from that screen accumulates budget on one side of the bound +without accumulating evidence on the other. Measured on a real iOS run: it went +from step 19 to step 136 without returning to the screen the property reads, +giving a transaction rise of 15 against a window of 37 submits. 15 is not more +than 37, so nothing was reported. The same run also gave 4 against 7, 6 against +13, and 1 against 1. Sound throughout, detected nothing. + +The obvious fix is to read the fact somewhere the run visits often. The better +one, when the wide window is the app's own shape rather than an accident, is to +**state the same rule a second time over a narrower window**, which is what +folio's spec now does: + +```ts +const submitCommitsOneTransactionPerAction = always( + next( + () => + !committedTransactionsExceedSubmits({ /* Home's counts, wide window */ }) && + !committedAmountExceedsOneSubmit({ /* this account's balance, narrow window */ }), + ), +); +``` + +The counting form can only close its window on a Home reading, and a walk that +stays inside the transaction flow leaves it hundreds of steps and dozens of +submits wide. The second conjunct says the same thing in money about the one +account whose screen the walk is already on, and the transaction flow redraws +that balance on nearly every frame, so its window is usually a single action +wide, narrow enough to tell one commit from two. One rule, two windows, and the +narrow one is where the detection actually comes from. + +That only works because the two readings are kept from spanning two accounts: +`readAccountBalance` drops its carrier on every route that is not the ledger or +the transaction screen, transition frames included. A narrow window buys nothing +if the pair it compares straddles two different subjects. + +**The window must not be spent on actions that provably could not cause the +effect.** A bound inflated by taps that commit nothing is a bound the app can +never exceed, which is slack a real double submit hides behind. Folio's +transaction submit is `clickable(enabled = amount.isNotBlank())`, so a tap with +an empty field never fires at all, and the app's own `parseCents` refuses +anything outside `^\d+(\.\d{1,2})?$` or parsing to zero. Measured over four +recorded Android runs, 19, 11, 25 and 25 of 35, 26, 42 and 42 submit taps landed +with the amount field empty, which is roughly half the budget in every one. + +`submitCouldCommit` in `predicates.ts` is that rule, and note how narrowly it is +drawn. It returns false only where folio's own code **must** have refused, and +returns true for anything it cannot rule out, including an undefined reading and +an amount too large for the app to hold. Over-counting costs a detection; +under-counting convicts a healthy app. + +Establishing that an action could not have had an effect is app knowledge, not +something the runner can tell you, and it has to come from the frame the tap +read. Folio reads the amount field on the landing frame, which is sound because +the tap changes nothing about it and one action runs per step. +`element.enabled` is on every `AccessibilityElement` for the general case, +though whether your platform populates it honestly is worth checking on a real +tree rather than assuming. + +The mirror of this rule matters just as much: an action whose effect you cannot +rule out **must** be counted. Leaving out submits whose dispatch the runner +could not confirm is what once convicted a healthy app here, when the property +saw a transaction rise of one against a window of zero. + +## 4. A value the user can reach must stay inside its legal range + +Anything the user can type, or a URL can carry, or a deep link can set, is +attacker-controlled input to your app even when the attacker is a fuzzer. The +property is that it stays legal, and it is cheap: one reading, no window, no +attribution. + +```ts +const selectedStepIsInRange = always(() => { + const current = toolbar.current; + if (!current || current.step === null || current.stepCount === null) return true; + return current.step >= 1 && current.step <= current.stepCount; +}); +``` + +Two things make that sound. The bound comes from the app's own reading of how +many steps the run has, not from a number you typed after looking at a fixture. +And it asserts legality rather than a prediction: + +```ts +// tempting: I tapped next, so it must now be on step n + 1 +toolbar.current.step === (toolbar.previous?.step ?? 0) + 1 +``` + +which is false at the end of the run, false when the tap did not land, and false +whenever the app is within its rights to clamp. Assert what must not happen. + +## 5. State machine and navigation invariants + +Every screen with a selection, a mode, or a route has invariants that are true +by construction and therefore worth stating, because "by construction" is +exactly what breaks. + +**Exactly one, not at least one.** The looser version is the tempting one and it +gives up the interesting half of the bug. + +```ts +const exactlyOneStepIsSelected = always(() => { + const rows = stepRows.current; + if (rows.length === 0) return true; + return rows.filter((row) => row.active).length === 1; +}); +``` + +Two selected rows is a stuck selection. Zero is the toolbar showing a step the +list has no row for, which is what an off-by-one or a failed clamp looks like +from the list's side, and `>= 1` would never see it. + +**A view change must not be a navigation.** Switching a tab, opening a menu or +toggling a theme must leave the app where it was. + +```ts +const switchingTabsKeepsTheStep = always( + next(() => { + const previousTabs = activeTabs.previous; + const previousToolbar = toolbar.previous; + const currentToolbar = toolbar.current; + if (previousTabs === undefined || previousTabs === activeTabs.current) return true; + if (!previousToolbar || !currentToolbar) return true; + return previousToolbar.step === currentToolbar.step; + }), +); +``` + +Note the guard: it declines unless the tab strip actually changed. A property +about an event must first establish that the event happened. + +The tempting unsound version of a navigation property is asserting the route you +were hoping for, `route.current === "home"` after tapping submit. The app is +within its rights to show a validation error and stay, and folio does exactly +that for an amount of zero. State what must not happen, not what you wanted to. + +Deriving the route at all deserves care, and folio's `routeOfFrame` is the +pattern: it returns the screen only when exactly one screen marker is in the +tree, and null otherwise. Android's hierarchy dump carries the outgoing and the +incoming screen together on 425 of 1879 steps measured across 17 runs, better +than one frame in five. Such a frame is evidence about neither screen, and +ranking the markers to pick one is how a spec convicts itself on an animation. + +## 6. The ones you get for free, on one platform each + +```ts +import { noUncaughtExceptions, noLogcatErrors } from "@sanderling/spec/defaults"; + +export const properties = { noUncaughtExceptions, /* yours */ }; +``` + +Both read a field the driver fills, and each field is filled on one platform, so +check which one is yours before counting either as coverage. Folio's spec exports +neither, and that is the tell: one spec drives its Android, iOS and web builds, +and neither of these holds anything on all three. + +`noUncaughtExceptions` fails when `state.exceptions` is non-empty. Only the web +runtime fills it, from `error` and `unhandledrejection` listeners installed in +the page by `pkg/spec/src/web-runtime.ts`. On web it is worth the line: a fuzzer +typing `'; DROP TABLE--` and a 4096-character string into every field it finds +will surface real breakage through it. On Android and iOS the field is never +populated, so the property holds at every step of a run that crashed. + +`noLogcatErrors` fails on a log line at level `E`. An uncaught Java or Kotlin +throwable is logged there, so on Android it is the nearest equivalent and worth +turning on once you know your app's log hygiene can support it. It holds +vacuously on web and iOS. + +That leaves iOS with neither, and it leaves both platforms uncovered for the +thing that matters most anyway. An app can be thoroughly wrong about money +without throwing once. + +## The rules that cut across all of them + +**Absence is unknown, never a default.** Extractors return null when the element +is not there, and a property handed null declines. `0`, `""` and `[]` are the +values that turn a property into one that fires on healthy runs: folio's +balances once parsed as `0` on web, so the check, an equality at the time, +became `|0 - 0| === typed` and was false at every healthy submit. Under today's +bound the same `0` reads as `|0 - 0| <= typed` and passes at every submit +instead, which is the same defect wearing green. An empty list has the same +problem in the other direction, and it is worse because it looks reasonable. +Android renders Home's own node a frame or two before its list, so `findAll` over the cards +comes back empty while the screen already claims to be Home. That is unknown, +not "no accounts", and reading it as zero accounts killed folio's counting +invariant outright: `countsBefore` was `{}` at every evaluation point of all 17 +runs measured. + +**Attribution needs injective keys.** If two distinct objects can produce the +same identity key, a value silently jumps between unrelated series. Merged UI +text is the usual culprit: web collapses an account card into a single node +whose text runs the name into the count, so an account named `Travel1` with 25 +transactions and one named `Travel12` with 5 both render `TRTravel125 +transactions`. No function of that string can separate them. Where a key can +collide, drop the reading rather than guess: `homeTxnCountsOf` leaves out any +name carried by more than one card, because subtracting two different accounts' +counts convicts a healthy app of double-submitting. + +**Match whole keys, not endings.** `endsWith` attribution judges an older +account named `Emergency Fund` when the user typed `Fund`: have the new card +clipped out of the reading, the way a list clips any card, and the old account +is convicted for money it has held all along. Substring matching is looser +still. Build every form of the key the platforms can produce and compare each +one whole, which is what `createdAccountHasNonZeroBalance` does with +`account.name === typed || account.name === initialsOf(typed) + typed`. Note +that this bought detections as well as soundness: under the suffix test, a name +that two cards ended with was thrown away as unattributable rather than matched +to the one card that actually carried it. + +**What the runner could not promise.** `state.lastAction` is +`Action & { applied: true | null; relaunched: true | null }`, and collapsing any +of its states is unsound: + +- `null`, the whole field, means no action ran +- `applied: true` means the runner saw the dispatch succeed +- `applied: null` means it was dispatched and nobody can find out whether it + landed, because an RPC deadline can fire after the tap arrived +- `relaunched: true` means the runner had to bring the app back to the + foreground after this action, so the two readings straddle a restart + +One rule covers the last two, and it is the rule that decides shape 2 for you. +An action the runner cannot fully vouch for **still counts toward a bound on +what the app could have done**, and it **never licenses attributing an effect to +it**. So a bound counts it and a property demanding an effect has to decline on +it. That is why `committedAmountExceedsOneSubmit`, which only bounds how far the +balance could have moved, needs no `confirmedApplied` guard, while +`createdAccountHasNonZeroBalance`, which demands that a card appear, does. +Demanding the effect of an action that may never have run convicts the app of +the runner's own uncertainty. + +A relaunch is not symmetric with that, and folio is a good illustration of why a +bound can still need the guard. `SqlLedgerStore` starts its flows on +`stateIn(Eagerly, emptyList())` and Home composes the total unconditionally, so +a restarted app draws `$0.00` until sqlite answers. That is not a commit the +restart swallowed, it is a reading of the wrong process, and its size is +arbitrary. A bound cannot absorb it, so the balance properties keep +`acrossRelaunch` even though they are bounds. Work out what a restart does to +the reading, not just to the effect. + +`relaunched` is the same shape of fact as `applied`, applied to app state rather +than to dispatch. The action itself did happen. What nobody can promise across +it is that the process ran continuously, that the commit survived, or that the +screen is showing the same slice of the same list it was. So a property assuming +continuous state declines, via `acrossRelaunch(lastAction)`. +`createdAccountHasNonZeroBalance` declines because Home redraws from the top and +the card carrying the typed name may be an older account laid out where the new +one used to be. `countSubmitsInWindow` uses the same call to **stop trusting its +own refusal evidence**, since a relaunch is the one thing that can put a form +state on screen other than the one the tap read. + +Both fields are `true | null` rather than booleans, and that is deliberate: only +the positive report is a fact the runner can vouch for, so `null` is "not +reported" rather than "did not happen". `relaunched` shows why it has to be that +way. Web and iOS cannot read the foreground at all, so they never relaunch the +app and equally cannot promise it never restarted, and a `false` there would be +a claim nobody is in a position to make. Read the absence as a guarantee and you +have made the same mistake as reading a missing value as zero, one level up. + +**Testing a property means both directions, every time.** + +- it fires on the bug it exists to catch +- it stays silent on a run where the app behaved + +The second is the one people skip and the one that catches unsoundness. Build +the fixture where the effect happens legitimately, at the boundary the property +draws, and assert silence: the commit that is still settling, the submit the app +refused, the card that scrolled into view rather than being created, the pair of +readings taken either side of a relaunch. A property you have only ever seen go +red is a property you have half tested. + +Then hand it to `sanderling-spec-review`, which will ask how many steps it +actually judged on a real run. diff --git a/skills/sanderling-run-triage/SKILL.md b/skills/sanderling-run-triage/SKILL.md new file mode 100644 index 0000000..1cc7583 --- /dev/null +++ b/skills/sanderling-run-triage/SKILL.md @@ -0,0 +1,211 @@ +--- +name: sanderling-run-triage +description: Work out what a finished sanderling run actually proves. Use before trusting a green run, before filing the bug a red run seems to show, and any time the exit code is the only thing anyone has looked at. +--- + +# Reading a run honestly + +A run produces one number that is easy to read and several that are worth +reading. The easy one says whether a process finished. It does not say whether +anything was checked, whether what was checked was your app, or whether the +violation it reports is about the app at all. + +Work through the sections in order and report what you established and what you +could not. "This run is not evidence, and here is the signal that says so" is a +complete and useful answer. + +## 1. The exit codes + +- **0** means the run completed. It does **not** mean no violations. Without + `--exit-on-violation` a run that recorded violations still exits 0: measured + on a ten step web run that recorded two, `run complete: 10 steps` and + `2 violation record(s)`, exit code 0. +- **2** means the run recorded a violation under `--exit-on-violation` and + stopped there. The same ten step run with the flag exits 2 after four steps. +- **1** means the harness broke. A bad target gives + `error: launch app: page load error net::ERR_UNSAFE_PORT` and exit 1, and + writes no run directory at all, because the trace is created after the launch + succeeds. + +Anything other than 0 and 2 means the run did not complete, and a missing +`trace.jsonl` under a 0 or a 2 means there is nothing to judge rather than +nothing to report. + +Exit 2 is not a conviction. `.github/scripts/folio-run.sh` is the worked example +worth reading in full: it exists because a thrown predicate reaches exit 2 by the +identical path a real conviction does, and so does a violation of a real but +unrelated property in the same spec. It sorts a trace's violations three ways, +by name and by `is_error`: convictions of the properties the leg gates on, +predicates that threw, and other real violations the leg has nothing to say +about. Do the same sort by hand before you call a 2 a finding. + +## 2. Reading a witness + +Witnesses live in `trace.jsonl`, one object per step under `witnesses`, keyed by +property name. A real conviction and a real throw from the same run: + +```json +{"step": 4, "violations": ["countStaysUnderThree"], + "witnesses": {"countStaysUnderThree": { + "reason": "predicate false", "step": 4, "detected_step": 4, + "extractors": {"count": 3}}}} + +{"step": 5, "violations": ["throwsOnceCountIsFour"], + "witnesses": {"throwsOnceCountIsFour": { + "reason": "Error: boom: no reading for this screen at :501:37(14)", + "is_error": true, "step": 5, "detected_step": 5, + "extractors": {"count": 4}}}} +``` + +`step` is where the failed obligation was armed and `detected_step` is where the +evaluation produced the violation; for a deferred obligation (a `next`, an +`eventually`) they differ, and `extractors` is `detected_step`'s state, not +`step`'s. + +The discipline is one sentence: open the witness and confirm those values could +actually produce that verdict. An iOS witness read `typedAmount = 0`, and +`submitChangesBalanceByAtMostTypedAmount` in +`examples/folio/sanderling/predicates.ts` returns true at `typedAmount === 0` +before it compares anything. So the trace +appeared to show a conviction that could not have happened. The verdict was real +and the artifact was lying, and until that was resolved neither the bug report +nor the fix could be trusted. + +When a witness value looks impossible, suspect the reading before you suspect +the property. Values reach a witness through the driver, and the driver can be +wrong in ways the spec cannot see: erasing a text field used to leave characters +behind, because a backspace only deletes to the left of the cursor and the +runner taps the field's centre, and 7 of 19 measured `InputText` observations +left residue that the spec then reasoned about as if it were the typed value. + +Two more things the witness tells you, if the spec extracts them. folio declares +`extract("lastAction", s => s.lastAction)` precisely so they land in the trace: +`applied: true` means the runner saw the dispatch succeed, `applied: null` means +it was dispatched and nobody knows whether it landed, and `relaunched: true` +means the app restarted between the two readings. A property attributing an +effect to an action of unknown fate is unsound; see `sanderling-spec-review`. + +Finally, a property violates once. After it fires, its residual stays `false` +(or `{"op": "error", ...}`) for every remaining step and it is never evaluated +again. Measured across steps 4 to 10 of that run, `countStaysUnderThree` reads +`{"op": "false"}` at every step after the first. So the violation count is a +count of distinct properties, not of occurrences, and everything after a +property's first violation is unchecked by that property. + +## 3. A green run fails in two ways + +Either it checked nothing, or it checked and the fuzzer never reached the bug. +These need opposite responses (fix the spec or the hooks; spend more budget or +better actions) and the exit code distinguishes neither. + +The first is not a hypothetical. Against an empty page, six steps, exit 0, `no +violations`, and `countNeverNegative` judged **0 of 6**: its extractor returned +null every step, its guard short-circuited, and its residual read `{"op": +"true"}` at every step, exactly as it reads when it compares real values. + +So count, per property, the steps where its guard passed and it compared +something (**judged**) against the steps where it returned true without +comparing anything (**declined**). `.github/scripts/replay-ui-summary.sh` does +this for the replay-ui spec and prints a judged/declined table for exactly this +reason. To do it by hand from a trace: + +- fold `extractor_changes` forward per step. Only extractors whose value changed + are recorded, so a step with no entry for an extractor means unchanged, not + absent. Measured: `{"count": {"prev": 0, "curr": 2}}` at one step and no + `count` entry at the next. +- skip steps carrying `skipped_verification` or `transitional`. They advance + nothing. +- apply each property's own guard to the folded values and count. + +That script also carries the honest warning about this technique: restating a +property's guard outside the property is a second copy that can drift, so it +checks that the trace's property names still match the ones it counts and that +the spec still declares the extractors it reads, and it fails loudly when either +moves. + +Do not try to read judged-versus-declined off `residuals`. `always(p)` residuals +to `{"op": "true"}` whether `p` compared real values or short-circuited, so the +two are indistinguishable there. + +## 4. A red run fails in two ways + +Either a property was proved false about the app, or a predicate threw. +`is_error` in the witness separates them and they mean opposite things. + +A conviction is a claim about the app. A throw is a claim about the spec, and it +is worse than an unhelpful result: the property is violated from that step on +whatever the app does, so it checks nothing for the rest of the run, and under +`--exit-on-violation` the run ended there so nothing past it was checked by +anything. The `reason` carries the JavaScript error and its location, which is +usually enough to find it: `Error: boom: no reading for this screen at +:501:37(14)`. + +The third case is a real violation of a property that is not the one you are +asking about. It is a finding, and it is somebody's bug, but the run has nothing +to say about the question you asked it. Name the property before you claim the +result. + +## 5. When a run is not evidence at all + +Some runs never got far enough for any of the above to matter, and every one of +them exits 0 and reports no violations. + +**It never reached the app.** A launch flake left a fuzzer on the device +launcher for 200 steps in 65 seconds, two nodes per snapshot, exit 0, no +violations (issue #81). The check is that the app's own marker appears in the +trace at all: the folio CI leg greps for `"AddTransactionScreen"` and fails the +run when it is absent, which is more honest than any exit code it could read. +The run's stdout also carries `app left foreground; relaunching` with the +package it found instead. + +**The hierarchy is a handful of nodes.** `nodes=` in each step line is the +cheapest signal there is. Measured: 6 on a four-element page, 2 on an empty one. +A run whose `nodes` never leaves single digits is looking at a launcher, a +crash screen, or a page that failed to boot. + +**It never left one screen.** `screen=` constant for the whole trace, or a +`route` extractor that never changes value. + +**Steps far faster than the run's own median.** Take the per-step deltas from +each step's `timestamp` and compare them against the run's median. A stretch of +steps at a fraction of it is a driver that is not waiting for an app, because +there is no app to wait for: 200 steps in 65 seconds is 325 ms a step, against +seconds a step for a run that is driving something real. + +**It spent its budget on one action.** Count `next_action` by kind and selector. +A run whose actions are one selector explored nothing, whatever its step count. + +None of these change the exit code. All of them change what the run proves, +which is nothing. + +## 6. `skipped_verification`, `transitional`, and the judged count + +The runner skips the verifier for a step whose hierarchy was still moving: an +Android NavHost mid cross-fade after the retry budget, or a hierarchy fetch that +failed or came back empty. Pushing such a tree would poison the previous/current +extractor advance and make the next clean step convict a healthy app, so the +step is recorded for replay and judged by nothing. `transitional` marks the +tree; `skipped_verification` is set exactly when the verifier was skipped. + +The run says so itself: + +``` +7 step(s) judged by nothing: the screen was still moving when it was read +``` + +Subtract it. `run complete: 240 steps` with that line is a 233 step run for +every purpose that matters, and `replay-ui-summary.sh` reports the pair as +"N steps recorded, M verified" for the same reason. A run with many of these is +telling you the driver could not get a clean read of your app, which is a +finding about the setup and worth chasing rather than quietly accepting the +smaller number. + +## Reporting + +For any run, report: the exit code and whether `--exit-on-violation` was set; +steps recorded against steps verified; per property, judged against declined; +for every violation, its `is_error` and the witness values you actually opened; +and which of the section 5 signals you checked. Name the step behind any claim. + +A run is evidence only for the properties that judged, and only for the app it +was actually looking at. Everything else it produced is a log. diff --git a/skills/sanderling-setup/SKILL.md b/skills/sanderling-setup/SKILL.md new file mode 100644 index 0000000..47d5402 --- /dev/null +++ b/skills/sanderling-setup/SKILL.md @@ -0,0 +1,269 @@ +--- +name: sanderling-setup +description: Get sanderling running against an app that is not folio. Use before writing a spec for a new app, when deciding what test hooks the app needs, and when a run will not start or starts and sees nothing. +--- + +# Getting sanderling onto your app + +The goal of setup is not a run that finishes. It is a run whose output you can +believe. Two things decide that, and both are usually treated as chores: the +handles the app exposes, and the state the app starts in. Everything else here +is plumbing. + +Every flag below is one the binary accepts, checked against `sanderling test -h` +on this revision. That command is the authority, not this file and not the +manual. Check before you use a flag you have not seen work. + +## 1. Install, then check the host + +The CLI installs from the release script, and the spec package from npm: + +```sh +curl -fsSL https://raw.githubusercontent.com/priyanshujain/sanderling/master/install.sh | bash +npm install --save-dev @sanderling/spec +``` + +Both come from the same release tag and the CLI bundles the package's TypeScript +when it evaluates your spec, so they move together. + +`sanderling doctor` reports the host's readiness per platform and exits non-zero +if anything is missing. Each line names the check and, on a failure, what to do +about it. On a Mac with the Android SDK installed but a CLI built by a plain +`go build`, `sanderling doctor --platform android` says: + +``` +OK adb on PATH or under the Android SDK +OK emulator on PATH or under the Android SDK +OK java 17+ on PATH +FAIL sidecar JAR is real (not placeholder): placeholder JAR embedded; run `make sidecar && make sanderling` to embed the real fat JAR +error: 1 check(s) failed +``` + +Scope it with `--platform web|android|ios|ios-device|all` (default `all`). Web +needs a Chromium that launches headless. Android needs `adb`, an emulator, Java +17 or newer, and the embedded sidecar JAR. iOS needs `xcrun` and `simctl`; +`ios-device` adds `devicectl`, the macOS usbmuxd socket, a connected paired +device, and App Store Connect signing credentials. + +The `adb` and `emulator` checks resolve through the same helpers a run uses, so +they search PATH, then `ANDROID_HOME` and `ANDROID_SDK_ROOT`, then +`~/Library/Android/sdk`, `~/Android/Sdk` and the Homebrew command-line-tools +paths. A host the doctor passes is a host a run can drive, and a failure names +every location it tried. What the doctor cannot tell you is the reverse: a +missing SDK can also surface during a run as `sidecar health check: context +deadline exceeded` about thirty seconds in, which names the symptom and not the +cause (issue #69). If you see it, go back to +`sanderling doctor --platform android` before believing anything about the +sidecar. + +Two traps if you build from source rather than installing a release. A plain +`go build ./cmd/sanderling` embeds a placeholder sidecar JAR, so every Android +run stops at `sidecar: binary built without -tags withsidecar`; `make sanderling` +(or `make sanderling-android`) embeds the real one. And `go run ./cmd/sanderling +test` collapses the process exit code: a run that exits 2 comes back from +`go run` as 1 with `exit status 2` printed. Use the built binary whenever the +exit code matters, which is always in CI. + +## 2. Point it at the app + +Android takes the applicationId, boots an AVD with `--avd`, and picks between +attached devices with `--device ` as `adb devices` prints it: + +```sh +sanderling test --spec spec.ts --bundle-id com.example.app --avd Pixel_7_API_34 +``` + +iOS takes `--platform ios` and `--ios-device`, which accepts a simulator name or +UDID, or a connected device's name, UDID, or CoreDevice id. `--ios-app-path` +points at the `.app` bundle and is what makes clear-state real; see section 4. + +Web takes a URL as the bundle id: + +```sh +sanderling test --spec spec.ts --platform web --bundle-id http://127.0.0.1:8799/index.html +``` + +The web target has to genuinely load. A page that boots to a blank canvas still +produces steps, still exits 0, and proves nothing: folio's own web leg needs +COOP/COEP headers or its sqlite worker never starts, which is why +`.github/scripts/folio-run.sh` serves the build itself instead of using a stock +static server. Confirm the app rendered before you read anything else. + +## 3. Test hooks are a prerequisite, not a polish step + +This is the part that decides whether a spec is possible at all. The header of +`replay-ui/sanderling/spec.ts` states it as the lesson it is: + +> The hooks it drives (data-testid, data-step, ...) were added to the UI for +> this spec. Needing them is the lesson: a UI with no stable handles is a UI +> nothing can assert on, and that is as true for a person writing a test as it +> is for a fuzzer. + +A fuzzer is not asking for anything a human test author does not need. It is +only less able to squint at a screenshot and guess. Budget the hooks as part of +adopting sanderling, before the spec, not after the first vacuous run. + +`testTag` is the portable name. `internal/hierarchy/hierarchy.go` aliases it to +`resource-id`, `identifier` and `accessibilityIdentifier`, so one selector +matches on every platform. What you have to add differs: + +**Compose on Android.** `Modifier.testTag("AddAccountSubmit")` alone does not +reach the accessibility tree. The tree only carries it when a root composable +sets `semantics { testTagsAsResourceId = true }`. folio does this once, at the +app root, through an expect/actual bridge: +`examples/folio/app/shared/src/androidMain/kotlin/app/folio/ui/TestTagBridge.android.kt`. +Without it every `testTag` selector matches nothing, every property over it +declines, and the run goes green having checked nothing. + +**Web.** `data-testid` is the hook. Every `data-*` attribute on the element +reaches the spec under `attrs`, camel-cased the way `dataset` does it, so +`data-step-count` reads as `attrs.stepCount`. That is how the replay-ui spec +reads a panel's own claim about which step it is showing rather than re-deriving +it. Hooks that carry a value, not just an identity, are what make cross-panel +agreement properties possible. + +**iOS.** `accessibilityIdentifier`, set via `.accessibilityIdentifier` in +SwiftUI or UIKit. Compose Multiplatform maps `testTag` to it for you. + +Two rules about the names themselves. On Android and iOS a `testTag` selector +falls through to a substring compare, so `{testTag: "Sub"}` matches +`AddAccountSubmit`; on web the same selector compiles to an exact CSS attribute +match and hits nothing. Make each hook a whole distinct name rather than a +fragment of another, and you are right on both. And give every screen a marker +of its own, because a route extractor is what lets a property decline on the +screens it has nothing to say about. + +The check that a hook exists is not that you added it. It is that you can point +at a step in a real trace where a selector over it resolved to a value. +`sanderling-spec-authoring` covers which hooks a spec needs and in what order to +add them; this section is about what each platform requires before any of that +reaches the tree. + +## 4. A run must start from a known state + +`--clear-data` defaults to true and is the difference between a repeatable run +and a measurement of your own leftovers. A second run that inherits the first +one's accounts, cache and completed onboarding diverges at step 1: the seed +reproduces nothing, the two runs' step counts are not comparable, and any number +you quote from the pair is noise. + +What "clear" reaches depends on the platform, and in two cases it silently +reaches less than you expect: + +- Android wipes app data through the sidecar. On OEM builds that deny + `pm clear`, pass `--android-app-path ` and it uninstalls and reinstalls + instead. +- iOS simulator without `--ios-app-path` resets the data container only and + prints `clear-state requested without an app path: resetting the data + container only`. With the path it does a full `simctl` uninstall and install. + The container wipe is a real reset and folio's own iOS leg relies on it; the + reinstall path is the one that races FrontBoard. +- iOS on a physical device without `--ios-app-path` does not clear at all. It + prints `clear-state on a physical device requires --ios-app-path for a + reinstall; skipping (state not cleared)` and carries on. A device run left on + the default flag inherits every previous run's data. +- Web clears cookies and the target origin's storage. It cannot touch your + backend. If your app's state lives on a server, reset it yourself between + runs. + +`--clear-data=false` is a legitimate choice in one situation: you have just +installed a fresh build, so the app is already in clear state and an in-run +reinstall would only add a failure mode. Outside that, a run that resumes is a +run you cannot repeat. + +## 5. The device does not have to be local + +Android talks to whatever adb server the environment names. +`ADB_SERVER_SOCKET=tcp:host:port` (or `tcp:port` for a server on this machine) +is read first, then the older `ANDROID_ADB_SERVER_ADDRESS` and +`ANDROID_ADB_SERVER_PORT` pair, then the loopback default. The CLI shells out to +`adb` and inherits it; the JVM sidecar resolves the same variables when it +attaches to a serial. + +Two things to get right. Pass `--device ` exactly as the remote server +reports it: with no serial the sidecar's target is a local `localhost:5555`, not +your remote device. And a serial that already looks like `host:port` is dialled +straight at adbd, bypassing any server, which is a different path with different +failure modes. A value the sidecar cannot parse fails the run rather than +falling back to loopback, and that is deliberate: emulator serials are numbered +per server, so a quiet fallback would drive whatever this machine calls +`emulator-5554` and report the results as the remote device's. + +## 6. What a first run prints + +A ten step web run, in full: + +``` +bundled spec: 16532 bytes (sha256=71375ed5bfc7) +bundled web spec: 33131 bytes (sha256=779bae3c8fee) +spec loaded into verifier +trace dir: runs/20260815-172356 +running for 1m30s or 10 steps, whichever comes first (seed=7) +step index=1 screen="/index.html" nodes=6 +... +step index=10 screen="/index.html" nodes=6 + +elapsed: 1.715s + +run complete: 10 steps +no violations. +``` + +`nodes=` is the first number to read and the cheapest lie detector you have. On +that page, four elements plus html and body gave `nodes=6`. The same command +against an empty page gives `nodes=2` for every step, and still exits 0 with no +violations. If `nodes` is a handful and never grows, the run is looking at +something that is not your app. + +`screen=` is web-only and has nothing to do with your spec's screen hooks: the +Chrome driver puts the URL hash, or the pathname when there is no hash, on the +root node, and only that driver writes the attribute. On Android and iOS it is +empty on every step, so an empty `screen=` there is the normal reading and not a +symptom. On web, a `screen=` that never changes means the run never left one +URL, which for a single-page app that routes in memory is also normal. Your +spec's own route extractor is the thing to trust on every platform. + +The summary can carry a third line you should never skim past: + +``` +7 step(s) judged by nothing: the screen was still moving when it was read +``` + +Those steps were recorded but no property judged them, so the run's step count +and its checked count are different numbers. `sanderling-run-triage` is about +what to do with that. + +Set `--max-steps` whenever you intend to compare two runs: a step budget is what +makes them comparable, since duration alone does not. `--seed` fixes the PRNG, +and seed 0 draws a random one and records it in `meta.json`. + +## 7. The run directory + +Each run writes `//`, containing `meta.json`, +`trace.jsonl`, and one PNG per step under `screenshots/`. `--output` defaults to +`./runs`. + +`meta.json` is the run's identity: seed, spec path, bundled spec sha256, +platform, bundle id, start and end times, generator, `max_steps`, +`duration_millis`, host, and the `--arm` label if you set one. Two runs that +differ in any of those are different runs and cannot be pooled. + +If the app never launched, there is no run directory at all: the launch error +comes before the trace is created. `error: launch app: ...` with nothing under +`./runs` means the run never began, which is a different thing from a run that +began and found nothing. + +Open a run with `sanderling replay `, which accepts either the parent runs +directory or a single run directory. + +## Reporting + +Say what you actually ran and what came back: the `doctor` output you got rather +than the one you expected, the exact `sanderling test` command, the step count +and the `nodes=` figure from the first run, and for each hook you added, the step +in a real trace where a selector over it resolved. Name what you could not +establish, particularly any platform you did not run on. + +Setup is finished when a property can be written that could fail. Write it with +`sanderling-spec-authoring`, review it with `sanderling-spec-review`, and read +the run it produces with `sanderling-run-triage`. diff --git a/skills/sanderling-spec-authoring/SKILL.md b/skills/sanderling-spec-authoring/SKILL.md new file mode 100644 index 0000000..e090d97 --- /dev/null +++ b/skills/sanderling-spec-authoring/SKILL.md @@ -0,0 +1,374 @@ +--- +name: sanderling-spec-authoring +description: Write a sanderling spec for an app: test hooks, extractors, selectors, properties, and an action tree that reaches the states the properties read. Use when adopting sanderling for a new app, when adding a property to an existing spec, and when a run is green because it never reached the state the property was written for. +--- + +# Writing a sanderling spec + +A spec is a TypeScript module the runner evaluates once per step. It exports +`properties` and `actionsRoot`, plus an optional `setup` and `generator`. +Writing one is easy. Writing one that would catch a real bug is not, because a +spec that checks nothing looks exactly like a spec that checks everything: both +are a green run. + +So the order below is arranged around getting evidence early that each piece +reads what you think it reads. When the spec is written, audit it with +`sanderling-spec-review` before trusting a green run from it. + +## Write it in this order + +Hooks, then extractors, then **one** property, then run it and read the witness, +then everything else. Writing six properties before the first run is how people +end up with six that cannot fire, and nothing in the output tells you which. + +## Hooks first + +Your app needs stable handles or nothing can name what it is asserting on. The +hooks the replay UI's spec drives (`data-testid`, `data-step`) were added to the +UI for that spec, and its header says why: a UI with no stable handles is a UI +nothing can assert on, and that is as true for a person writing a test as it is +for a fuzzer. + +Put a hook on the screen or route markers, on every container you will scope a +lookup to, and on every fact you will read. `testTag` is the portable one: it +surfaces as resource-id on Android and accessibilityIdentifier on iOS, and on +web it resolves to `data-testid` or `id`. + +## Extractors + +`extract(name, fn)` reads one fact off `state.ax` per step. `.current` is this +step's value, `.previous` the last step's, `undefined` on the first step. + +Name every one. The name is what you get back later: `extractor_changes` in +`trace.jsonl` carries the prev/curr pair for each extractor whose value moved, +and the witness recorded at a violation carries the extractor values behind it, +by name. An unnamed extractor shows up as `extractor_3`, which tells you nothing +at the point you most need to know what the property was looking at. + +The rule that decides whether the spec is worth anything: + +> **Return `null` when the element is absent. Never `0`, `""`, or `[]`.** + +An unreadable fact is unknown, and a default turns unknown into a claim. Both +directions bite. Folio parsed a missing balance as `0` and its property became +`Math.abs(0 - 0) === typedAmount`, false at every healthy submit. Read a missing +panel's row count as `0` and the property says the cart is empty when the truth +is that the cart is not on screen. Where the ambiguity is real, call it unknown: +an empty `findAll` is both "no rows" and "not drawn yet", and folio treats it as +unknown, which costs the very first account of a run and buys back every card +that arrived late. + +Extractors run before properties and action generators, and they may not read +each other. If two readings must come off one parse, put the parse in a helper +both call: `examples/folio/sanderling/predicates.ts` does this with +`oncePerFrame`, keyed on the state object, since both hosts build a new state +object per step. + +## Selectors + +`ax.find` and `ax.findAll` take a string (`"id:CartBadge"`), an object +(`{id: "CartBadge"}`), or an array of objects for a path. Element handles carry +their own `.find` / `.findAll` scoped to their subtree. + +The two forms resolve identically: an object key is matched by the same rule its +string form uses, so `{id: "X"}` and `"id:X"` can never pick different elements. +What differs is the rule per key. Measured against an Android dump holding +`com.app:id/CartBadge`, whose content-desc is `Cart, 3 items`, alongside +`AddAccountSubmit`: + +| Selector | Resolves to | +|---|---| +| `{id: "CartBadge"}` | the badge: `id` matches the whole resource-id, or the part after `:id/` | +| `{id: "Sub"}` | nothing: `id` wants a whole name, not a fragment | +| `{testTag: "CartBadge"}` | the badge: `testTag` reaches resource-id on Android and accessibilityIdentifier on iOS | +| `{testTag: "Sub"}` | `AddAccountSubmit`, because every key outside the `id` / `desc` / `descPrefix` special cases is a **substring** match | +| `{desc: "Cart"}` | the badge: `desc` takes the whole description, or an iOS merged label starting `Cart, ` | + +`testTag` is the portable key and the one to reach for, but name the element in +full. A substring match on `Sub` is not a match, it is a coincidence, and it +will one day pick a different control. + +A selector that matches nothing makes every property over it vacuous, and +nothing anywhere reports that. This is why the selector you verify is the one +you found a real value for in a witness, not the one that looked right when you +wrote it. + +**Scope the lookup to a container instead of taking the first match on the +page.** From `replay-ui/sanderling/spec.ts`: an earlier draft of +`screenshotShowsTheSelectedStep` took the first screenshot on the page, the +fuzzer put the "before" panel on another tab, which left the "after" panel's +image first, and the property fired against a UI that was behaving correctly. It +now reads `s.ax.find([{ "data-testid": "state-before" }, { "data-testid": "screenshot" }])` +and is scoped to the panel it means. + +A screen marker is not enough scope on its own during a navigation. Android's +hierarchy dump carries the outgoing and the incoming screen together on 425 of +1879 steps measured across 17 runs, better than one frame in five, so a find +scoped to a screen the app has already left still resolves. Decide the route once +per frame, return `null` when more than one screen marker is present, and have +every reading take its answer from there. `routeOfFrame` in folio's +`predicates.ts` is that rule and carries the measurements. + +## Properties + +`always(f)` requires `f` at every step. `next(f)` inside it compares this step +to the next, which is how you state "this action had that effect". `now(f)` +evaluates at the current step inside a formula body. +`eventually(f).within(n, "steps" | "seconds" | "milliseconds")` requires `f` +before the window closes and convicts at the step it does not. Unbounded, it +does not stop being a liveness obligation: one that never fires is violated when +the run ends, with the reason `eventually never satisfied`. So an `eventually` +over a state your run may not reach fires on every run that does not reach it, +and that is the usual way a first spec ends up red for no reason. At the top +level an `eventually` is one goal for the whole run, armed once and discharged +for good the first time it holds; written inside `always` it re-arms at every +step, which asks for the window to be met from everywhere. Every formula has +`.implies`, `.and`, `.or`, `.not`. + +The stock properties are in `@sanderling/spec/defaults`. Both are cheap and both +are narrower than their names suggest, so know which platform yours runs on. + +`noUncaughtExceptions` fails when `state.exceptions` is non-empty, and today only +the web runtime fills it: `pkg/spec/src/web-runtime.ts` installs `error` and +`unhandledrejection` listeners in the page. On Android and iOS nothing populates +the field, so it holds at every step whatever the app does. Export it on web, +where it is free and real; on native, understand that a green run says nothing +about crashes. + +`noLogcatErrors` fails on any log line the driver reports at level `E`, which is +where an uncaught Java or Kotlin throwable lands, so on Android it is the closest +thing to `noUncaughtExceptions`. It holds vacuously on web and iOS. Neither +platform has an equivalent today: an iOS crash is invisible to both properties. + +## What makes a good first property + +Prefer a **cross-panel agreement**: two parts of the UI that derive the same fact +by different paths must say the same thing. The toolbar prints a step count and +the list renders rows; a badge counts violation records and the panel counts the +rows it can show for them. Those hold on any run, so they never need +recalibrating against a fixture, and an app that gets the fact wrong in one of +the two places cannot satisfy them however it was driven there. +Three of the seven properties in `replay-ui/sanderling/spec.ts` are this shape: +`stepCountMatchesTheList`, `screenshotShowsTheSelectedStep` and +`badgeCountMatchesThePanel`. The rest of that spec shows what to write when no +second panel derives the fact: a range invariant on user input +(`selectedStepIsInRange`), a counting invariant inside one panel +(`exactlyOneStepIsSelected`), a no-effect property across an action +(`switchingTabsKeepsTheStep`), and the stock `noUncaughtExceptions`. All of them +still hold on any run, which is the property worth keeping. + +Contrast a property that needs the fuzzer to reach a specific state, like +folio's "a submit moves the balance by no more than the amount typed". That is where +the real bugs are, and it is the harder thing to keep honest: it needs an action +tree that reaches the state, a window that closes often enough to bound what +happened inside it, and attribution that cannot blame the wrong action. Folio's +counting form went 117 steps between two readings on one iOS run and gathered 37 +submits against a rise of 15 transactions, which is perfectly sound and says +nothing at all; the fix was to state the same rule over a number the app redraws +on nearly every frame, so the window is usually one action wide. Write these +second, and read `sanderling-spec-review` before you believe one. + +Whichever you write, name the input that makes it return false before you move +on. If you cannot, it is decoration. + +## Actions + +`actions(() => Action[])` returns the candidate actions for this step and the +picker chooses one. The verbs are `Tap`, `DoubleTap`, `LongPress`, `InputText`, +`Scroll`, `Swipe`, `PressKey`, and `Wait`. The built-in generators are `taps`, +`doubleTaps`, `longPresses`, `typing`, `scrolls`, `swipes`, `pressKeys`, and +`waitOnce`. `defaultActions` bundles five of them: taps and typing at 100, +scrolls 50, swipes 25, double taps 10. `longPresses`, `pressKeys` and `waitOnce` +are not in it, so a spec that only exports `defaultActions` never presses android +back, never long-presses, and never waits. Weight those in yourself if the app +has behaviour behind them. + +`weighted([n, generator], ...)` composes them with relative weights. +`whenRoute(routeExtractor, routes, body)` runs `body` only on the named screens. +The optional `setup` export runs before `actionsRoot` for as long as it returns +actions, which is where login and onboarding belong; it re-engages on its own if +the app logs itself out mid-run. Values come from `from(items)`, +`integers().between(min, max)`, `strings().length(min, max).alpha()`, +`emails().domain(host)`, and `edgeCaseText()`, all drawn from the run's seeded +PRNG so a seed replays exactly. + +**The default enumeration explores, but reaching a specific interesting state +usually needs a weighted action of your own.** With about 15 clickable elements +on the replay UI's page, an undirected run went 40 steps without switching a +single tab, which left both tab-facing properties vacuously true. Its badge +agreement is worse: it needs two readings on one step, a badge, which only a +step that has a violation renders, and a violations panel to compare it against. +Undirected actions put both on the same step **0 times in the 80 steps of the +first dogfood run**. The property was reachable in principle and judged nothing +in practice, and aiming at the step alone just moved the misses to the other +side, 0 judged either way. It took an action that selects a violating step and +then opens a panel if none is up. Folio weights its transaction chain at 45 for +the same reason: both balance properties observe that flow and nothing else +reaches it. + +So for every property, name the action in the tree that puts everything it reads +on screen at the same step. If there is none, add one, and give it enough weight +that a short run gets there. + +## Soundness outranks everything else here + +A property must never convict an app that behaved correctly. A property that +convicts more often and is sometimes wrong is strictly worse than one that +convicts less and is never wrong, because a false conviction costs someone a day +and then costs the whole suite its credibility. **When in doubt, a property +should decline to judge.** + +Declining costs at most a detection. Convicting a healthy app costs the spec. +Concretely that means unknown stays `null`, a bound is preferred to an equality +where the window can hold more than one cause, and a value carried across a +screen change is dropped rather than compared. + +It also means reading `state.lastAction` for what it actually promises, which is +three different things and not one: + +- `state.lastAction === null`: no action ran. +- `applied: true`: the runner saw the dispatch succeed. +- `applied: null`: it was dispatched and nobody knows whether it landed. + +The rule is short. **An action of unknown fate still counts toward bounds on +what the app could have done, but it never licenses attributing an effect to +it.** Leave it out of the bound and you convict a healthy app: folio saw a +transaction rise of one against a window of zero submits and called it a double +submit. Demand its effect and you convict the app of the runner's own +uncertainty. + +`relaunched: true` says the runner had to bring the app back to the foreground +after the action, so the two readings straddle a restart. The action still +happened and still counts toward the bound, but nothing about state running +continuously between the two readings survives it, and a property demanding that +action's effect has to decline. Like `applied`, its null is "not reported", not +"the app never restarted": web and iOS cannot read the foreground at all, so only +an explicit `true` licenses declining. + +## Run it, then read the witness + +**Write one property, run it, and read the witness before you write the second.** +This is the step that gets skipped and it is the one that pays. A spec that has +never had its readings confirmed against a real run looks exactly like a spec +that has, right up until you find out that an extractor reads `null` on the +platform you care about, or that a selector matches nothing, or that the value +being compared is not the value you thought. + +```sh +sanderling test --spec spec.ts --bundle-id com.example.app --platform android --duration 2m +sanderling replay +``` + +Confirm two things before adding anything. First, that each extractor holds a +real value at some step, by finding it in `extractor_changes` in `trace.jsonl`; +the replay UI's hierarchy panel separately tells you whether the element your +selector names is in the tree at all. Second, that the property actually +compared values on some step rather than short-circuiting on its own guard. A +green run is evidence only if you can point at a step where a property fired. + +Only then write the next property. + +Once a property is worth keeping, its logic is worth testing away from the +device. Folio keeps its predicates in a plain module and unit-tests them in +`pkg/spec/test/folio-*.test.ts`, run by `make test-spec-api`; the app's own +Kotlin tests are `make test-folio`. Those files are the model for testing a +property in isolation, including the direction people skip: a fixture where the +effect happens legitimately, asserting that the predicate stays silent. + +## A spec to adapt + +Complete and self-contained: a storefront whose header badge and cart panel both +know how many things are in the cart. + +```ts +import { InputText, Tap, actions, always, extract, from, integers, weighted } from "@sanderling/spec"; +import { defaultActions, noUncaughtExceptions } from "@sanderling/spec/defaults"; + +function wholeNumber(text: string | undefined): number | null { + if (!text) return null; + const parsed = Number(text.trim()); + return Number.isInteger(parsed) ? parsed : null; +} + +// The header badge: the app's own count of what is in the cart. +const badgeCount = extract("badgeCount", s => + wholeNumber(s.ax.find({ testTag: "CartBadge" })?.text)); + +// The same fact by another path: the rows the cart panel renders. No panel is +// null rather than 0, because nothing on screen is a fact we do not have, and +// 0 would claim the cart is empty. +const cartRowCount = extract("cartRowCount", s => { + const panel = s.ax.find({ testTag: "CartPanel" }); + return panel ? panel.findAll({ testTag: "CartRow" }).length : null; +}); + +const checkoutEnabled = extract("checkoutEnabled", s => { + const button = s.ax.find({ testTag: "CheckoutButton" }); + return button ? button.enabled === true : null; +}); + +// Two parts of the UI count the cart by different routes through the app's own +// state, so they cannot disagree about how many things are in it. +const badgeMatchesTheCart = always(() => { + const badge = badgeCount.current; + const rows = cartRowCount.current; + if (badge === null || rows === null) return true; + return badge === rows; +}); + +// Checkout is offered exactly when there is something to check out. +const emptyCartCannotCheckOut = always(() => { + const rows = cartRowCount.current; + const enabled = checkoutEnabled.current; + if (rows === null || enabled === null) return true; + return rows > 0 || !enabled; +}); + +export const properties = { + noUncaughtExceptions, + badgeMatchesTheCart, + emptyCartCannotCheckOut, +}; + +const productCards = extract("productCards", s => s.ax.findAll({ testTag: "ProductCard" })); +const cartButton = extract("cartButton", s => s.ax.find({ testTag: "CartButton" })); +const quantityField = extract("quantityField", s => + s.ax.find([{ testTag: "CartPanel" }, { testTag: "QuantityField" }])); + +const addAProduct = actions(() => { + const cards = productCards.current; + return cards.length === 0 ? [] : [Tap({ on: from(cards).generate() })]; +}); + +const openTheCart = actions(() => { + const button = cartButton.current; + return button ? [Tap({ on: button })] : []; +}); + +const quantities = integers().between(1, 5); + +const changeAQuantity = actions(() => { + const field = quantityField.current; + return field ? [InputText({ into: field, text: String(quantities.generate()) })] : []; +}); + +// Both properties read the cart panel, so a run that never opens it judges +// nothing. defaultActions carries the rest of the app. +export const actionsRoot = weighted( + [35, addAProduct], + [25, openTheCart], + [15, changeAQuantity], + [25, defaultActions], +); +``` + +Both properties here decline whenever the panel is off screen, which is honest +and also the thing to measure first: if `openTheCart` never wins the draw, they +judge nothing, exactly like the replay UI's badge property did for 80 steps. + +The two real specs in the repo are the fuller references. +`replay-ui/sanderling/spec.ts` is the cross-panel spec written the way this page +recommends. `examples/folio/sanderling/spec.ts` with its `predicates.ts` is the +harder kind, a spec that attributes effects to actions across screens, and every +comment in it records a way it was once wrong. `docs/manual/spec-language.md` is +the lookup reference for anything not covered here. diff --git a/skills/sanderling-spec-review/SKILL.md b/skills/sanderling-spec-review/SKILL.md new file mode 100644 index 0000000..4307afa --- /dev/null +++ b/skills/sanderling-spec-review/SKILL.md @@ -0,0 +1,157 @@ +--- +name: sanderling-spec-review +description: Review a sanderling spec for properties that cannot fail, cannot pass, or convict a healthy app. Use before trusting any spec, after any spec change, and whenever a run is green but you are not sure it checked anything. +--- + +# Reviewing a sanderling spec + +A spec that is wrong does not look wrong. It looks like a passing run. Every +failure below was found in a real spec that had been green for weeks, and each +was caught by reading a witness rather than an exit code. + +Work through the checks in order. Report what you actually verified and what you +could not; a review that says "I could not establish this" is worth more than one +that implies coverage it did not check. + +## 1. Can each property ever fail? + +For every property, find the input that makes it return false, and say what it is. +If you cannot name one, the property is decoration. + +The common shape is a guard that short-circuits on absent elements: + +```ts +const badgeMatchesPanel = always(() => { + const badges = violationBadges.current; + const panels = panelCounts.current; + if (badges.length === 0 || panels.length === 0) return true; // declines + return panels.every((c) => c === badges[0]); +}); +``` + +That guard is correct in isolation: with nothing on screen there is nothing to +disagree about. It is also how a property judges zero steps in an eighty step run +and reports success. Measured on a real run, that exact property judged **0 of 80 +steps** while the job went green. + +So counting matters. For each property, count the steps where its guard passed +and it actually compared values (**judged**) against the steps where it returned +true without comparing anything (**declined**). A property that judged nothing +proved nothing, whatever the exit code said. + +You can reconstruct this from a trace: fold `extractor_changes` forward per step +to recover each extractor's value, then apply the property's own guard. Do not +try to read it from `residuals`: `always(p)` residuals back to `{"op":"true"}` +whether `p` compared real values or short-circuited, so the two are +indistinguishable there. + +## 2. Can each property ever pass? + +The mirror failure. A missing fact read as a value instead of as unknown turns a +property into one that fires on every healthy run. + +```ts +// balances parse to 0 when the element is missing +Math.abs(currBalance - prevBalance) === typedAmount // 0 - 0 === typed, always +``` + +Check every extractor: does it return `null` when the element is absent, or does +it return `0`, `""`, or `[]`? An unreadable fact is unknown, never a default. + +## 3. Would it convict an app that behaved correctly? + +This is the only unforgivable failure. A property that convicts more often and is +sometimes wrong is strictly worse than one that convicts less and is never wrong, +because a false conviction costs someone a day and then costs the whole suite its +credibility. + +Test both directions for every property, always: + +- it fires on the bug it exists to catch +- it stays silent on a run where the app behaved + +The second test is the one that matters and the one people skip. Build a fixture +where the effect happens legitimately and assert silence. + +## 4. Is the attribution sound? + +When a property blames an effect on an action, check it cannot blame the wrong one. + +**Identity keys must be injective.** If two distinct objects can produce the same +key, a value can jump between unrelated series without anything noticing. Merged +UI text is the usual culprit: an account named `Travel1` with 25 transactions and +one named `Travel12` with 5 can both render `TRTravel125 transactions`. No +function of that string can separate them. + +**Match whole keys, not endings or substrings.** `endsWith` attribution judges an +older account named `Emergency Fund` when the user typed `Fund`. + +Selector matching has the same trap on Android and iOS, and it is easy to miss +which keys carry it. `id`, `desc` and `descPrefix` resolve by rules of their own +(exact or `:id/`-suffixed, exact or comma-prefixed, starts-with), so +`{id: "Sub"}` correctly matches nothing. Every other key, `text` and `testTag` +included, falls through to a substring compare, so `{testTag: "Sub"}` matches +`AddAccountSubmit`. That is not a match, it is a coincidence, and a property +built on it judges whichever element happens to contain the fragment. + +The web path does not share the rule, which is its own trap. `web-runtime.ts` +compiles an object selector to CSS, and every key becomes an exact attribute +match (`descPrefix` alone becomes a `^=` prefix). So the loose selector that +resolved on Android resolves to nothing on web, and every property over it goes +vacuously true rather than red. Reviewing a cross-platform spec means checking +that each selector is exact enough for native and literal enough for web. + +**Drop the carrier when the screen changes.** A value carried across a route +change is a value read from a screen that is no longer there. + +## 5. Are the windows bounded? + +A property that compares two readings and counts actions between them is only as +good as how often it closes the window. + +Real numbers from a real leg: a run went from step 19 to step 136 without +returning to the screen the property read, so it saw a rise of 15 against a window +of 37 actions. 15 is not more than 37, so nothing was reported, and the same run +also gave 4 against 7, 6 against 13, and 1 against 1. The property was sound the +whole time and detected nothing. + +Two fixes, and prefer the first: + +- **Close the window more often.** Read the fact somewhere the run visits often, + not somewhere it visits rarely. +- **Do not spend budget on actions that cannot have caused anything.** If the + submit button is disabled when the field is empty, a tap on it committed + nothing and must not count. That one change halved the window on a real spec. + +Prefer an **upper bound** to an equality. `|delta| > typedAmount` is sound where +`|delta| === typedAmount` convicts a commit still in flight, a refused submit, and +a tap that never landed. + +## 6. Does every selector actually resolve? + +A selector that matches nothing makes every property over it vacuous, and nothing +reports it. Verify by finding a step whose witness holds a real value for it, not +by reading the selector and believing it. + +## 7. Does the property know what the runner could not promise? + +`state.lastAction` distinguishes three things, and a property that collapses them +is unsound: + +- `null` means no action ran +- `applied: true` means the runner saw the dispatch succeed +- `applied: null` means it was dispatched and nobody knows whether it landed + +An action of unknown fate still counts toward **bounds on what the app could have +done**, and never licenses attributing an effect **to** it. `relaunched: true` +says the app restarted between two readings, so a property assuming continuous +state must decline. + +## Reporting + +For each property give: can it fail, can it pass, does it convict a healthy app, +how many steps it judged on a real run, and what you could not check. Name the +step and the witness values behind any claim that a property works. + +A green run is evidence only if you can point at a step where a property actually +fired. Read the witness, not the exit code. diff --git a/test/browser/browser_test.go b/test/browser/browser_test.go index 6c9bc9e..e1c4184 100644 --- a/test/browser/browser_test.go +++ b/test/browser/browser_test.go @@ -26,6 +26,7 @@ import ( "github.com/priyanshujain/sanderling/internal/bundler" "github.com/priyanshujain/sanderling/internal/driver" "github.com/priyanshujain/sanderling/internal/driver/chrome" + "github.com/priyanshujain/sanderling/internal/hierarchy" chromerunner "github.com/priyanshujain/sanderling/internal/runner" "github.com/priyanshujain/sanderling/internal/trace" "github.com/priyanshujain/sanderling/internal/verifier" @@ -275,3 +276,86 @@ func TestBrowserUncaughtExceptionReachesTheTrace(t *testing.T) { t.Errorf("exception recorded without class or message: %+v", recorded[0]) } } + +// One page, one selector, two hosts, two answers. +// +// The V8 host resolves state.ax.find against the live DOM; the goja host +// resolves the same selector against the hierarchy dump. The fixture puts a +// shadow-hosted #x above a light-DOM #x, the one shape where the two walks can +// disagree, and on web it is V8's answer that reaches the properties. Each host +// is driven through its production path (EvaluateExtractors in the page, +// PushSnapshot over the dump) and neither is asked what the other said, so the +// comparison is evidence rather than an assertion about one of them. +func TestBrowserAxFindAgreesAcrossHosts(t *testing.T) { + server := httptest.NewServer(http.FileServer(http.Dir(testdataDir(t)))) + t.Cleanup(server.Close) + + gojaBundle, webBundle := bundleSpec(t, filepath.Join(testdataDir(t), "find-order", "spec.ts")) + + driverInstance := chrome.New() + t.Cleanup(func() { + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + _ = driverInstance.Terminate(ctx) + }) + ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second) + defer cancel() + + if err := driverInstance.Launch(ctx, server.URL+"/find-order/", false, nil); err != nil { + t.Fatalf("launch: %v", err) + } + if err := driverInstance.InstallBundle(ctx, webBundle); err != nil { + t.Fatalf("install web bundle: %v", err) + } + readings, err := driverInstance.EvaluateExtractors(ctx) + if err != nil { + t.Fatalf("evaluate extractors in the page: %v", err) + } + fromV8 := string(readings[0]) + + dump, err := driverInstance.Hierarchy(ctx) + if err != nil { + t.Fatalf("hierarchy: %v", err) + } + tree, err := hierarchy.Parse(dump) + if err != nil { + t.Fatalf("parse hierarchy: %v", err) + } + verifierInstance, err := verifier.New( + verifier.WithSeed(fixtureSeed), + verifier.WithPlatform("web"), + ) + if err != nil { + t.Fatalf("verifier: %v", err) + } + if err := verifierInstance.Load(string(gojaBundle)); err != nil { + t.Fatalf("load spec: %v", err) + } + if err := verifierInstance.PushSnapshot(verifier.SnapshotInput{Tree: tree}); err != nil { + t.Fatalf("push snapshot: %v", err) + } + if count := verifierInstance.ExtractorCount(); count != 1 { + t.Fatalf("the goja host registered %d extractors, want the fixture's 1", count) + } + // ChangedExtractors omits an extractor whose reading is null and unchanged, + // which is exactly what a selector resolving nothing produces. With the + // count checked above, an absent entry is a null reading and not a missing + // extractor, so reporting it as null names the real failure. + fromGoja := "null" + if change, ok := verifierInstance.ChangedExtractors()["found"]; ok { + fromGoja = string(change.Curr) + } + + // Pinned, not just compared: two hosts that both resolved nothing would + // agree on undefined and prove nothing about the walk. + if fromGoja != `"shadow"` { + t.Errorf("the goja host read %s off the dump, want the shadow-hosted %q", fromGoja, "shadow") + } + if fromV8 != fromGoja { + t.Fatalf( + "one page, one selector, two answers: the V8 host read %s and the goja host read %s", + fromV8, + fromGoja, + ) + } +} diff --git a/test/browser/console_levels_test.go b/test/browser/console_levels_test.go new file mode 100644 index 0000000..f977dc9 --- /dev/null +++ b/test/browser/console_levels_test.go @@ -0,0 +1,215 @@ +//go:build browser + +package browser_test + +import ( + "context" + "net/http" + "net/http/httptest" + "path/filepath" + "slices" + "strings" + "testing" + "time" + + "github.com/priyanshujain/sanderling/internal/driver" + "github.com/priyanshujain/sanderling/internal/driver/chrome" + "github.com/priyanshujain/sanderling/internal/hierarchy" + "github.com/priyanshujain/sanderling/internal/verifier" +) + +// TestBrowserConsoleErrorReachesTheSpec drives a page that calls console.error +// and follows the entry the whole way a run does: the driver's log fetch at the +// runner's minimum level, then into the verifier state a property reads. The +// default noLogcatErrors counts entries whose level is "E", so a driver that +// spells the level any other way leaves the property permanently satisfied on +// web with nothing reporting that it never saw anything. +func TestBrowserConsoleErrorReachesTheSpec(t *testing.T) { + ctx, driverInstance, since := launchConsoleFixture(t) + + entries, err := driverInstance.RecentLogs(ctx, since, "E") + if err != nil { + t.Fatalf("recent logs: %v", err) + } + if len(entries) != 2 { + t.Fatalf("the runner's error-level fetch returned %d entries, want the page's two console.error calls: %+v", len(entries), entries) + } + for _, entry := range entries { + if entry.Level != "E" { + t.Errorf("console.error %q arrived as level %q, want %q", entry.Message, entry.Level, "E") + } + } + // console.error(err) is the ordinary way a page reports a failure, and the + // argument is then an object rather than a string. An entry that arrives + // with the right level and no message names nothing a reader can act on. + if !messageSeen(entries, "boom from the page") { + t.Errorf("the page's console.error string never arrived: %+v", entries) + } + if !messageSeen(entries, "object arg detail") { + t.Errorf("console.error(new Error(...)) arrived with no message: %+v", entries) + } + + dump, err := driverInstance.Hierarchy(ctx) + if err != nil { + t.Fatalf("hierarchy: %v", err) + } + tree, err := hierarchy.Parse(dump) + if err != nil { + t.Fatalf("parse hierarchy: %v", err) + } + + gojaBundle, _ := bundleSpec(t, filepath.Join(testdataDir(t), "console-levels", "spec.ts")) + verifierInstance, err := verifier.New( + verifier.WithSeed(fixtureSeed), + verifier.WithPlatform("web"), + ) + if err != nil { + t.Fatalf("verifier: %v", err) + } + if err := verifierInstance.Load(string(gojaBundle)); err != nil { + t.Fatalf("load spec: %v", err) + } + if err := verifierInstance.PushSnapshot(verifier.SnapshotInput{ + Tree: tree, + Logs: asVerifierLogs(entries), + }); err != nil { + t.Fatalf("push snapshot: %v", err) + } + verifierInstance.EvaluateProperties() + if violated := verifierInstance.NewlyViolatedProperties(); !slices.Contains(violated, "noLogcatErrors") { + t.Fatalf("a console.error on the page left noLogcatErrors satisfied; violations=%v", violated) + } +} + +// TestBrowserConsoleErrorFiresTheLogProperty drives the same page through the +// whole bundle -> run -> verify pipeline instead of hand-assembling a snapshot. +// The driver holding the entry is not enough on web: every extractor reading is +// replaced by the one the page computed, so state.logs is whatever the page +// says it is, and a page that answers "no logs" leaves noLogcatErrors green on +// a run whose console was full of errors. +func TestBrowserConsoleErrorFiresTheLogProperty(t *testing.T) { + violations := runFixture(t, "console-levels") + if !slices.Contains(violations, "noLogcatErrors") { + t.Fatalf("a page calling console.error ran a whole run without noLogcatErrors firing; violations=%v", violations) + } +} + +// TestBrowserQuietPageKeepsTheLogPropertySatisfied is the other half: a page +// whose console never reaches the error level must leave noLogcatErrors alone, +// so the property is reporting what the page logged rather than being on +// whenever the run is web. +func TestBrowserQuietPageKeepsTheLogPropertySatisfied(t *testing.T) { + violations := runFixture(t, "console-quiet") + if slices.Contains(violations, "noLogcatErrors") { + t.Errorf("noLogcatErrors fired on a page that logged nothing at error level; violations=%v", violations) + } + if !slices.Contains(violations, "counterNeverMoves") { + t.Fatalf("nothing was ever pressed, so the run proves nothing about a property that can fire; violations=%v", violations) + } +} + +// TestBrowserConsoleLevelsMapToTheLogcatScale pins what each console verb +// becomes once it crosses the driver. driver.LogEntry.Level is the single-letter +// logcat scale on every platform, so a spec asking for warnings or debug lines +// by letter has to get the same answer on web as it does on Android, and a +// console verb the driver has no mapping for still has to arrive rather than be +// silently discarded. +func TestBrowserConsoleLevelsMapToTheLogcatScale(t *testing.T) { + ctx, driverInstance, since := launchConsoleFixture(t) + + entries, err := driverInstance.RecentLogs(ctx, since, "V") + if err != nil { + t.Fatalf("recent logs: %v", err) + } + if len(entries) != 7 { + t.Fatalf("the page made 7 console calls, the driver kept %d: %+v", len(entries), entries) + } + + wantLevels := map[string]string{ + "boom from the page": "E", + "a warning": "W", + "a plain log": "I", + "a debug line": "D", + "an info line": "I", + } + seen := map[string]bool{} + for _, entry := range entries { + if !slices.Contains([]string{"V", "D", "I", "W", "E", "F"}, entry.Level) { + t.Errorf("entry %q carries level %q, which is not on the logcat scale a spec compares against", entry.Message, entry.Level) + } + for message, level := range wantLevels { + if !strings.Contains(entry.Message, message) { + continue + } + seen[message] = true + if entry.Level != level { + t.Errorf("console message %q arrived as level %q, want %q", message, entry.Level, level) + } + } + } + for message := range wantLevels { + if !seen[message] { + t.Errorf("console message %q never reached the driver's log buffer", message) + } + } + + errorsOnly, err := driverInstance.RecentLogs(ctx, since, "E") + if err != nil { + t.Fatalf("recent logs: %v", err) + } + if len(errorsOnly) != 2 { + t.Fatalf("the error-level fetch kept %d of the 7 entries, want only the console.error calls: %+v", len(errorsOnly), errorsOnly) + } + warningsUp, err := driverInstance.RecentLogs(ctx, since, "W") + if err != nil { + t.Fatalf("recent logs: %v", err) + } + if len(warningsUp) != 3 { + t.Fatalf("the warning-level fetch kept %d entries, want the console.error calls and the console.warn: %+v", len(warningsUp), warningsUp) + } +} + +// launchConsoleFixture serves the console-levels page and drives headless Chrome +// to it, returning the driver plus the instant before the page ran so a log +// fetch can ask for everything the page emitted. +func launchConsoleFixture(t *testing.T) (context.Context, *chrome.Driver, time.Time) { + t.Helper() + + server := httptest.NewServer(http.FileServer(http.Dir(testdataDir(t)))) + t.Cleanup(server.Close) + + driverInstance := chrome.New() + t.Cleanup(func() { + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + _ = driverInstance.Terminate(ctx) + }) + + ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second) + t.Cleanup(cancel) + + since := time.Now() + if err := driverInstance.Launch(ctx, server.URL+"/console-levels/", false, nil); err != nil { + t.Fatalf("launch: %v", err) + } + return ctx, driverInstance, since +} + +func messageSeen(entries []driver.LogEntry, want string) bool { + return slices.ContainsFunc(entries, func(entry driver.LogEntry) bool { + return strings.Contains(entry.Message, want) + }) +} + +func asVerifierLogs(entries []driver.LogEntry) []verifier.LogEntry { + out := make([]verifier.LogEntry, 0, len(entries)) + for _, entry := range entries { + out = append(out, verifier.LogEntry{ + UnixMillis: entry.UnixMillis, + Level: entry.Level, + Tag: entry.Tag, + Message: entry.Message, + }) + } + return out +} diff --git a/test/browser/testdata/console-levels/index.html b/test/browser/testdata/console-levels/index.html new file mode 100644 index 0000000..1683068 --- /dev/null +++ b/test/browser/testdata/console-levels/index.html @@ -0,0 +1,25 @@ + + + + + console-levels + + + +
0
+ + + diff --git a/test/browser/testdata/console-levels/spec.ts b/test/browser/testdata/console-levels/spec.ts new file mode 100644 index 0000000..e4a513d --- /dev/null +++ b/test/browser/testdata/console-levels/spec.ts @@ -0,0 +1,6 @@ +import { taps } from "@sanderling/spec"; +import { noLogcatErrors } from "@sanderling/spec/defaults/properties"; + +export const properties = { noLogcatErrors }; + +export const actionsRoot = taps; diff --git a/test/browser/testdata/console-quiet/index.html b/test/browser/testdata/console-quiet/index.html new file mode 100644 index 0000000..2020cf6 --- /dev/null +++ b/test/browser/testdata/console-quiet/index.html @@ -0,0 +1,23 @@ + + + + + console-quiet + + + +
0
+ + + diff --git a/test/browser/testdata/console-quiet/spec.ts b/test/browser/testdata/console-quiet/spec.ts new file mode 100644 index 0000000..842a8f4 --- /dev/null +++ b/test/browser/testdata/console-quiet/spec.ts @@ -0,0 +1,15 @@ +import { always, extract, taps } from "@sanderling/spec"; +import { noLogcatErrors } from "@sanderling/spec/defaults/properties"; + +const presses = extract((s) => { + const el = s.ax.find({ id: "count" }); + return el ? parseInt(el.text, 10) || 0 : 0; +}).named("presses"); + +// The run has to actually be driving the page, or noLogcatErrors staying +// satisfied says nothing about whether it can fire at all. +const counterNeverMoves = always(() => presses.current === 0); + +export const properties = { noLogcatErrors, counterNeverMoves }; + +export const actionsRoot = taps; diff --git a/test/browser/testdata/find-order/index.html b/test/browser/testdata/find-order/index.html new file mode 100644 index 0000000..9e81425 --- /dev/null +++ b/test/browser/testdata/find-order/index.html @@ -0,0 +1,16 @@ + + + + +
+ light + + + diff --git a/test/browser/testdata/find-order/spec.ts b/test/browser/testdata/find-order/spec.ts new file mode 100644 index 0000000..2145321 --- /dev/null +++ b/test/browser/testdata/find-order/spec.ts @@ -0,0 +1,17 @@ +import { always, extract, taps } from "@sanderling/spec"; + +// Which of the two #x a selector means is the whole fixture. The reading is +// compared between the two hosts by TestBrowserAxFindAgreesAcrossHosts, which +// drives each host's production path and never asks one what the other said. +// +// Written in the object form on purpose. It used to reach the goja host as a +// plain attribute filter looking for an `id` attribute a dump never carries +// (the web dump files the DOM id under resource-id), so it resolved nothing +// there while the V8 host resolved it against the live DOM. +const found = extract((s) => s.ax.find({ id: "x" })?.text).named("found"); + +const findsTheShadowMatch = always(() => found.current === "shadow"); + +export const properties = { findsTheShadowMatch }; + +export const actionsRoot = taps;