mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 11:07:10 +00:00
Merge origin/master into llm-recording-and-analysis
Brings in #73, which landed web fact-parity work overlapping this branch: selector-tagged ax handles, aria-disabled in `enabled`, shadow-DOM traversal and a `scrollable`/`editable` dump, plus a third folio property and two new testTags on HomeScreen. Six files conflicted. The option-carrying ones took the union of both sides' fields, so `--label-source` and `--exit-on-violation` both reach the pipeline. In web-runtime.ts both sides changed how an element is described: master gave `elementHandle` a selector to tag handles with, this branch gave it raw attribute names and a field hint. Both survive, and `enabled` now answers through master's isEnabled while `editable` stays. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
commit
d987526e47
73 files changed
+6587
-405
No files matched your search
Executable
+217
@@ -0,0 +1,217 @@
|
||||
#!/usr/bin/env bash
|
||||
# Runs examples/folio/sanderling/spec.ts against one platform and checks the
|
||||
# exit code the job expects. Kept out of the workflow YAML so it can be run by
|
||||
# hand, which is how it was calibrated:
|
||||
#
|
||||
# SEED=9 MAX_STEPS=200 .github/scripts/folio-run.sh android
|
||||
#
|
||||
# web and ios expect a conviction: folio's double-submit bug is still there, and
|
||||
# a run that no longer finds it is a regression in the fuzzer, not a pass. The
|
||||
# exit code alone does not say that much, so this script reads the trace: see
|
||||
# GATED_PROPERTIES below. android is a health gate instead, because it convicts
|
||||
# in four runs out of five rather than five; see docs/development/ci.md.
|
||||
set -uo pipefail
|
||||
|
||||
platform="${1:?usage: folio-run.sh android|ios|web}"
|
||||
seed="${SEED:-1}"
|
||||
max_steps="${MAX_STEPS:-240}"
|
||||
duration="${DURATION:-20m}"
|
||||
sanderling="${SANDERLING:-./bin/sanderling}"
|
||||
output="runs/folio-$platform"
|
||||
spec="examples/folio/sanderling/spec.ts"
|
||||
summary="${GITHUB_STEP_SUMMARY:-/dev/null}"
|
||||
|
||||
folio_args=(--bundle-id app.folio)
|
||||
case "$platform" in
|
||||
android)
|
||||
folio_args+=(--android-app-path
|
||||
examples/folio/app/androidApp/build/outputs/apk/debug/androidApp-debug.apk)
|
||||
;;
|
||||
ios)
|
||||
# --clear-data=false because the caller has just installed a fresh build (a
|
||||
# freshly installed app IS clear state). The in-run reinstall path is worth
|
||||
# avoiding here: `simctl uninstall` + `install` immediately followed by the
|
||||
# XCTest runner's own launch hits "app.folio is unknown to FrontBoard"
|
||||
# perhaps half the time. The launch RPC is bounded now, so that surfaces
|
||||
# as an error rather than an indefinite hang, but a failed leg is still a
|
||||
# failed leg and a fresh install is already clear state.
|
||||
folio_args+=(--platform ios
|
||||
--clear-data=false
|
||||
--ios-device "${IOS_DEVICE:-iPhone 16 Pro}")
|
||||
;;
|
||||
web)
|
||||
dist="examples/folio/app/webApp/build/dist/wasmJs/developmentExecutable"
|
||||
port="${PORT:-8791}"
|
||||
# The stock static servers do not set COOP/COEP, and without cross-origin
|
||||
# isolation the app's sqlite worker never starts, so folio loads to a blank
|
||||
# canvas and every step observes an empty accessibility tree.
|
||||
python3 - "$dist" "$port" <<'PY' &
|
||||
import functools, http.server, sys
|
||||
|
||||
class Isolated(http.server.SimpleHTTPRequestHandler):
|
||||
def end_headers(self):
|
||||
self.send_header("Cross-Origin-Opener-Policy", "same-origin")
|
||||
self.send_header("Cross-Origin-Embedder-Policy", "require-corp")
|
||||
self.send_header("Cross-Origin-Resource-Policy", "cross-origin")
|
||||
super().end_headers()
|
||||
|
||||
def log_message(self, *args):
|
||||
pass
|
||||
|
||||
directory, port = sys.argv[1], int(sys.argv[2])
|
||||
handler = functools.partial(Isolated, directory=directory)
|
||||
http.server.HTTPServer(("127.0.0.1", port), handler).serve_forever()
|
||||
PY
|
||||
server_pid=$!
|
||||
trap 'kill "$server_pid" 2>/dev/null' EXIT
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf "http://127.0.0.1:$port/index.html" >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "folio/web: the app server never served index.html on 127.0.0.1:$port from $dist" >&2
|
||||
exit 1
|
||||
fi
|
||||
folio_args=(--platform web --bundle-id "http://127.0.0.1:$port/index.html")
|
||||
;;
|
||||
*)
|
||||
echo "unknown platform: $platform" >&2
|
||||
exit 64
|
||||
;;
|
||||
esac
|
||||
|
||||
"$sanderling" test \
|
||||
--spec "$spec" \
|
||||
"${folio_args[@]}" \
|
||||
--duration "$duration" \
|
||||
--max-steps "$max_steps" \
|
||||
--seed "$seed" \
|
||||
--exit-on-violation \
|
||||
--output "$output"
|
||||
code=$?
|
||||
|
||||
run_dir="$(ls -d "$output"/*/ 2>/dev/null | tail -1)"
|
||||
trace="${run_dir:-.}/trace.jsonl"
|
||||
steps=0
|
||||
[ -f "$trace" ] && steps=$(wc -l < "$trace" | tr -d ' ')
|
||||
|
||||
# The two properties that state folio's double-submit. Anything else the spec
|
||||
# proves false is a different finding, and this leg has nothing to say about it.
|
||||
GATED_PROPERTIES="submitMovesBalanceByTypedAmount,submitCommitsOneTransactionPerAction"
|
||||
|
||||
# Exit 2 means "the run recorded a violation", and that is NOT the same as "the
|
||||
# run convicted folio". A predicate that THROWS is recorded as a violation too,
|
||||
# with is_error set and the thrown text as its reason, and it reaches exit 2 by
|
||||
# the identical path. So does a violation of newAccountBalanceIsZero, a real but
|
||||
# unrelated property in the same spec, which one android seed produces. Reading
|
||||
# the exit code alone leaves the gate green with detection dead.
|
||||
#
|
||||
# So sort the trace's violations three ways, by name and by is_error:
|
||||
# line 1 convictions: a gated property that was proved false
|
||||
# line 2 thrown: a predicate that blew up, with the reason it gave
|
||||
# line 3 other: a real violation of some property this leg does not gate on
|
||||
classified=$(TRACE="$trace" GATED="$GATED_PROPERTIES" python3 - <<'PY'
|
||||
import json, os
|
||||
|
||||
gated = set(os.environ["GATED"].split(","))
|
||||
convictions, thrown, other = [], [], []
|
||||
try:
|
||||
lines = open(os.environ["TRACE"], encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
lines = []
|
||||
for line in lines:
|
||||
try:
|
||||
step = json.loads(line)
|
||||
except ValueError:
|
||||
continue
|
||||
witnesses = step.get("witnesses") or {}
|
||||
for name in step.get("violations") or []:
|
||||
witness = witnesses.get(name) or {}
|
||||
if witness.get("is_error"):
|
||||
reason = " ".join(str(witness.get("reason") or "").split())
|
||||
thrown.append("%s: %s" % (name, reason) if reason else name)
|
||||
elif name in gated:
|
||||
convictions.append(name)
|
||||
else:
|
||||
other.append(name)
|
||||
for names in (convictions, thrown, other):
|
||||
print(", ".join(names))
|
||||
PY
|
||||
) || {
|
||||
echo "folio/$platform: could not classify $trace, so exit $code cannot be read as a verdict" >&2
|
||||
exit 1
|
||||
}
|
||||
convicted=$(printf '%s\n' "$classified" | sed -n '1p')
|
||||
thrown=$(printf '%s\n' "$classified" | sed -n '2p')
|
||||
other=$(printf '%s\n' "$classified" | sed -n '3p')
|
||||
|
||||
{
|
||||
echo "### folio on $platform"
|
||||
echo
|
||||
echo "- seed \`$seed\`, budget $max_steps steps / $duration"
|
||||
echo "- $steps steps recorded, exit $code"
|
||||
[ -n "$convicted" ] && echo "- convicted on: $convicted"
|
||||
[ -n "$other" ] && echo "- also violated: $other"
|
||||
[ -n "$thrown" ] && echo "- **a predicate threw**: $thrown"
|
||||
} >> "$summary"
|
||||
|
||||
# A thrown predicate fails every leg, android included. It is not evidence about
|
||||
# folio: the property that threw is violated from that step on whatever the app
|
||||
# does, and --exit-on-violation ends the run there, so nothing past it was
|
||||
# checked. Reporting that as a conviction, or as android's bonus, is the hole
|
||||
# this check exists to close.
|
||||
if [ -n "$thrown" ]; then
|
||||
echo "folio/$platform: a predicate threw, so exit $code is not a verdict about folio: $thrown" >&2
|
||||
echo "folio/$platform: fix the spec (examples/folio/sanderling/) and run again" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Only 0 and 2, the two codes that claim the run completed: any other code is a
|
||||
# harness failure, which the branches below already report as one. Without this,
|
||||
# a missing trace is judged as an empty trace and reported as a verdict on folio.
|
||||
if [ ! -f "$trace" ] && { [ "$code" = 0 ] || [ "$code" = 2 ]; }; then
|
||||
echo "folio/$platform: the run exited $code but wrote no trace under $output/, so there is nothing to judge" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ "$platform" = "android" ]; then
|
||||
# A health gate, not a conviction gate: android convicts in four runs out of
|
||||
# five, and a gate that fails the fifth would report a regression it had not
|
||||
# found. Reaching the transaction screen is what this leg proves: the app
|
||||
# built, installed, launched and drove.
|
||||
case "$code" in
|
||||
0) ;;
|
||||
2)
|
||||
if [ -n "$convicted" ]; then
|
||||
echo "folio/android: found the submit bug in $steps steps (a bonus, not required)"
|
||||
else
|
||||
echo "folio/android: violated $other, which is not the double-submit; judging health only"
|
||||
fi
|
||||
;;
|
||||
*) echo "folio/android: the harness failed with exit $code" >&2; exit "$code" ;;
|
||||
esac
|
||||
if ! grep -q '"AddTransactionScreen"' "$trace"; then
|
||||
echo "folio/android: the run never reached AddTransactionScreen, so it never got past login" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "folio/android: healthy run over $steps steps, reached the transaction screen"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
case "$code" in
|
||||
2)
|
||||
if [ -z "$convicted" ]; then
|
||||
echo "folio/$platform: exit 2, but the violation was ${other:-nothing this trace records}, not the double-submit this leg gates on" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "folio/$platform: found the submit bug in $steps steps ($convicted)"
|
||||
exit 0
|
||||
;;
|
||||
0)
|
||||
echo "folio/$platform: the run finished clean; the double-submit bug was NOT found in $steps steps (seed $seed)" >&2
|
||||
echo "folio/$platform: the spec ran without throwing, so this is the fuzzer no longer reaching the bug, not a broken spec" >&2
|
||||
exit 1
|
||||
;;
|
||||
*) echo "folio/$platform: the harness failed with exit $code" >&2; exit "$code" ;;
|
||||
esac
|
||||
@@ -8,6 +8,9 @@ on:
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
@@ -53,8 +56,16 @@ jobs:
|
||||
restore-keys: |
|
||||
bun-${{ runner.os }}-
|
||||
|
||||
# The token is what stops this step flaking: without it the action pulls
|
||||
# buf's release tarball from github.com anonymously, on the shared runner
|
||||
# IP's rate limit, and a throttled connection shows up as `socket hang
|
||||
# up` after three retries. The version is the action's own default, made
|
||||
# explicit so a new action release cannot move the buf we build with.
|
||||
- name: Install buf
|
||||
uses: bufbuild/buf-setup-action@v1
|
||||
uses: bufbuild/buf-setup-action@a47c93e0b1648d5651a065437926377d060baa99 # v1.50.0
|
||||
with:
|
||||
version: "1.50.0"
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Install protoc plugins
|
||||
run: |
|
||||
|
||||
@@ -0,0 +1,266 @@
|
||||
name: folio
|
||||
|
||||
# One spec, three platforms. Dispatch-only: each job boots a device or a
|
||||
# browser, builds the folio app for that platform, and runs
|
||||
# examples/folio/sanderling/spec.ts against it.
|
||||
#
|
||||
# web and ios are expect-the-bug jobs: folio double-submits a transaction on a
|
||||
# double tap, so the run is supposed to end with exit 2. Exit 0 means the fuzzer
|
||||
# stopped finding a bug that is still there; exit 1 means the harness broke. The
|
||||
# two are worth telling apart, which is why --exit-on-violation exits 2 and not
|
||||
# 1.
|
||||
#
|
||||
# android is a health gate. It convicts in four runs out of five, which is real
|
||||
# evidence but not a gate: the fifth would report a regression it had not found.
|
||||
# The budget is set so the conviction it usually gets is reported as a bonus.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
platforms:
|
||||
description: which legs to run
|
||||
type: choice
|
||||
options: [all, android, ios, web]
|
||||
default: all
|
||||
seed:
|
||||
description: seed override (0 = each job's calibrated seed)
|
||||
default: "0"
|
||||
duration:
|
||||
description: wall-clock budget per run
|
||||
default: 20m
|
||||
max-steps:
|
||||
description: step budget override (0 = each job's calibrated budget)
|
||||
default: "0"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
android:
|
||||
timeout-minutes: 90
|
||||
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'android' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up JDK 17
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "17"
|
||||
|
||||
- name: Set up Android SDK
|
||||
uses: android-actions/setup-android@v3
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
- name: Cache Gradle
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
|
||||
restore-keys: |
|
||||
folio-gradle-${{ runner.os }}-
|
||||
|
||||
# Without this the emulator falls back to software rendering and every
|
||||
# step costs several seconds.
|
||||
- name: Enable KVM
|
||||
run: |
|
||||
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \
|
||||
| sudo tee /etc/udev/rules.d/99-kvm4all.rules
|
||||
sudo udevadm control --reload-rules
|
||||
sudo udevadm trigger --name-match=kvm
|
||||
|
||||
- name: Build the folio APK
|
||||
run: ./gradlew :app:androidApp:assembleDebug
|
||||
working-directory: examples/folio
|
||||
|
||||
- name: Build sanderling
|
||||
run: make sanderling-android
|
||||
|
||||
- name: Fuzz folio on an emulator
|
||||
uses: reactivecircus/android-emulator-runner@v2
|
||||
with:
|
||||
api-level: 34
|
||||
target: google_apis
|
||||
arch: x86_64
|
||||
emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim
|
||||
disable-animations: true
|
||||
script: .github/scripts/folio-run.sh android
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || '9' }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '200' }}
|
||||
DURATION: ${{ inputs.duration }}
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: folio-android
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
|
||||
ios:
|
||||
timeout-minutes: 90
|
||||
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'ios' }}
|
||||
runs-on: macos-15
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
# idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so
|
||||
# it has to be named by its full tap path. xcodegen and just are core.
|
||||
- name: Install idb-companion, xcodegen and just
|
||||
run: brew install facebook/fb/idb-companion xcodegen just
|
||||
|
||||
- name: Set up JDK 17
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "17"
|
||||
|
||||
# The iOS app builds its Kotlin framework through the folio gradle
|
||||
# project, which configures :app:androidApp and so needs an Android SDK
|
||||
# even on this leg.
|
||||
- name: Set up Android SDK
|
||||
uses: android-actions/setup-android@v3
|
||||
|
||||
# Both asset tarballs are built by the prepare scripts, and the runner
|
||||
# bundle is an xcodebuild of companion/Sources. Keyed on the scripts and
|
||||
# the versions the Makefile embeds, so a later run reuses them.
|
||||
- name: Cache the companion and runner bundles
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
internal/driver/ioscompanion/companionassets/assets
|
||||
internal/driver/ioscompanion/runnerassets/assets
|
||||
key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }}
|
||||
|
||||
- name: Build sanderling
|
||||
run: make sanderling-ios
|
||||
|
||||
- name: Boot a simulator
|
||||
run: |
|
||||
xcrun simctl boot "$IOS_DEVICE" || true
|
||||
xcrun simctl bootstatus "$IOS_DEVICE" -b
|
||||
env:
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
|
||||
- name: Build and install folio
|
||||
run: just ios
|
||||
working-directory: examples/folio
|
||||
env:
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
|
||||
# `just ios` leaves the app running, and the run's own clear-data
|
||||
# reinstall on top of a live app has raced FrontBoard into refusing the
|
||||
# launch ("app.folio is unknown to FrontBoard") with the run then hanging.
|
||||
- name: Stop the app before the run
|
||||
run: xcrun simctl terminate booted app.folio || true
|
||||
|
||||
- name: Fuzz folio on the simulator
|
||||
run: .github/scripts/folio-run.sh ios
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || '7' }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }}
|
||||
DURATION: ${{ inputs.duration }}
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: folio-ios
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
|
||||
web:
|
||||
timeout-minutes: 60
|
||||
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'web' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up JDK 17
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "17"
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
- name: Cache Gradle
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
|
||||
restore-keys: |
|
||||
folio-gradle-${{ runner.os }}-
|
||||
|
||||
- name: Set up Chrome
|
||||
uses: browser-actions/setup-chrome@v1
|
||||
with:
|
||||
chrome-version: stable
|
||||
|
||||
- name: Allow Chrome under unprivileged user namespaces
|
||||
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
|
||||
- name: Verify headless Chrome starts
|
||||
run: |
|
||||
chrome --version
|
||||
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
|
||||
--dump-dom 'data:text/html,<title>ok</title>'
|
||||
|
||||
- name: Build the folio wasmJs app
|
||||
run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution
|
||||
working-directory: examples/folio
|
||||
|
||||
- name: Build sanderling
|
||||
run: make sanderling-web
|
||||
|
||||
- name: Fuzz folio in the browser
|
||||
run: .github/scripts/folio-run.sh web
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || '3' }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }}
|
||||
DURATION: ${{ inputs.duration }}
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: folio-web
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
@@ -0,0 +1,148 @@
|
||||
name: replay-ui
|
||||
|
||||
# Sanderling fuzzing sanderling's own replay UI. Dispatch-only: it takes minutes
|
||||
# and it is a demo of the product loop, not a merge gate.
|
||||
#
|
||||
# The shape is: produce a real trace, serve it with `sanderling replay`, then run
|
||||
# a spec against that UI. Any violation fails the job. Six of the seven
|
||||
# properties in replay-ui/sanderling/spec.ts are cross-panel agreements that hold
|
||||
# for any trace; the seventh is the stock noUncaughtExceptions. None of them
|
||||
# needs recalibrating when the fixture changes.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
seed:
|
||||
description: seed for the dogfood run
|
||||
default: "3"
|
||||
max-steps:
|
||||
description: step budget for the dogfood run
|
||||
default: "80"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
dogfood:
|
||||
timeout-minutes: 45
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
# Pinned stable plus the AppArmor sysctl: the same setup ci.yml's browser
|
||||
# job needs to get headless Chrome up on ubuntu-latest.
|
||||
- name: Set up Chrome
|
||||
uses: browser-actions/setup-chrome@v1
|
||||
with:
|
||||
chrome-version: stable
|
||||
|
||||
- name: Allow Chrome under unprivileged user namespaces
|
||||
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
|
||||
- name: Verify headless Chrome starts
|
||||
run: |
|
||||
chrome --version
|
||||
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
|
||||
--dump-dom 'data:text/html,<title>ok</title>'
|
||||
|
||||
# The UI the spec drives is the one embedded in this binary, so the build
|
||||
# has to come after any change to replay-ui/src.
|
||||
- name: Build sanderling
|
||||
run: make sanderling-web
|
||||
|
||||
# A trace with a violation and uncaught exceptions in it, so the UI has
|
||||
# something to render in every panel the spec looks at. No
|
||||
# --exit-on-violation here: the run is the fixture, and stopping it at the
|
||||
# first violation would leave a four-step trace to fuzz.
|
||||
- name: Record a fixture trace
|
||||
run: |
|
||||
python3 -m http.server 8792 --bind 127.0.0.1 \
|
||||
--directory test/browser/testdata/throwing &
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf http://127.0.0.1:8792/ >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "the fixture http server never answered on 127.0.0.1:8792" >&2
|
||||
exit 1
|
||||
fi
|
||||
./bin/sanderling test \
|
||||
--platform web \
|
||||
--spec test/browser/testdata/throwing/spec.ts \
|
||||
--bundle-id http://127.0.0.1:8792/ \
|
||||
--duration 5m --max-steps 25 --seed 7 \
|
||||
--output runs/fixture
|
||||
|
||||
- name: Serve the trace with sanderling replay
|
||||
run: |
|
||||
# Flags before the positional argument: Go's flag package stops
|
||||
# parsing at the first non-flag word.
|
||||
./bin/sanderling replay --port 8793 --no-open runs/fixture &
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf http://127.0.0.1:8793/api/runs >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "sanderling replay never served /api/runs on 127.0.0.1:8793" >&2
|
||||
exit 1
|
||||
fi
|
||||
run_id="$(ls runs/fixture | head -1)"
|
||||
echo "RUN_URL=http://127.0.0.1:8793/runs/$run_id/steps/1" >> "$GITHUB_ENV"
|
||||
curl -sf "http://127.0.0.1:8793/runs/$run_id/steps/1" >/dev/null
|
||||
|
||||
# Inputs go through env rather than into the script text: a `${{ }}` is
|
||||
# substituted before bash ever sees the line, so a seed of `$(id)` would
|
||||
# run as a command.
|
||||
- name: Fuzz the replay UI
|
||||
run: |
|
||||
./bin/sanderling test \
|
||||
--platform web \
|
||||
--spec replay-ui/sanderling/spec.ts \
|
||||
--bundle-id "$RUN_URL" \
|
||||
--duration 10m \
|
||||
--max-steps "$MAX_STEPS" \
|
||||
--seed "$SEED" \
|
||||
--exit-on-violation \
|
||||
--output runs/dogfood
|
||||
env:
|
||||
SEED: ${{ inputs.seed }}
|
||||
MAX_STEPS: ${{ inputs.max-steps }}
|
||||
|
||||
- name: Summarise
|
||||
if: always()
|
||||
run: |
|
||||
{
|
||||
echo "### replay-ui dogfood"
|
||||
echo
|
||||
echo "- seed \`$SEED\`, budget $MAX_STEPS steps"
|
||||
for dir in runs/dogfood/*/; do
|
||||
[ -f "$dir/trace.jsonl" ] || continue
|
||||
steps=$(wc -l < "$dir/trace.jsonl" | tr -d ' ')
|
||||
violations=$(grep -c '"violations":\[' "$dir/trace.jsonl" || true)
|
||||
echo "- $steps steps recorded, $violations step(s) with violations"
|
||||
done
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
env:
|
||||
SEED: ${{ inputs.seed }}
|
||||
MAX_STEPS: ${{ inputs.max-steps }}
|
||||
|
||||
- name: Upload runs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: replay-ui-runs
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
@@ -36,6 +36,9 @@
|
||||
|
||||
## PR Rules
|
||||
|
||||
- Never merge a pull request. I merge, nobody else. That covers `gh pr merge`, enabling
|
||||
auto-merge, squash, rebase, fast-forward, and merging the branch into anything locally.
|
||||
Push the branch, say it is ready, and stop there.
|
||||
- Simple PR title, few-line description. Never write a wall of text. Nobody reads it.
|
||||
- Everything lowercase in PR titles, descriptions, and comments.
|
||||
|
||||
|
||||
@@ -29,7 +29,7 @@ WEB_DIST := replay-ui/dist
|
||||
|
||||
GOLINES := $(shell $(GO) env GOPATH)/bin/golines
|
||||
|
||||
.PHONY: bootstrap proto sidecar sanderling install test test-go test-browser test-companion test-kotlin test-spec-api spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift
|
||||
.PHONY: bootstrap proto sidecar sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift
|
||||
|
||||
bootstrap:
|
||||
$(GO) mod download
|
||||
@@ -48,6 +48,22 @@ $(SANDERLING_BIN): $(SIDECAR_EMBED) $(COMPANION_EMBED) $(RUNNER_EMBED) web-build
|
||||
mkdir -p bin
|
||||
$(GO) build -tags "withsidecar withcompanion" -o $(SANDERLING_BIN) ./cmd/sanderling
|
||||
|
||||
# Per-platform builds, each taking only the tags that platform needs. `make
|
||||
# sanderling` cannot run on Linux at all: preparing the iOS companion asset
|
||||
# shells out to a macOS homebrew path. The CI workflows build through these so
|
||||
# a job and a developer produce the same binary.
|
||||
sanderling-web: web-build
|
||||
mkdir -p bin
|
||||
$(GO) build -o $(SANDERLING_BIN) ./cmd/sanderling
|
||||
|
||||
sanderling-android: $(SIDECAR_EMBED) web-build
|
||||
mkdir -p bin
|
||||
$(GO) build -tags withsidecar -o $(SANDERLING_BIN) ./cmd/sanderling
|
||||
|
||||
sanderling-ios: $(COMPANION_EMBED) $(RUNNER_EMBED) web-build
|
||||
mkdir -p bin
|
||||
$(GO) build -tags withcompanion -o $(SANDERLING_BIN) ./cmd/sanderling
|
||||
|
||||
# Installs `sanderling` into $GOBIN (or $GOPATH/bin) so it's directly on PATH for
|
||||
# anyone with a standard Go toolchain setup.
|
||||
install: $(SIDECAR_EMBED) $(COMPANION_EMBED) $(RUNNER_EMBED) web-build
|
||||
@@ -57,7 +73,7 @@ install: $(SIDECAR_EMBED) $(COMPANION_EMBED) $(RUNNER_EMBED) web-build
|
||||
web-build:
|
||||
cd replay-ui && bun install --frozen-lockfile && bun run build
|
||||
mkdir -p $(REPLAY_DIST)
|
||||
rm -rf $(REPLAY_DIST)/assets $(REPLAY_DIST)/fonts
|
||||
rm -rf $(REPLAY_DIST)/assets
|
||||
cp -R $(WEB_DIST)/. $(REPLAY_DIST)/
|
||||
|
||||
web-dev:
|
||||
@@ -101,7 +117,7 @@ fmt-ts:
|
||||
fmt-swift:
|
||||
xcrun swift-format format -i -r companion/Sources
|
||||
|
||||
test: test-go spec-typecheck test-spec-api web-typecheck web-test
|
||||
test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test
|
||||
|
||||
test-go:
|
||||
$(GO) test $(GO_PACKAGES)
|
||||
|
||||
+24
-8
@@ -11,6 +11,8 @@ import (
|
||||
"os/signal"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/priyanshujain/sanderling/internal/testrun"
|
||||
)
|
||||
|
||||
// Version is stamped at build time via goreleaser ldflags.
|
||||
@@ -34,6 +36,7 @@ type testOptions struct {
|
||||
clearData bool
|
||||
generator string
|
||||
labelSource string
|
||||
exitOnViolation bool
|
||||
}
|
||||
|
||||
const topUsage = `sanderling is a property-based UI fuzzer for mobile apps.
|
||||
@@ -70,6 +73,7 @@ func parseTestArgs(args []string, stderr io.Writer) (testOptions, error) {
|
||||
flagSet.StringVar(&options.arm, "arm", "", "experiment cell label, recorded in meta.json so a directory of runs can be attributed to a cell")
|
||||
flagSet.StringVar(&options.generator, "generator", "seeded", "action generator: seeded (weighted random) or llm (model picks from the same candidate set; requires generator = llm() in the spec)")
|
||||
flagSet.StringVar(&options.labelSource, "label-source", "visible-text", "how candidates are named to the llm generator: visible-text (what a user reads) or resource-id (the identifier the app assigned). The seeded generator picks by index and ignores this")
|
||||
flagSet.BoolVar(&options.exitOnViolation, "exit-on-violation", false, "stop the run at the first property violation and exit 2, so CI can tell a found bug (2) from a broken harness (1)")
|
||||
if err := flagSet.Parse(args); err != nil {
|
||||
return testOptions{}, err
|
||||
}
|
||||
@@ -151,13 +155,25 @@ func run(args []string, stdout, stderr io.Writer) error {
|
||||
}
|
||||
|
||||
func main() {
|
||||
if err := run(os.Args, os.Stdout, os.Stderr); err != nil {
|
||||
// flag.ErrHelp means -h/--help was requested; flag already printed
|
||||
// usage to stderr, so exit 0 rather than treating it as a failure.
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return
|
||||
}
|
||||
fmt.Fprintf(os.Stderr, "error: %v\n", err)
|
||||
os.Exit(1)
|
||||
if code := exitCode(run(os.Args, os.Stdout, os.Stderr), os.Stderr); code != 0 {
|
||||
os.Exit(code)
|
||||
}
|
||||
}
|
||||
|
||||
// exitCode maps a command result to the process status and reports it on
|
||||
// stderr. 2 means the run did its job and found violations under
|
||||
// --exit-on-violation; 1 stays "something went wrong", so CI can tell a found
|
||||
// bug from a broken harness. flag.ErrHelp means -h/--help was requested and
|
||||
// flag already printed usage, so it exits 0 rather than reading as a failure.
|
||||
func exitCode(err error, stderr io.Writer) int {
|
||||
if err == nil || errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
var violations testrun.ViolationsError
|
||||
if errors.As(err, &violations) {
|
||||
fmt.Fprintf(stderr, "violations: %d\n", violations.Count)
|
||||
return 2
|
||||
}
|
||||
fmt.Fprintf(stderr, "error: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
@@ -2,11 +2,14 @@ package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"errors"
|
||||
"flag"
|
||||
"io"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/priyanshujain/sanderling/internal/testrun"
|
||||
"github.com/priyanshujain/sanderling/internal/verifier"
|
||||
)
|
||||
|
||||
@@ -370,3 +373,58 @@ func TestParseTestArgs_ArmLabel(t *testing.T) {
|
||||
t.Errorf("arm: got %q, want seeded-identifier", options.arm)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseTestArgs_ExitOnViolationDefaultsOff(t *testing.T) {
|
||||
options, err := parseTestArgs([]string{
|
||||
"--spec", "s.ts",
|
||||
"--bundle-id", "com.example",
|
||||
}, io.Discard)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if options.exitOnViolation {
|
||||
t.Error("exitOnViolation default: got true, want false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseTestArgs_ExitOnViolation(t *testing.T) {
|
||||
options, err := parseTestArgs([]string{
|
||||
"--spec", "s.ts",
|
||||
"--bundle-id", "com.example",
|
||||
"--exit-on-violation",
|
||||
}, io.Discard)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !options.exitOnViolation {
|
||||
t.Error("expected exitOnViolation=true")
|
||||
}
|
||||
}
|
||||
|
||||
// TestExitCode_SeparatesFoundBugsFromBrokenHarnesses pins the three statuses CI
|
||||
// reads: 0 clean, 2 the run found violations, 1 everything else. A workflow
|
||||
// that asserts "the known bug is still found" is only meaningful while 2 and 1
|
||||
// stay distinct.
|
||||
func TestExitCode_SeparatesFoundBugsFromBrokenHarnesses(t *testing.T) {
|
||||
for _, testCase := range []struct {
|
||||
name string
|
||||
err error
|
||||
want int
|
||||
says string
|
||||
}{
|
||||
{"clean run", nil, 0, ""},
|
||||
{"help", flag.ErrHelp, 0, ""},
|
||||
{"violations found", testrun.ViolationsError{Count: 2}, 2, "violations: 2"},
|
||||
{"broken harness", errors.New("launch app: no device"), 1, "error: launch app"},
|
||||
} {
|
||||
t.Run(testCase.name, func(t *testing.T) {
|
||||
var stderr bytes.Buffer
|
||||
if got := exitCode(testCase.err, &stderr); got != testCase.want {
|
||||
t.Errorf("exit code: got %d, want %d", got, testCase.want)
|
||||
}
|
||||
if testCase.says != "" && !strings.Contains(stderr.String(), testCase.says) {
|
||||
t.Errorf("stderr %q does not mention %q", stderr.String(), testCase.says)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -32,5 +32,6 @@ func pipelineOptions(options testOptions) testrun.Options {
|
||||
Generator: options.generator,
|
||||
LabelSource: options.labelSource,
|
||||
Arm: options.arm,
|
||||
ExitOnViolation: options.exitOnViolation,
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
---
|
||||
title: CI
|
||||
---
|
||||
|
||||
# CI
|
||||
|
||||
`ci.yml` runs on every pull request: it builds, unit-tests, and drives three
|
||||
small web fixtures through headless Chrome (`test/browser/testdata`). It never
|
||||
runs sanderling against a real app.
|
||||
|
||||
Two other workflows do, and both are `workflow_dispatch` only. They boot devices,
|
||||
build apps and take minutes, which is not what you want on every push, and
|
||||
neither is a merge gate.
|
||||
|
||||
## folio
|
||||
|
||||
Actions -> folio -> Run workflow. Inputs pick the legs (`all`, `android`, `ios`,
|
||||
`web`), and override the seed, the step budget and the wall-clock budget. Seed
|
||||
and step budget take `0` to mean "use the calibrated value in the workflow"; the
|
||||
wall-clock budget has no such sentinel and is passed through as written, so
|
||||
every leg gets whatever you type there.
|
||||
|
||||
Each leg builds `examples/folio` for its platform, builds the CLI with only the
|
||||
tags that platform needs (`make sanderling-android` and friends), and runs
|
||||
`examples/folio/sanderling/spec.ts` through `.github/scripts/folio-run.sh`. That
|
||||
script is plain bash so you can reproduce a job locally, with that leg's pinned
|
||||
numbers:
|
||||
|
||||
```
|
||||
SEED=9 MAX_STEPS=200 .github/scripts/folio-run.sh android
|
||||
```
|
||||
|
||||
**web and ios expect the bug.** Folio double-submits a transaction when the
|
||||
submit button is double-tapped, and two properties catch it:
|
||||
`submitMovesBalanceByTypedAmount`, which demands the total balance move by
|
||||
exactly the amount typed, and `submitCommitsOneTransactionPerAction`, which
|
||||
demands no more transactions committed over a window than there were submit
|
||||
actions in it. A double tap is one action committing two transactions, so it
|
||||
breaks both.
|
||||
|
||||
The runs pass `--exit-on-violation`, which exits 2 when the run recorded a
|
||||
violation and 1 when something went wrong. Telling those apart is the whole
|
||||
point of the exit code: a job that only knew "non-zero" could not tell a working
|
||||
fuzzer from a broken emulator.
|
||||
|
||||
Exit 2 on its own is not a conviction, though, so the script reads the trace
|
||||
before it decides:
|
||||
|
||||
| what the trace says | job |
|
||||
|---|---|
|
||||
| a violation of one of the two properties above, with no `is_error` on its witness | green |
|
||||
| a violation whose witness carries `is_error` | red: a predicate threw, and a thrown predicate is recorded as a violation like any other |
|
||||
| a violation of any other property (`newAccountBalanceIsZero` fires on one android seed) | red: a real finding, but not the one this leg gates on |
|
||||
| no violation and exit 0 | red: the fuzzer stopped finding a bug that is still there |
|
||||
| no trace at all, on exit 0 or 2 | red: the run recorded nothing, so there is no verdict to read |
|
||||
| exit 1, or any other code | red: the harness broke, and the code propagates |
|
||||
|
||||
The first two rows are why the check is worth the code it takes. A `TypeError`
|
||||
in `predicates.ts` and a fuzzer that no longer reaches the bug both used to
|
||||
print "found the submit bug" and exit 0, and they need opposite responses: one
|
||||
is a spec to fix, the other is a seed to recalibrate. A thrown predicate fails
|
||||
android as well, where a conviction is otherwise only a bonus, because a spec
|
||||
that stopped running is not evidence about the app.
|
||||
|
||||
The balance property only judges a window holding exactly one submit action.
|
||||
Without that rule the recorded delta covers every transaction since the last Home
|
||||
visit, and the property convicts on arithmetic it cannot attribute: an early
|
||||
version of this gate went green on a witness whose delta was 3.16x the typed
|
||||
amount. Honest windows are rare, so the counting invariant carries most of the
|
||||
detection: it needs no amount and survives a wide window.
|
||||
|
||||
**android is a health gate**, not a conviction gate, because it convicts in four
|
||||
runs out of five rather than five. The leg asserts that the run stayed healthy
|
||||
and reached the transaction screen, and reports the conviction it usually gets as
|
||||
a bonus. A gate that fails one run in five would be useless here: its failure
|
||||
message, "the double-submit bug was NOT found", is indistinguishable from the
|
||||
regression the gate exists to catch.
|
||||
|
||||
It used to be two runs in five. The android backend now waits out a route
|
||||
cross-fade before it snapshots, the way the ios companion and the chrome driver
|
||||
do, because a dump holding two screens at once makes the runner refuse to act on
|
||||
it and a quarter of android steps therefore applied no action at all. The number
|
||||
of those wasted steps varied run to run, so the same seed never walked the same
|
||||
trajectory. With the wait, four runs of one seed produced byte-identical decision
|
||||
sequences over 179 steps.
|
||||
|
||||
The fifth diverged for the remaining reason: a route can settle before its
|
||||
content composes, so the tree holds one screen and almost nothing in it, and the
|
||||
fuzzer acts on a screen that is still filling in. That happened once in about a
|
||||
thousand steps. Catching it needs structural stability polling on every snapshot,
|
||||
which costs roughly 2.8s per mutating step and was removed for that reason, so it
|
||||
is the open item between android and a real conviction gate.
|
||||
|
||||
The wasmJs app is served with `Cross-Origin-Opener-Policy` and
|
||||
`Cross-Origin-Embedder-Policy` headers, because its sqlite worker needs
|
||||
cross-origin isolation. Served without them the app loads a blank canvas and
|
||||
every step observes an empty accessibility tree.
|
||||
|
||||
The seeds are calibrated, not guessed. On an M-series mac, web seed 3 convicts at
|
||||
step 185-187 and ios seed 7 at step 97-101, each 3 runs out of 3 and each with a
|
||||
delta of exactly twice the typed amount. Both run a 240-step budget. Keep them
|
||||
pinned: honest evidence is rare, and across 2261 ios steps only one submit tap
|
||||
landing on Home had a single-submit window.
|
||||
|
||||
Android runs seed 9 over 200 steps: its conviction lands at step 178, so a
|
||||
shorter budget would never see the bonus. A full run costs about five minutes.
|
||||
|
||||
Repeating the ios leg by hand is not the same as running it in CI: with
|
||||
`--clear-data=false` a second local run inherits the first one's accounts, so
|
||||
`simctl uninstall` before each repeat or the numbers drift.
|
||||
|
||||
The ios leg passes `--clear-data=false`, because the job installs a fresh build
|
||||
immediately before the run and a freshly installed app is already clear state.
|
||||
The in-run reinstall is worth avoiding: `simctl uninstall` + `install` followed
|
||||
straight away by the XCTest runner's own launch fails with `app.folio is unknown
|
||||
to FrontBoard` maybe half the time. That used to hang the run outright; the
|
||||
launch RPC is bounded now, so it fails in about 90 seconds with a real error
|
||||
instead, but a failing leg is still a failing leg. The job timeouts are the
|
||||
backstop if it happens anyway.
|
||||
|
||||
Only one sanderling run may drive a given simulator at a time. The driver takes
|
||||
an advisory lock on the target's UDID and a second run is refused with the lock
|
||||
path in the message, because two runs interleaving app lifecycle leave the first
|
||||
run's automation session bound to a bundle the simulator no longer knows.
|
||||
|
||||
## replay-ui
|
||||
|
||||
Actions -> replay-ui -> Run workflow. This one is dogfooding: it records a trace
|
||||
from `test/browser/testdata/throwing` (violations and uncaught exceptions, so
|
||||
every panel has something to render), serves it with `sanderling replay`, and
|
||||
fuzzes that UI with `replay-ui/sanderling/spec.ts`.
|
||||
|
||||
Six of the seven properties there are cross-panel agreements - two panels
|
||||
deriving the same fact by different paths have to say the same thing - so they
|
||||
hold for any trace and need no recalibrating when the fixture changes. The
|
||||
seventh is the stock `noUncaughtExceptions`, which asks nothing of the panels
|
||||
and only fails if the UI throws. Any violation fails the job.
|
||||
|
||||
## Reading a failure
|
||||
|
||||
Both workflows upload their run directories as artifacts, and write the step
|
||||
count, seed and violations to the job summary. To replay a failure:
|
||||
|
||||
```
|
||||
gh run download <run-id> -n folio-android
|
||||
sanderling replay <the downloaded runs directory>
|
||||
```
|
||||
|
||||
That is the same UI the replay-ui workflow fuzzes. Open the step the summary
|
||||
named, and the Violations tab shows the witness: the property, the reason, and
|
||||
the extractor values at the step that caused it.
|
||||
|
||||
## When a device leg flakes
|
||||
|
||||
Expecting a violation from a single seed is timing-sensitive, most of all on an
|
||||
emulator. Calibration reduces that; it does not remove it. If a platform starts
|
||||
failing across repeated dispatches with "the double-submit bug was NOT found",
|
||||
do not raise the step budget blindly - run a seed sweep with the campaign tool
|
||||
(`cmd/internal-tools/campaign`), which exists for exactly this, and pin a seed
|
||||
that finds the bug with room to spare. A leg failing with "a predicate threw" is
|
||||
a different problem entirely and no seed will fix it.
|
||||
@@ -51,3 +51,25 @@ There is exactly one authoring surface: the TypeScript spec. There is no separat
|
||||
|
||||
This is intentional. A test harness with two authoring languages (YAML plus code, JSON plus code) splits concerns in a way that always drifts. Something works in one surface and not the other, and debugging requires holding both in your head.
|
||||
|
||||
|
||||
## 8. A property that cannot fail is worse than one that fails
|
||||
|
||||
A spec fails loudly when it is wrong about the app. It fails silently when it is wrong about
|
||||
itself, and that is the failure this project has to design against.
|
||||
|
||||
Two shapes recur. A property is *vacuously true* when a fact it reads is missing, so it never
|
||||
fires and every run is green: `state.lastAction` was hardcoded null on web, `ax.findAll` on a
|
||||
selector path returned nothing there, and no element on iOS carried `text` because the mapping
|
||||
only ever read `AXValue`. In each case a correct property proved nothing. A property is
|
||||
*degenerately false* when a missing fact is read as a value instead: folio's balances parsed as
|
||||
`0` on web, so the check became `|0 - 0| === typedAmount` and fired on every healthy submit.
|
||||
Both shapes look exactly like a working property from the outside.
|
||||
|
||||
So an unreadable fact is unknown, never a default. Extractors return null rather than zero or
|
||||
empty string, and a property given an unknown declines to judge instead of convicting. The same
|
||||
rule covers arithmetic the representation cannot hold: integer cents in a float64 stop being
|
||||
exact past 2^53, and a comparison that cannot pass is not a comparison that failed.
|
||||
|
||||
Which means a green run is evidence only if you can point at a step where the property actually
|
||||
fired, and a red one only if the witness holds real values. Read the witness, not the exit code.
|
||||
Every bug listed above was found that way, and each had already survived a run that looked fine.
|
||||
@@ -6,6 +6,7 @@ title: Development
|
||||
|
||||
- [Design principles](./design-principles/)
|
||||
- [Architecture](./architecture/)
|
||||
- [CI](./ci/)
|
||||
- v0.1.0 scope: [milestone #1](https://github.com/priyanshujain/sanderling/milestone/1)
|
||||
|
||||
## Building the docs site locally
|
||||
|
||||
+67
-17
@@ -33,45 +33,95 @@ const submitMovesBalanceByTypedAmount = always(
|
||||
const action = lastAction.current;
|
||||
if (action?.kind !== "Tap" && action?.kind !== "DoubleTap") return true;
|
||||
if (!JSON.stringify(action.on ?? "").includes("TxnSubmit")) return true;
|
||||
if (submitsInWindow.current !== 1) return true;
|
||||
const typed = parseTypedAmount(txnAmountField.previous?.text);
|
||||
if (typed === 0) return true;
|
||||
return Math.abs(totalBalance.current - (totalBalance.previous ?? 0)) === typed;
|
||||
const before = totalBalance.previous;
|
||||
if (before === null || totalBalance.current === null) return true;
|
||||
return Math.abs(totalBalance.current - before) === typed;
|
||||
})
|
||||
);
|
||||
```
|
||||
|
||||
`always` checks the formula at every step; `next` lets it compare the step before a submit to the step after. The guards narrow it to the one transition that matters, a submit that lands back on home, and the last line states the rule: the balance moved by exactly the typed amount. Double-submit moves it by twice that, and the formula is false.
|
||||
|
||||
The window guard is the difference between a property and a false conviction. `totalBalance.previous` is the last total we read, not the total as of the last transaction, so the two numbers being compared can straddle any number of commits: a real run produced a delta of 13000 against a typed 19600, because the window held a double-submit's two 19600 debits and an unrelated 26200 credit. A delta like that is not evidence about the amount typed into any one submit. Exactly one submit action in the window still catches the bug, because the double tap is a single action.
|
||||
|
||||
The null guard is not defensive clutter either. Read a balance you could not parse as `0` and the comparison becomes `0 - 0 === typed`, which is false at every healthy submit. A reading you do not have is not evidence, so the property declines to judge. The real spec guards the same way against a balance too large for exact integer arithmetic.
|
||||
|
||||
The values it reads come from extractors, which pull state out of the UI tree once per step:
|
||||
|
||||
```ts
|
||||
const route = extract<string | null>("route", s => {
|
||||
if (s.ax.find({ testTag: "AddTransactionScreen" })) return "add-transaction";
|
||||
if (s.ax.find({ testTag: "HomeScreen" })) return "home";
|
||||
// ...other screens
|
||||
return null;
|
||||
});
|
||||
const SCREENS = {
|
||||
login: "LoginScreen",
|
||||
"add-account": "AddAccountScreen",
|
||||
"add-transaction": "AddTransactionScreen",
|
||||
ledger: "LedgerScreen",
|
||||
home: "HomeScreen",
|
||||
} as const;
|
||||
type Route = keyof typeof SCREENS;
|
||||
|
||||
const totalBalance = extract("totalBalance", s =>
|
||||
s.ax.findAll([{ testTag: "HomeScreen" }, { testTag: "AccountCard" }])
|
||||
.reduce((sum, c) => sum + parseDollarCents(c.find({ testTag: "AccountBalance" })?.text), 0));
|
||||
// The screen this frame shows, or null when it does not show exactly one.
|
||||
// Android's hierarchy dump carries the outgoing and the incoming screen
|
||||
// together on better than one frame in five, and such a frame is a navigation
|
||||
// transition: evidence about neither screen. Ranking the markers and returning
|
||||
// the first one found is how a spec convicts itself on a half-drawn screen.
|
||||
const routeOf = (s: State): Route | null => {
|
||||
let shown: Route | null = null;
|
||||
for (const [name, tag] of Object.entries(SCREENS) as [Route, string][]) {
|
||||
if (!s.ax.find({ testTag: tag })) continue;
|
||||
if (shown !== null) return null;
|
||||
shown = name;
|
||||
}
|
||||
return shown;
|
||||
};
|
||||
const route = extract<Route | null>("route", routeOf);
|
||||
|
||||
// Home's own TOTAL BALANCE node, which the app computes over every account.
|
||||
// Summing the AccountCard balances reads only the cards laid out inside the
|
||||
// viewport, and a card clipped at the bottom edge looks exactly like money
|
||||
// moving. Off Home there is nothing to read, so the last total we did read is
|
||||
// carried forward and `previous` and `current` stay on the same scale.
|
||||
let lastHomeTotal: number | null = null;
|
||||
const totalBalance = extract<number | null>("totalBalance", s => {
|
||||
if (routeOf(s) !== "home") return lastHomeTotal;
|
||||
const total = parseDollarCents(
|
||||
s.ax.find([{ testTag: "HomeScreen" }, { testTag: "TotalBalance" }])?.text);
|
||||
if (total === null) return null; // unreadable is unknown; the carrier keeps its value
|
||||
lastHomeTotal = total;
|
||||
return total;
|
||||
});
|
||||
```
|
||||
|
||||
Every Folio screen and control carries a `testTag`. Compose exposes it as the resource-id on Android and the accessibility identifier on iOS, so one selector resolves on both. `extract` runs against the live tree each step; properties and actions read `.current` and `.previous`, never the raw state.
|
||||
Every Folio screen and control carries a `testTag`. Compose exposes it as the resource-id on Android and the accessibility identifier on iOS, so one selector resolves on both. `extract` runs against the live tree each step; properties and actions read `.current` and `.previous`, never the raw state. An extractor may not read another extractor's handle, which is why `totalBalance` calls `routeOf` rather than `route.current`.
|
||||
|
||||
A second property states what new accounts must look like: a freshly created account starts at zero.
|
||||
A second property states what new accounts must look like: a freshly created account starts at zero. The work is in naming the account it judges.
|
||||
|
||||
```ts
|
||||
const newAccountBalanceIsZero = always(
|
||||
next(() => {
|
||||
const before = new Set((accounts.previous ?? []).map(a => a.name));
|
||||
return accounts.current
|
||||
.filter(a => !before.has(a.name))
|
||||
.every(a => a.balance === 0);
|
||||
if (route.current !== "home") return true;
|
||||
if (!isAddAccountSubmitTap(lastAction.current)) return true;
|
||||
const typed = accountNameField.previous?.text?.trim();
|
||||
const before = accounts.previous ?? null;
|
||||
const after = accounts.current;
|
||||
if (!typed || before === null || after === null) return true;
|
||||
// The only card attributable to a creation is the one named what the fuzzer
|
||||
// typed, on the step its submit landed. Anything else that turned up is a
|
||||
// card that scrolled into view, not an account that came into existence.
|
||||
const matches = after.filter(a => a.name.endsWith(typed));
|
||||
if (matches.length !== 1) return true;
|
||||
const created = matches[0];
|
||||
if (before.some(a => a.name === created.name)) return true;
|
||||
return created.balance === null || created.balance === 0;
|
||||
})
|
||||
);
|
||||
```
|
||||
|
||||
Diffing the two account lists and judging whatever is new is the version that reads better and does not work: Home lists the accounts that fit the viewport, so a card that scrolls in is indistinguishable from an account that was just created. That version convicted this property on android over a Travel account holding $24,112.00.
|
||||
|
||||
A third property, `submitCommitsOneTransactionPerAction`, states the same bug without arithmetic: over any window, no more transactions may be committed than there were submit actions. It needs no amount and no float comparison, so it survives a window of any width, and it does most of the detecting in practice.
|
||||
|
||||
## Reaching the screens that matter
|
||||
|
||||
A fuzzer that pokes at random never logs in, and never reaches a transaction form. The spec gives sanderling enough to drive the real flows, no more.
|
||||
@@ -125,7 +175,7 @@ export const actionsRoot = weighted(
|
||||
);
|
||||
```
|
||||
|
||||
That is the whole input. Two invariants, a way in, and a weighted sense of where to spend time. Nothing here names the bug.
|
||||
That is the whole input. Three invariants, a way in, and a weighted sense of where to spend time. Nothing here names the bug.
|
||||
|
||||
## What the run does
|
||||
|
||||
|
||||
+4
-2
@@ -22,11 +22,15 @@ Run a spec against an app for a fixed duration.
|
||||
| `--ios-device` | optional (ios) | iOS target: a simulator name/UDID to boot, or a connected device's name, UDID, or CoreDevice id. |
|
||||
| `--ios-app-path` | optional (ios) | Path to the `.app` bundle for clear-state reinstall (simulator via `simctl`, device via `devicectl`). |
|
||||
| `--duration` | `5m` | Total test duration (`30s`, `5m`, `2h`, `1d`). |
|
||||
| `--max-steps` | `0` | Stop after this many steps (`0` = no cap, the duration governs). A step budget is what makes two generators comparable. |
|
||||
| `--exit-on-violation` | `false` | Stop the run at the first property violation and exit `2`. |
|
||||
| `--seed` | `0` | PRNG seed. `0` uses a random seed and records it in `meta.json`. |
|
||||
| `--generator` | `seeded` | Who picks each action: `seeded` (the run's PRNG) or `llm` (a vision model). See [the LLM generator](../spec-language/#llm-generator). |
|
||||
| `--output` | `./runs` | Output directory for traces. |
|
||||
| `--clear-data` | `true` | Clear app data before launching so the run starts from a fresh install. Pass `--clear-data=false` to resume prior state. |
|
||||
|
||||
Exit codes: `0` the run finished (violations, if any, are in the summary), `2` the run stopped on a violation under `--exit-on-violation`, `1` something went wrong. CI reads the difference between `2` and `1` to tell a found bug from a broken harness.
|
||||
|
||||
## `sanderling replay [run-or-runs-dir]`
|
||||
|
||||
Serve a local web UI for browsing traces. The positional argument is optional and may point at either a runs directory (the parent of many runs) or a single run directory (auto-detected by the presence of `meta.json`). Defaults to `./runs`.
|
||||
@@ -63,7 +67,5 @@ Print the CLI version.
|
||||
## Flags coming in v0.1.0
|
||||
|
||||
- `--permissions` to pre-set OS-level permissions (for example `--permissions location=allow,notifications=deny`).
|
||||
- `--max-steps` hard cap on step count.
|
||||
- `--exit-on-violation` stop the run on the first property violation.
|
||||
|
||||
Tracked in the [v0.1.0 milestone](https://github.com/priyanshujain/sanderling/milestone/1).
|
||||
@@ -80,6 +80,8 @@ The loop runs until the duration you set elapses. Every step is written to a tra
|
||||
|
||||
Kotlin Multiplatform apps need nothing special: the Android build is tested through the Android driver and the iOS build through the iOS driver.
|
||||
|
||||
Canvas-rendered web apps are the exception. On web, `InputText` types through the DevTools Protocol's `Input.insertText`, which needs a focused DOM element to deliver to, and a bare `<canvas>` is not one: the call produces no events at all and the app sees nothing typed. The driver cannot work around that, because there is no event target. Compose Multiplatform for Web is fine because it keeps a hidden `<input>` IME proxy in its shadow root, a real focusable element that receives the text. A canvas app that exposes no such element is tap-fuzzable but not text-fuzzable.
|
||||
|
||||
## Reading this manual
|
||||
|
||||
- [Case study: Folio](../case-study/) follows sanderling finding a real bug in the example app, and shows how its spec is written.
|
||||
|
||||
+12
-2
@@ -72,7 +72,12 @@ fun HomeScreen(state: HomeUiState, onEvent: (HomeEvent) -> Unit) {
|
||||
Row(verticalAlignment = Alignment.Bottom) {
|
||||
Column {
|
||||
Text("TOTAL BALANCE", style = Type.label, color = t.textMuted)
|
||||
Text(formatCents(total), style = Type.balance, color = t.text)
|
||||
Text(
|
||||
formatCents(total),
|
||||
style = Type.balance,
|
||||
color = t.text,
|
||||
modifier = Modifier.testTag("TotalBalance"),
|
||||
)
|
||||
}
|
||||
Box(Modifier.weight(1f))
|
||||
Text(
|
||||
@@ -153,7 +158,12 @@ private fun AccountCard(
|
||||
overflow = TextOverflow.Ellipsis,
|
||||
modifier = Modifier.testTag("AccountName"),
|
||||
)
|
||||
Text(txnLabel, style = Type.caption, color = t.textMuted)
|
||||
Text(
|
||||
txnLabel,
|
||||
style = Type.caption,
|
||||
color = t.textMuted,
|
||||
modifier = Modifier.testTag("AccountTxnCount"),
|
||||
)
|
||||
}
|
||||
Text(
|
||||
formatCents(balance),
|
||||
|
||||
@@ -1,62 +1,505 @@
|
||||
// Computes the Home-screen total balance from the visible AccountCard balances.
|
||||
// When no Home cards are visible (Ledger, AddTransaction, etc.) the carrier
|
||||
// value is returned so the property compares apples to apples across screen
|
||||
// transitions. The Ledger LedgerBalance is intentionally ignored because it is
|
||||
// a single-account number on a different scale than the Home multi-account sum.
|
||||
export function computeHomeTotalBalance(args: {
|
||||
cardBalanceTexts: (string | undefined)[];
|
||||
previousCarrier: number;
|
||||
}): number {
|
||||
const { cardBalanceTexts, previousCarrier } = args;
|
||||
if (cardBalanceTexts.length === 0) return previousCarrier;
|
||||
return cardBalanceTexts.reduce((sum, text) => sum + parseDollarCents(text), 0);
|
||||
// The screen a frame shows, or null when it does not show exactly one.
|
||||
//
|
||||
// `present` answers whether a screen's marker node is in the accessibility
|
||||
// tree. Android puts two markers there constantly: its hierarchy dump carries
|
||||
// the outgoing and the incoming screen together on 425 of 1879 steps measured
|
||||
// across 17 runs, better than one frame in five, occasionally three at once.
|
||||
// iOS produced none in 2558 steps and web one in 560, so this is not a rule
|
||||
// invented for a hypothetical.
|
||||
//
|
||||
// Such a frame is a navigation transition, and it is evidence about neither
|
||||
// screen. Ranking the markers and taking the first is what convicted folio's
|
||||
// spec in 11 android runs: `route` answered add-transaction off the screen
|
||||
// being torn down while a separate unscoped find answered "on Home" off the one
|
||||
// being built, so a half-rendered Home counted as a fresh balance reading and
|
||||
// reset the submit window under a tap that had not landed yet. Every reading
|
||||
// below takes the route as its only input, so there is no second answer left
|
||||
// for it to disagree with.
|
||||
//
|
||||
// The engine already draws this line: a transitional tree yields no builtin
|
||||
// targets at all (see targets() in internal/verifier), so a spec that keeps
|
||||
// picking one of the two screens is the only thing still handing out
|
||||
// coordinates from an animation.
|
||||
export function routeOfFrame<R extends string>(
|
||||
screens: Record<R, string>,
|
||||
present: (tag: string) => boolean,
|
||||
): R | null {
|
||||
let shown: R | null = null;
|
||||
for (const route of Object.keys(screens) as R[]) {
|
||||
if (!present(screens[route])) continue;
|
||||
if (shown !== null) return null;
|
||||
shown = route;
|
||||
}
|
||||
return shown;
|
||||
}
|
||||
|
||||
// Parses formatCents output like "$5.00", "-$1,234.56", "+$0.50" back to integer cents.
|
||||
function parseDollarCents(text: string | undefined): number {
|
||||
if (!text) return 0;
|
||||
const sign = text.startsWith("-") ? -1 : 1;
|
||||
const digits = text.replace(/[^0-9]/g, "");
|
||||
return digits ? sign * parseInt(digits, 10) : 0;
|
||||
// A reading taken once per frame, keyed on the identity of the state object the
|
||||
// reading came off.
|
||||
//
|
||||
// Both hosts build a NEW state object per step and hand that one object to
|
||||
// every extractor: goja's PushSnapshot builds it through stateObject
|
||||
// (internal/verifier/worker.go), the web runtime's evaluateExtractors through
|
||||
// buildState (pkg/spec/src/web-runtime.ts). So the object IS the frame, and a
|
||||
// value cached against it cannot outlive the frame it was read on. Holding the
|
||||
// reference is what keeps that true rather than merely likely: the cached state
|
||||
// cannot be collected, so no later step's state can be the same object.
|
||||
//
|
||||
// Worth having because these readings walk the whole tree. On web every
|
||||
// `find` is a querySelectorAll across the document and each shadow root
|
||||
// beneath it, and the spec takes a dozen of them per step off one frame.
|
||||
export function oncePerFrame<S extends object, T>(read: (frame: S) => T): (frame: S) => T {
|
||||
let last: { frame: S; value: T } | null = null;
|
||||
return frame => {
|
||||
if (last === null || last.frame !== frame) last = { frame, value: read(frame) };
|
||||
return last.value;
|
||||
};
|
||||
}
|
||||
|
||||
// Reads the Home screen's own TOTAL BALANCE node and advances the carrier the
|
||||
// spec holds between Home visits.
|
||||
//
|
||||
// The app computes that number over ALL accounts, so it does not care which
|
||||
// account cards happen to be laid out inside the viewport. Summing the visible
|
||||
// AccountCard balances did: a card clipped at the bottom edge exposes no
|
||||
// AccountBalance child, and a partial sum looks exactly like money moving.
|
||||
//
|
||||
// Three cases, and the difference between the reported value and the carrier is
|
||||
// the whole point:
|
||||
// - anywhere but Home there is nothing to read, so the last total we actually
|
||||
// read is reported and carried on unchanged. A transition frame is one of
|
||||
// those places: its route is null, which is not "home";
|
||||
// - on Home with an unreadable total the reading is UNKNOWN, so null is
|
||||
// reported (the property treats null as vacuous) while the carrier keeps
|
||||
// the last value we did read. Writing null into the carrier is what used to
|
||||
// poison every later step: off-Home steps hand the carrier back, so one
|
||||
// unreadable Home turned the property vacuous for the rest of the run;
|
||||
// - on Home with a readable total, that total is both the reading and the new
|
||||
// carrier, and `fresh` says the comparison window closes here.
|
||||
export interface HomeTotalReading {
|
||||
value: number | null;
|
||||
carrier: number | null;
|
||||
fresh: boolean;
|
||||
}
|
||||
|
||||
export function readHomeTotalBalance(args: {
|
||||
route: string | null;
|
||||
totalText: string | undefined;
|
||||
previousCarrier: number | null;
|
||||
}): HomeTotalReading {
|
||||
const { route, totalText, previousCarrier } = args;
|
||||
if (route !== "home") return { value: previousCarrier, carrier: previousCarrier, fresh: false };
|
||||
const total = parseDollarCents(totalText);
|
||||
if (total === null) return { value: null, carrier: previousCarrier, fresh: false };
|
||||
return { value: total, carrier: total, fresh: true };
|
||||
}
|
||||
|
||||
// The same reading, taken off Home's account cards instead of its total, for
|
||||
// the readings the spec carries between Home visits (the account list and the
|
||||
// per-account transaction counts).
|
||||
//
|
||||
// The empty reading is the whole point. Android renders Home's own node a frame
|
||||
// or two before its list, so `findAll` over the cards comes back empty while the
|
||||
// screen is already claiming to be Home. That is UNKNOWN, not "no accounts":
|
||||
// writing it into the carrier is the poisoning readHomeTotalBalance was already
|
||||
// fixed for. It killed the counting invariant outright on android, where
|
||||
// counts_prev was {} at every evaluation point of all 17 runs measured.
|
||||
//
|
||||
// Callers pass null for a reading they could not take, so an empty list and an
|
||||
// empty map are one case here rather than two.
|
||||
export interface HomeCardReading<T> {
|
||||
value: T | null;
|
||||
carrier: T | null;
|
||||
fresh: boolean;
|
||||
}
|
||||
|
||||
export function readHomeCards<T>(args: {
|
||||
route: string | null;
|
||||
reading: T | null;
|
||||
previousCarrier: T | null;
|
||||
}): HomeCardReading<T> {
|
||||
const { route, reading, previousCarrier } = args;
|
||||
if (route !== "home") return { value: previousCarrier, carrier: previousCarrier, fresh: false };
|
||||
if (reading === null) return { value: null, carrier: previousCarrier, fresh: false };
|
||||
return { value: reading, carrier: reading, fresh: true };
|
||||
}
|
||||
|
||||
function isTapOn(
|
||||
lastAction: { kind?: string; on?: string | object } | null,
|
||||
target: string,
|
||||
): boolean {
|
||||
if (lastAction == null) return false;
|
||||
if (lastAction.kind !== "Tap" && lastAction.kind !== "DoubleTap") return false;
|
||||
const on = lastAction.on;
|
||||
const onString = typeof on === "string" ? on : on != null ? JSON.stringify(on) : "";
|
||||
return onString.includes(target);
|
||||
}
|
||||
|
||||
// Is this the action that commits a transaction? Nothing else in Folio reaches
|
||||
// Repository.createTransaction: AddTransactionViewModel.submit() is the only
|
||||
// caller, AddTransactionEvent.Submit is the only thing that runs it, and the
|
||||
// TxnSubmit button's onClick is the only thing that sends that event.
|
||||
export function isTxnSubmitTap(
|
||||
lastAction: { kind?: string; on?: string | object } | null,
|
||||
): boolean {
|
||||
return isTapOn(lastAction, "TxnSubmit");
|
||||
}
|
||||
|
||||
// Likewise the only action that creates an account.
|
||||
export function isAddAccountSubmitTap(
|
||||
lastAction: { kind?: string; on?: string | object } | null,
|
||||
): boolean {
|
||||
return isTapOn(lastAction, "AddAccountSubmit");
|
||||
}
|
||||
|
||||
// Counts the submit actions inside the window the balance property compares
|
||||
// over: from the last Home total we read to this step, inclusive of this step's
|
||||
// action.
|
||||
//
|
||||
// Counting ACTIONS rather than transactions is deliberate. A double-tap is one
|
||||
// action that commits two transactions, which is precisely the bug, so a rule
|
||||
// phrased in transactions could not tell the bug apart from two healthy
|
||||
// submits. A rule phrased in actions can: one action in the window means the
|
||||
// whole balance delta belongs to that action, and comparing it against the
|
||||
// amount typed for it is fair.
|
||||
//
|
||||
// The reset lands on `fresh`, the same event that advances the carrier, so the
|
||||
// count always describes exactly the interval the two compared totals span.
|
||||
export function countSubmitsInWindow(args: {
|
||||
previousCount: number;
|
||||
lastAction: { kind?: string; on?: string | object } | null;
|
||||
fresh: boolean;
|
||||
}): { reported: number; next: number } {
|
||||
const { previousCount, lastAction, fresh } = args;
|
||||
const reported = previousCount + (isTxnSubmitTap(lastAction) ? 1 : 0);
|
||||
return { reported, next: fresh ? 0 : reported };
|
||||
}
|
||||
|
||||
// Parses formatCents output like "$5.00", "-$1,234.56", "+$0.50" back to
|
||||
// integer cents. Anything that is not a complete amount is null, not 0: a
|
||||
// balance we could not read is unknown, and reading it as zero silently moves
|
||||
// the Home total.
|
||||
export function parseDollarCents(text: string | undefined): number | null {
|
||||
if (!text) return null;
|
||||
const match = text.trim().match(/^([-+]?)\$?(\d{1,3}(?:,\d{3})*|\d+)\.(\d{2})$/);
|
||||
if (!match) return null;
|
||||
const [, sign, dollars, cents] = match;
|
||||
if (dollars === undefined || cents === undefined) return null;
|
||||
return (sign === "-" ? -1 : 1) * (parseInt(dollars.replace(/,/g, ""), 10) * 100 + parseInt(cents, 10));
|
||||
}
|
||||
|
||||
// Compose Multiplatform for Web merges an AccountCard's whole subtree into one
|
||||
// accessibility node: the AccountName and AccountBalance children that Android
|
||||
// and iOS expose do not exist there, and the card's own text is
|
||||
// initials + name + "N transaction(s)" + balance run together with no
|
||||
// separator, e.g. "INInvestments12 transactions$2,589.00". The two helpers
|
||||
// below prefer the structured child and only parse the merged text when the
|
||||
// platform did not give us one.
|
||||
|
||||
// The balance is the amount at the very END of the card text. Anchoring there
|
||||
// is the whole trick: digit-scraping the merged string would swallow the "12"
|
||||
// of "12 transactions" into the amount.
|
||||
const TRAILING_BALANCE = /[-+]?\$[\d,]+\.\d{2}\s*$/;
|
||||
|
||||
// The transaction-count label sits between the name and the balance. The group
|
||||
// captures the digit run so the count can be read from the same match.
|
||||
const TRAILING_TXN_COUNT = /(\d+)\s*transactions?\s*$/;
|
||||
|
||||
export function cardBalanceText(args: {
|
||||
childText: string | undefined;
|
||||
cardText: string | undefined;
|
||||
}): string | undefined {
|
||||
const { childText, cardText } = args;
|
||||
if (childText) return childText;
|
||||
const match = cardText?.match(TRAILING_BALANCE);
|
||||
return match ? match[0].trim() : undefined;
|
||||
}
|
||||
|
||||
// The result is an identity key, not a display name: off web it is the
|
||||
// AccountName text, on web it is whatever the merged card text leaves in front
|
||||
// of the count label, initials and all ("T2Travel" for "Travel 2024"). Its only
|
||||
// consumer compares it for set membership. Keep it stable, not pretty: the
|
||||
// count label always starts at the first digit of the run before it, so the
|
||||
// key does not move as an account's transaction count grows.
|
||||
export function cardAccountName(args: {
|
||||
childText: string | undefined;
|
||||
cardText: string | undefined;
|
||||
}): string {
|
||||
const { childText, cardText } = args;
|
||||
if (childText) return childText;
|
||||
if (!cardText) return "";
|
||||
const head = cardText.replace(TRAILING_BALANCE, "");
|
||||
const label = head.match(TRAILING_TXN_COUNT);
|
||||
if (!label || label.index === undefined) return head.trim();
|
||||
return head.slice(0, label.index).trim();
|
||||
}
|
||||
|
||||
// One card's transaction count, in the strongest form its SOURCE supports. The
|
||||
// two forms are the whole reason this is not just a number:
|
||||
//
|
||||
// - a NUMBER when the count came from the card's own AccountTxnCount node.
|
||||
// That node's text is the count label and nothing else, so its digits are
|
||||
// the count and subtracting two readings is exact arithmetic;
|
||||
// - the digit RUN as TEXT when the count had to be recovered from merged card
|
||||
// text. Web merges the AccountCard subtree into one node, and an account
|
||||
// whose name ends in digits runs them into the count: the account named
|
||||
// "-1" holding 2 transactions merges to "-1-12 transactions", whose maximal
|
||||
// digit run reads 12. That prefix is fixed for a given account, so two
|
||||
// readings whose runs are the SAME LENGTH still differ by exactly the true
|
||||
// difference (19 to 120 is impossible; 19 to 110 is a length change), while
|
||||
// two of different lengths do not: 9 to 10 reads as 19 to 110, a delta of
|
||||
// 91 out of a delta of 1. Keeping the run as text is what lets
|
||||
// committedTransactionsExceedSubmits see the length change and drop the
|
||||
// pair instead of convicting on it.
|
||||
//
|
||||
// Carrying the distinction is what keeps that length rule where it belongs.
|
||||
// There is no name in front of a dedicated node's digits for it to protect
|
||||
// against, so applied there it buys nothing and throws away real evidence every
|
||||
// time an account crosses a decade: of the three android seed-9 runs that
|
||||
// finished without convicting, two had dropped a window on this rule, one
|
||||
// reading 7 against 12 and the other 4 against 10.
|
||||
//
|
||||
// This is a fact about the reading, not about the platform. A platform that
|
||||
// starts exposing the node gets exact counts by exposing it, and one that stops
|
||||
// falls back to the text rule on the same step it stops.
|
||||
export type TxnCount = number | string;
|
||||
|
||||
export function cardTxnCount(args: {
|
||||
childText: string | undefined;
|
||||
cardText: string | undefined;
|
||||
}): TxnCount | undefined {
|
||||
const { childText, cardText } = args;
|
||||
const source = childText ?? cardText?.replace(TRAILING_BALANCE, "");
|
||||
const digits = source?.match(TRAILING_TXN_COUNT)?.[1];
|
||||
if (digits === undefined) return undefined;
|
||||
return childText === undefined ? digits : parseInt(digits, 10);
|
||||
}
|
||||
|
||||
export interface Account {
|
||||
// Identity key, not a display name: on web it carries the card's initials.
|
||||
name: string;
|
||||
// null when the card's balance could not be read at all (see cardBalanceText).
|
||||
balance: number | null;
|
||||
}
|
||||
|
||||
// One Home card, already parsed by the three helpers above.
|
||||
export interface CardReading extends Account {
|
||||
count: TxnCount | undefined;
|
||||
}
|
||||
|
||||
// The two readings Home's card list yields, each null when there is nothing in
|
||||
// it to read. Both feed readHomeCards, which is where null stops the carrier
|
||||
// from being overwritten.
|
||||
//
|
||||
// An account list is empty for two reasons a frame cannot tell apart: the app
|
||||
// has no accounts, or it has not drawn them yet. Calling both unknown costs one
|
||||
// detection, the very first account of a run, and buys back every card that
|
||||
// arrived late.
|
||||
export function homeAccountsOf(cards: readonly CardReading[]): Account[] | null {
|
||||
if (cards.length === 0) return null;
|
||||
return cards.map(({ name, balance }) => ({ name, balance }));
|
||||
}
|
||||
|
||||
// A card whose name or count is unreadable is left out rather than guessed at;
|
||||
// committedTransactionsExceedSubmits treats a missing account as no evidence.
|
||||
// Every card being unreadable leaves nothing to compare, which is unknown.
|
||||
//
|
||||
// A name carried by more than one card is left out for the same reason, the
|
||||
// rule createdAccountHasNonZeroBalance applies with `matches.length === 1`:
|
||||
// nothing here can say which of them a count came from. Folio accepts the same
|
||||
// account name twice and Home lists whatever fits the viewport, so a reading
|
||||
// that saw one Travel card and a later one that saw two would otherwise
|
||||
// subtract two DIFFERENT accounts' counts and convict a healthy app of
|
||||
// double-submitting. The twin does not have to be readable to spoil the
|
||||
// identity, so duplicates are counted over every card, not just the usable
|
||||
// ones. Dropping a card can only ever cost a detection.
|
||||
export function homeTxnCountsOf(cards: readonly CardReading[]): Record<string, TxnCount> | null {
|
||||
const cardsPerName = new Map<string, number>();
|
||||
for (const card of cards) cardsPerName.set(card.name, (cardsPerName.get(card.name) ?? 0) + 1);
|
||||
const counts: Record<string, TxnCount> = {};
|
||||
for (const card of cards) {
|
||||
if (card.name === "" || card.count === undefined) continue;
|
||||
if (cardsPerName.get(card.name) !== 1) continue;
|
||||
counts[card.name] = card.count;
|
||||
}
|
||||
return Object.keys(counts).length === 0 ? null : counts;
|
||||
}
|
||||
|
||||
// Did the account the fuzzer just created come into existence holding money?
|
||||
//
|
||||
// "Appeared in the visible set" is not "was created". Home lists the accounts
|
||||
// that fit the viewport, so an existing account arrives in a later reading
|
||||
// whenever the list scrolls, whenever a card that was clipped becomes laid out,
|
||||
// and (before the route fix) whenever the earlier reading came off a
|
||||
// half-rendered Home mid-transition. All three read as a brand new account, and
|
||||
// the last two convicted this property on android over a Travel account holding
|
||||
// $24,112.00 and a Savings account holding $429,585.00.
|
||||
//
|
||||
// The one appearance that IS attributable to a creation is the account the
|
||||
// fuzzer asked for: the name it typed into AddAccountScreen, judged on the step
|
||||
// where that screen's submit landed on Home. Everything else that shows up is a
|
||||
// card that came into view, and this property has nothing to say about it.
|
||||
//
|
||||
// Returns true only for a violation it can attribute. Unattributable is not the
|
||||
// same as fine, and both come back false here.
|
||||
export function createdAccountHasNonZeroBalance(args: {
|
||||
route: string | null;
|
||||
lastAction: { kind?: string; on?: string | object } | null;
|
||||
typedName: string | undefined;
|
||||
before: Account[] | null;
|
||||
after: Account[] | null;
|
||||
}): boolean {
|
||||
const { route, lastAction, before, after } = args;
|
||||
if (route !== "home") return false;
|
||||
if (!isAddAccountSubmitTap(lastAction)) return false;
|
||||
if (before === null || after === null) return false;
|
||||
const typed = (args.typedName ?? "").trim();
|
||||
if (typed === "") return false;
|
||||
// Web merges the card into one node whose text opens with the avatar
|
||||
// initials, so the identity key is "INInvestments" where android and iOS give
|
||||
// "Investments"; endsWith covers both. Two cards answering to the same typed
|
||||
// name (a second "Travel", or a card the tree exposed twice) leave the
|
||||
// appearance unattributable, so nothing is judged.
|
||||
const matches = after.filter(account => account.name.endsWith(typed));
|
||||
const created = matches.length === 1 ? matches[0] : undefined;
|
||||
if (created === undefined) return false;
|
||||
if (before.some(account => account.name === created.name)) return false;
|
||||
return created.balance !== null && created.balance !== 0;
|
||||
}
|
||||
|
||||
// Every accepted submit commits exactly one transaction, so over any window the
|
||||
// number of transactions committed cannot exceed the number of submit actions
|
||||
// taken. A double-submit is one action committing two, which is the only way to
|
||||
// break it. Rejected submits (parseCents refuses the amount) commit nothing, so
|
||||
// the normal case sits comfortably under the bound.
|
||||
//
|
||||
// Only accounts present in BOTH readings are counted, and only upward movement.
|
||||
// A card that scrolled out of the viewport, or one whose count could not be
|
||||
// read, simply drops out of the sum. That makes the result a LOWER BOUND on the
|
||||
// transactions committed, and a lower bound that already exceeds the submit
|
||||
// count is still a real violation. Missing cards can cost a detection; they
|
||||
// cannot manufacture one.
|
||||
//
|
||||
// Transactions are never deleted (Ledger.sq has no DELETE for them), so a
|
||||
// per-account count only ever rises; the max(0, ...) is defensive, not load
|
||||
// bearing.
|
||||
export function committedTransactionsExceedSubmits(args: {
|
||||
countsBefore: Record<string, TxnCount> | null;
|
||||
countsAfter: Record<string, TxnCount> | null;
|
||||
submitsInWindow: number;
|
||||
}): boolean {
|
||||
const { countsBefore, countsAfter, submitsInWindow } = args;
|
||||
if (countsBefore === null || countsAfter === null) return false;
|
||||
if (!Number.isSafeInteger(submitsInWindow)) return false;
|
||||
let committed = 0;
|
||||
for (const name of Object.keys(countsAfter)) {
|
||||
const before = countsBefore[name];
|
||||
const after = countsAfter[name];
|
||||
if (before === undefined || after === undefined) continue;
|
||||
const rise = countRise(before, after);
|
||||
if (rise !== null && rise > 0) committed += rise;
|
||||
}
|
||||
return committed > submitsInWindow;
|
||||
}
|
||||
|
||||
// How far one account's count rose between two readings, or null when the pair
|
||||
// is not comparable. Not comparable is not zero: the account drops out of the
|
||||
// sum entirely, which can only cost a detection.
|
||||
function countRise(before: TxnCount, after: TxnCount): number | null {
|
||||
if (typeof before === "number" && typeof after === "number") {
|
||||
if (!Number.isSafeInteger(before) || !Number.isSafeInteger(after)) return null;
|
||||
return after - before;
|
||||
}
|
||||
// Recovered from merged card text, where the run may carry an account-name
|
||||
// prefix, so only equal-length runs subtract to the true difference: see
|
||||
// TxnCount. A pair whose two readings came from different sources is one
|
||||
// neither rule can vouch for, and it is dropped with them.
|
||||
if (typeof before !== "string" || typeof after !== "string") return null;
|
||||
if (before.length !== after.length) return null;
|
||||
const from = parseInt(before, 10);
|
||||
const to = parseInt(after, 10);
|
||||
if (!Number.isSafeInteger(from) || !Number.isSafeInteger(to)) return null;
|
||||
return to - from;
|
||||
}
|
||||
|
||||
// Parses raw user input in the transaction amount field into integer cents.
|
||||
// Mirrors the Folio app's parseCents: whole numbers like "50" become 5000
|
||||
// cents, decimals like "5.50" become 550, more than 2 decimals or non-numeric
|
||||
// input return 0. Leading +/- signs are tolerated and treated as positive.
|
||||
// Mirrors the Folio app's parseCents (app/shared/.../util/Format.kt): whole
|
||||
// numbers like "50" become 5000 cents, decimals like "5.50" become 550, more
|
||||
// than 2 decimals or non-numeric input return 0. A sign is NOT tolerated,
|
||||
// because parseCents does not tolerate one either: it rejects "-50" outright,
|
||||
// so the app creates no transaction, and reading the field as a real amount
|
||||
// would make the property demand a balance move that never happened.
|
||||
//
|
||||
// 0 also means "an amount this reading cannot represent". parseCents returns
|
||||
// null once the whole part no longer fits a Kotlin Long, and its cents fit a
|
||||
// Long where ours are float64 doubles, exact only to Number.MAX_SAFE_INTEGER.
|
||||
// Past that the digits we compute are not the digits the app applied, so the
|
||||
// reading is unknown rather than large, and 0 routes it to the property's
|
||||
// vacuous branch.
|
||||
export function parseTypedAmount(text: string | undefined | null): number {
|
||||
if (!text) return 0;
|
||||
let trimmed = text.trim().replace(/,/g, "");
|
||||
if (trimmed.startsWith("+") || trimmed.startsWith("-")) trimmed = trimmed.slice(1);
|
||||
const trimmed = text.trim().replace(/,/g, "");
|
||||
if (!/^\d+(\.\d{1,2})?$/.test(trimmed)) return 0;
|
||||
const dot = trimmed.indexOf(".");
|
||||
const whole = dot < 0 ? trimmed : trimmed.slice(0, dot);
|
||||
const frac = dot < 0 ? "" : trimmed.slice(dot + 1);
|
||||
const fracPadded = (frac + "00").slice(0, 2);
|
||||
return parseInt(whole, 10) * 100 + parseInt(fracPadded, 10);
|
||||
const cents = parseInt(whole, 10) * 100 + parseInt(fracPadded, 10);
|
||||
return Number.isSafeInteger(cents) ? cents : 0;
|
||||
}
|
||||
|
||||
// When the last action is a tap (or double-tap) on the transaction Submit
|
||||
// button, the absolute change in total balance must equal the amount the
|
||||
// user typed. A double-submit lands two transactions and shifts the balance
|
||||
// by 2x the typed amount, tripping this check. The route gate skips steps
|
||||
// whose landing screen is not Home: totalBalance is only freshly computed
|
||||
// from visible AccountCards on Home, so off-Home comparisons would read a
|
||||
// stale carrier value and false-fire.
|
||||
// whose landing screen is not Home: totalBalance is only freshly read from
|
||||
// Home's own TOTAL BALANCE node, so off-Home comparisons would read a stale
|
||||
// carrier value and false-fire.
|
||||
//
|
||||
// submitsInWindow is what keeps the comparison honest. prevTotalBalance is the
|
||||
// last total we READ, not the total as of the previous transaction, so the two
|
||||
// compared numbers can straddle any number of commits: a real run produced a
|
||||
// 13000 delta against a typed 19600 because the window held a double-submit's
|
||||
// two 19600 debits AND an unrelated 26200 credit. A delta like that is not
|
||||
// evidence about the amount typed into any one submit, so anything other than
|
||||
// exactly one submit action in the window is vacuous. Exactly one still catches
|
||||
// the bug: the double-tap is a single action.
|
||||
export function submitChangesBalanceByTypedAmount(args: {
|
||||
route: string | null;
|
||||
lastAction: { kind?: string; on?: string | object } | null;
|
||||
submitsInWindow: number;
|
||||
typedAmount: number;
|
||||
prevTotalBalance: number;
|
||||
currTotalBalance: number;
|
||||
prevTotalBalance: number | null;
|
||||
currTotalBalance: number | null;
|
||||
}): boolean {
|
||||
const { route, lastAction, typedAmount, prevTotalBalance, currTotalBalance } = args;
|
||||
const { route, lastAction, submitsInWindow, typedAmount } = args;
|
||||
const { prevTotalBalance, currTotalBalance } = args;
|
||||
|
||||
if (route !== "home") return true;
|
||||
if (lastAction == null) return true;
|
||||
if (lastAction.kind !== "Tap" && lastAction.kind !== "DoubleTap") return true;
|
||||
const on = lastAction.on;
|
||||
const onString = typeof on === "string" ? on : on != null ? JSON.stringify(on) : "";
|
||||
if (!onString.includes("TxnSubmit")) return true;
|
||||
if (!isTxnSubmitTap(lastAction)) return true;
|
||||
if (submitsInWindow !== 1) return true;
|
||||
if (typedAmount === 0) return true;
|
||||
// An unknown total on either side is not evidence of anything. Comparing one
|
||||
// would turn every unreadable Home into a violation.
|
||||
if (prevTotalBalance === null || currTotalBalance === null) return true;
|
||||
// Same rule for a total we cannot hold exactly. These are integer cents in
|
||||
// float64, exact only up to Number.MAX_SAFE_INTEGER. The app caps a
|
||||
// transaction at whatever fits a Kotlin Long, so a balance of ~1e18 cents is
|
||||
// one accepted amount away, and up there the gap between representable
|
||||
// values is 128 cents: a real 1600-cent move reads back as something else
|
||||
// entirely. The equality below is then false for a healthy single submit
|
||||
// exactly as readily as for a double one, and a check that cannot pass is not
|
||||
// a check that failed.
|
||||
//
|
||||
// The three guarded quantities are the ones that lose precision on their own:
|
||||
// each balance, because the number parsed out of the card text stops being
|
||||
// the number the card shows, and the typed amount, because parseTypedAmount
|
||||
// multiplies by 100. Their difference needs no guard of its own: IEEE
|
||||
// subtraction of two exact integers is correctly rounded, so if the real
|
||||
// difference equals a safe typedAmount it is itself safe and comes out exact,
|
||||
// and if it does not, it cannot round INTO a safe typedAmount either. Adding
|
||||
// a magnitude check there would only silence honest mismatches.
|
||||
if (!Number.isSafeInteger(prevTotalBalance)) return true;
|
||||
if (!Number.isSafeInteger(currTotalBalance)) return true;
|
||||
if (!Number.isSafeInteger(typedAmount)) return true;
|
||||
return Math.abs(currTotalBalance - prevTotalBalance) === typedAmount;
|
||||
}
|
||||
@@ -9,91 +9,203 @@ import {
|
||||
weighted,
|
||||
whenRoute,
|
||||
} from "@sanderling/spec";
|
||||
import type { AccessibilityElement, State } from "@sanderling/spec";
|
||||
import { defaultActions, doubleTaps } from "@sanderling/spec/defaults";
|
||||
import {
|
||||
computeHomeTotalBalance,
|
||||
cardAccountName,
|
||||
cardBalanceText,
|
||||
cardTxnCount,
|
||||
committedTransactionsExceedSubmits,
|
||||
countSubmitsInWindow,
|
||||
createdAccountHasNonZeroBalance,
|
||||
homeAccountsOf,
|
||||
homeTxnCountsOf,
|
||||
oncePerFrame,
|
||||
parseDollarCents,
|
||||
parseTypedAmount,
|
||||
readHomeCards,
|
||||
readHomeTotalBalance,
|
||||
routeOfFrame,
|
||||
submitChangesBalanceByTypedAmount,
|
||||
} from "./predicates";
|
||||
import type { Account, CardReading, TxnCount } from "./predicates";
|
||||
|
||||
interface Account {
|
||||
name: string;
|
||||
balance: number;
|
||||
}
|
||||
// Screen markers, and the route each one names. Detection is by testTag
|
||||
// (resource-id on Android, accessibilityIdentifier on iOS).
|
||||
const SCREENS = {
|
||||
login: "LoginScreen",
|
||||
"add-account": "AddAccountScreen",
|
||||
"add-transaction": "AddTransactionScreen",
|
||||
ledger: "LedgerScreen",
|
||||
home: "HomeScreen",
|
||||
} as const;
|
||||
type Route = keyof typeof SCREENS;
|
||||
|
||||
// Parses formatCents output like "$5.00", "-$1,234.56", "+$0.50" back to integer cents.
|
||||
function parseDollarCents(text: string | undefined): number {
|
||||
if (!text) return 0;
|
||||
const sign = text.startsWith("-") ? -1 : 1;
|
||||
const digits = text.replace(/[^0-9]/g, "");
|
||||
return digits ? sign * parseInt(digits, 10) : 0;
|
||||
}
|
||||
// The screen this frame shows, or null when it does not show exactly one: see
|
||||
// routeOfFrame, which owns that rule and the reason for it. Everything below
|
||||
// takes its answer from here, so no two readings can disagree about which
|
||||
// screen the app is on. Every extractor asks, so the answer is read once per
|
||||
// frame: see oncePerFrame for why that stays fresh.
|
||||
const routeOf = oncePerFrame(
|
||||
(s: State): Route | null =>
|
||||
routeOfFrame<Route>(SCREENS, tag => s.ax.find({ testTag: tag }) != null),
|
||||
);
|
||||
|
||||
// Route detection via testTag (resource-id on Android, accessibilityIdentifier on iOS)
|
||||
const loggedIn = extract("loggedIn", s => s.ax.find({ testTag: "LoginScreen" }) == null);
|
||||
const route = extract<string | null>("route", s => {
|
||||
if (s.ax.find({ testTag: "LoginScreen" })) return "login";
|
||||
if (s.ax.find({ testTag: "AddAccountScreen" })) return "add-account";
|
||||
if (s.ax.find({ testTag: "AddTransactionScreen" })) return "add-transaction";
|
||||
if (s.ax.find({ testTag: "LedgerScreen" })) return "ledger";
|
||||
if (s.ax.find({ testTag: "HomeScreen" })) return "home";
|
||||
return null;
|
||||
});
|
||||
// An element is a reading, and a target, only when the route says we are on its
|
||||
// screen. Scoping a find to the screen's own node is not enough on a transition
|
||||
// frame: both screens are in the tree, so the one the app has already left
|
||||
// still resolves and the tap goes to whatever now occupies those pixels.
|
||||
const on =
|
||||
(route: Route, tag: string) =>
|
||||
(s: State): AccessibilityElement | undefined =>
|
||||
routeOf(s) === route ? s.ax.find([{ testTag: SCREENS[route] }, { testTag: tag }]) : undefined;
|
||||
|
||||
// Account cards on Home: identity is the AccountName text; balance comes from AccountBalance.
|
||||
const accounts = extract<Account[]>("accounts", s =>
|
||||
s.ax.findAll([{ testTag: "HomeScreen" }, { testTag: "AccountCard" }]).map(card => ({
|
||||
name: card.find({ testTag: "AccountName" })?.text ?? "",
|
||||
balance: parseDollarCents(card.find({ testTag: "AccountBalance" })?.text),
|
||||
const allOn =
|
||||
(route: Route, tag: string) =>
|
||||
(s: State): AccessibilityElement[] =>
|
||||
routeOf(s) === route ? s.ax.findAll([{ testTag: SCREENS[route] }, { testTag: tag }]) : [];
|
||||
|
||||
const loggedIn = extract("loggedIn", s => routeOf(s) !== "login");
|
||||
const route = extract<Route | null>("route", routeOf);
|
||||
|
||||
// One parse of Home's account cards. Identity comes from AccountName, balance
|
||||
// from AccountBalance, the transaction count from AccountTxnCount. Web exposes
|
||||
// none of those children (the card is one merged node there), so every reading
|
||||
// goes through predicates.ts, which falls back to parsing the card's own text.
|
||||
//
|
||||
// Everything that comes off the card list shares this parse so the readings
|
||||
// cannot disagree with each other about what was on screen.
|
||||
const homeCards = oncePerFrame((s: State): CardReading[] =>
|
||||
allOn("home", "AccountCard")(s).map(card => ({
|
||||
name: cardAccountName({
|
||||
childText: card.find({ testTag: "AccountName" })?.text,
|
||||
cardText: card.text,
|
||||
}),
|
||||
balance: parseDollarCents(
|
||||
cardBalanceText({
|
||||
childText: card.find({ testTag: "AccountBalance" })?.text,
|
||||
cardText: card.text,
|
||||
})),
|
||||
count: cardTxnCount({
|
||||
childText: card.find({ testTag: "AccountTxnCount" })?.text,
|
||||
cardText: card.text,
|
||||
}),
|
||||
})));
|
||||
|
||||
// Total balance: sum of AccountCard balances visible on Home. The carrier
|
||||
// deliberately tracks only the Home multi-account total. Ledger's
|
||||
// Total balance: Home's own TOTAL BALANCE node, which the app computes over
|
||||
// every account rather than over the cards that happen to be laid out inside
|
||||
// the viewport. The carrier deliberately tracks only that Home total. Ledger's
|
||||
// LedgerBalance is a single-account number on a different scale and would
|
||||
// corrupt cross-screen comparisons if mixed in. Off-Home steps carry forward
|
||||
// the last-seen Home sum so `previous` and `current` stay on the same scale.
|
||||
let lastHomeTotal = 0;
|
||||
const totalBalance = extract("totalBalance", s => {
|
||||
const cards = s.ax.findAll([{ testTag: "HomeScreen" }, { testTag: "AccountCard" }]);
|
||||
const cardBalanceTexts = cards.map(c => c.find({ testTag: "AccountBalance" })?.text);
|
||||
lastHomeTotal = computeHomeTotalBalance({ cardBalanceTexts, previousCarrier: lastHomeTotal });
|
||||
return lastHomeTotal;
|
||||
// the last-read Home total so `previous` and `current` stay on the same scale.
|
||||
const homeTotalText = (s: State) => on("home", "TotalBalance")(s)?.text;
|
||||
|
||||
let lastHomeTotal: number | null = null;
|
||||
const totalBalance = extract<number | null>("totalBalance", s => {
|
||||
const reading = readHomeTotalBalance({
|
||||
route: routeOf(s),
|
||||
totalText: homeTotalText(s),
|
||||
previousCarrier: lastHomeTotal,
|
||||
});
|
||||
lastHomeTotal = reading.carrier;
|
||||
return reading.value;
|
||||
});
|
||||
|
||||
// Submit actions inside the window `totalBalance.previous` and
|
||||
// `totalBalance.current` span. Recomputed rather than shared with the extractor
|
||||
// above because extractor getters may not read one another; both derive
|
||||
// freshness from the same reading, so they reset on the same step.
|
||||
let submitsSinceHomeTotal = 0;
|
||||
const submitsInWindow = extract("submitsInWindow", s => {
|
||||
const fresh = readHomeTotalBalance({
|
||||
route: routeOf(s),
|
||||
totalText: homeTotalText(s),
|
||||
previousCarrier: null,
|
||||
}).fresh;
|
||||
const window = countSubmitsInWindow({
|
||||
previousCount: submitsSinceHomeTotal,
|
||||
lastAction: s.lastAction,
|
||||
fresh,
|
||||
});
|
||||
submitsSinceHomeTotal = window.next;
|
||||
return window.reported;
|
||||
});
|
||||
|
||||
// The account list, carried across off-Home steps exactly like totalBalance and
|
||||
// for the same reason: the pair a property compares has to be two Home
|
||||
// readings, not a Home reading and whatever happened to be on screen.
|
||||
let lastHomeAccounts: Account[] | null = null;
|
||||
const accounts = extract<Account[] | null>("accounts", s => {
|
||||
const reading = readHomeCards({
|
||||
route: routeOf(s),
|
||||
reading: homeAccountsOf(homeCards(s)),
|
||||
previousCarrier: lastHomeAccounts,
|
||||
});
|
||||
lastHomeAccounts = reading.carrier;
|
||||
return reading.value;
|
||||
});
|
||||
|
||||
// Transactions committed per account, same carrier rule.
|
||||
let lastHomeTxnCounts: Record<string, TxnCount> | null = null;
|
||||
const homeTxnCounts = extract<Record<string, TxnCount> | null>("homeTxnCounts", s => {
|
||||
const reading = readHomeCards({
|
||||
route: routeOf(s),
|
||||
reading: homeTxnCountsOf(homeCards(s)),
|
||||
previousCarrier: lastHomeTxnCounts,
|
||||
});
|
||||
lastHomeTxnCounts = reading.carrier;
|
||||
return reading.value;
|
||||
});
|
||||
|
||||
// The counting invariant gets its own window because its carrier advances on a
|
||||
// different event than the total's: a Home frame can render the footer total
|
||||
// while its card list is still empty. Sharing submitsInWindow would let the
|
||||
// count reset without the counts pair moving, and a window whose submit count
|
||||
// is smaller than the interval its two readings span is a false conviction
|
||||
// waiting to happen.
|
||||
let submitsSinceHomeCards = 0;
|
||||
const submitsSinceCounts = extract("submitsSinceCounts", s => {
|
||||
const fresh = readHomeCards({
|
||||
route: routeOf(s),
|
||||
reading: homeTxnCountsOf(homeCards(s)),
|
||||
previousCarrier: null,
|
||||
}).fresh;
|
||||
const window = countSubmitsInWindow({
|
||||
previousCount: submitsSinceHomeCards,
|
||||
lastAction: s.lastAction,
|
||||
fresh,
|
||||
});
|
||||
submitsSinceHomeCards = window.next;
|
||||
return window.reported;
|
||||
});
|
||||
|
||||
const lastAction = extract("lastAction", s => s.lastAction);
|
||||
|
||||
const loginEmailField = extract("loginEmailField", s =>
|
||||
s.ax.find([{ testTag: "LoginScreen" }, { testTag: "LoginEmail" }]));
|
||||
const loginPasswordField = extract("loginPasswordField", s =>
|
||||
s.ax.find([{ testTag: "LoginScreen" }, { testTag: "LoginPassword" }]));
|
||||
const loginSubmit = extract("loginSubmit", s =>
|
||||
s.ax.find([{ testTag: "LoginScreen" }, { testTag: "LoginSubmit" }]));
|
||||
const addAccountButton = extract("addAccountButton", s =>
|
||||
s.ax.find([{ testTag: "HomeScreen" }, { testTag: "AddAccountButton" }]));
|
||||
const accountNameField = extract("accountNameField", s =>
|
||||
s.ax.find([{ testTag: "AddAccountScreen" }, { testTag: "AccountNameField" }]));
|
||||
const addAccountSubmit = extract("addAccountSubmit", s =>
|
||||
s.ax.find([{ testTag: "AddAccountScreen" }, { testTag: "AddAccountSubmit" }]));
|
||||
const addTxnButton = extract("addTxnButton", s =>
|
||||
s.ax.find([{ testTag: "LedgerScreen" }, { testTag: "AddTransactionButton" }]));
|
||||
const txnAmountField = extract("txnAmountField", s =>
|
||||
s.ax.find([{ testTag: "AddTransactionScreen" }, { testTag: "TxnAmountField" }]));
|
||||
const txnSubmit = extract("txnSubmit", s =>
|
||||
s.ax.find([{ testTag: "AddTransactionScreen" }, { testTag: "TxnSubmit" }]));
|
||||
const accountCards = extract("accountCards", s =>
|
||||
s.ax.findAll([{ testTag: "HomeScreen" }, { testTag: "AccountCard" }]));
|
||||
const loginEmailField = extract("loginEmailField", on("login", "LoginEmail"));
|
||||
const loginPasswordField = extract("loginPasswordField", on("login", "LoginPassword"));
|
||||
const loginSubmit = extract("loginSubmit", on("login", "LoginSubmit"));
|
||||
const addAccountButton = extract("addAccountButton", on("home", "AddAccountButton"));
|
||||
const accountNameField = extract("accountNameField", on("add-account", "AccountNameField"));
|
||||
const addAccountSubmit = extract("addAccountSubmit", on("add-account", "AddAccountSubmit"));
|
||||
const addTxnButton = extract("addTxnButton", on("ledger", "AddTransactionButton"));
|
||||
const txnAmountField = extract("txnAmountField", on("add-transaction", "TxnAmountField"));
|
||||
const txnSubmit = extract("txnSubmit", on("add-transaction", "TxnSubmit"));
|
||||
const accountCards = extract("accountCards", allOn("home", "AccountCard"));
|
||||
|
||||
// Property 1: every newly-appearing account starts with balance === 0.
|
||||
// Identity is by visible name. Guard against navigation transitions where
|
||||
// accounts vanish from the visible tree.
|
||||
// Property 1: an account starts life holding nothing. The account it judges is
|
||||
// the one the fuzzer just created, on the step that creation landed on Home:
|
||||
// a card that merely turns up in a later reading is a card that came into view,
|
||||
// not an account that came into existence. See createdAccountHasNonZeroBalance.
|
||||
const newAccountBalanceIsZero = always(
|
||||
next(() => {
|
||||
const prev = accounts.previous ?? [];
|
||||
const curr = accounts.current;
|
||||
if (prev.length === 0 || curr.length === 0) return true;
|
||||
const prevNames = new Set(prev.map(a => a.name));
|
||||
return curr.filter(a => !prevNames.has(a.name)).every(a => a.balance === 0);
|
||||
})
|
||||
next(() =>
|
||||
!createdAccountHasNonZeroBalance({
|
||||
route: route.current,
|
||||
lastAction: lastAction.current,
|
||||
typedName: accountNameField.previous?.text,
|
||||
before: accounts.previous ?? null,
|
||||
after: accounts.current,
|
||||
}),
|
||||
),
|
||||
);
|
||||
|
||||
// Property 2: a tap on TxnSubmit must move the total balance by exactly the
|
||||
@@ -105,13 +217,29 @@ const submitMovesBalanceByTypedAmount = always(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: route.current,
|
||||
lastAction: lastAction.current,
|
||||
submitsInWindow: submitsInWindow.current,
|
||||
typedAmount: parseTypedAmount(txnAmountField.previous?.text),
|
||||
prevTotalBalance: totalBalance.previous ?? 0,
|
||||
prevTotalBalance: totalBalance.previous ?? null,
|
||||
currTotalBalance: totalBalance.current,
|
||||
}),
|
||||
),
|
||||
);
|
||||
|
||||
// Property 3: one submit action commits at most one transaction. Counting
|
||||
// actions against transactions needs no amounts and no float arithmetic, and it
|
||||
// stays sound however wide the window between two Home readings gets, because
|
||||
// both sides of the comparison accumulate over the same window. It is the
|
||||
// double-submit stated directly: one tap, two rows.
|
||||
const submitCommitsOneTransactionPerAction = always(
|
||||
next(() =>
|
||||
!committedTransactionsExceedSubmits({
|
||||
countsBefore: homeTxnCounts.previous ?? null,
|
||||
countsAfter: homeTxnCounts.current,
|
||||
submitsInWindow: submitsSinceCounts.current,
|
||||
}),
|
||||
),
|
||||
);
|
||||
|
||||
const DEMO_EMAIL = "[email protected]";
|
||||
const DEMO_PASSWORD = "ledger123";
|
||||
|
||||
@@ -176,6 +304,7 @@ const addTxn = whenRoute(route, ["home", "ledger", "add-transaction"], () => {
|
||||
export const properties = {
|
||||
newAccountBalanceIsZero,
|
||||
submitMovesBalanceByTypedAmount,
|
||||
submitCommitsOneTransactionPerAction,
|
||||
};
|
||||
|
||||
export const setup = login;
|
||||
|
||||
@@ -256,12 +256,7 @@ func (d *Driver) InputText(callerCtx context.Context, text string) error {
|
||||
defer cancel()
|
||||
return chromedp.Run(runCtx,
|
||||
chromedp.ActionFunc(func(ctx context.Context) error {
|
||||
// Select any existing content so InsertText replaces rather than appends.
|
||||
if err := chromedp.Evaluate(`
|
||||
(function() {
|
||||
const el = document.activeElement;
|
||||
if (el && typeof el.select === 'function') el.select();
|
||||
})()`, nil).Do(ctx); err != nil {
|
||||
if err := selectFocusedText(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
return input.InsertText(text).Do(ctx)
|
||||
@@ -269,6 +264,28 @@ func (d *Driver) InputText(callerCtx context.Context, text string) error {
|
||||
)
|
||||
}
|
||||
|
||||
// selectAllScript selects everything in the focused field so the InsertText
|
||||
// that follows replaces rather than appends.
|
||||
//
|
||||
// document.activeElement stops at a shadow boundary: it names the HOST, not the
|
||||
// focused node inside. Compose for Web focuses a hidden <input> inside the
|
||||
// shadow root it mounts, so the host answer has no select() and the selection
|
||||
// never happened - every InputText appended to the last one, and a fuzzer that
|
||||
// types into the same field twice built up garbage it could never clear.
|
||||
// Descending activeElement through each shadow root finds the real field.
|
||||
const selectAllScript = `
|
||||
(function() {
|
||||
let el = document.activeElement;
|
||||
while (el && el.shadowRoot && el.shadowRoot.activeElement) {
|
||||
el = el.shadowRoot.activeElement;
|
||||
}
|
||||
if (el && typeof el.select === 'function') el.select();
|
||||
})()`
|
||||
|
||||
func selectFocusedText(ctx context.Context) error {
|
||||
return chromedp.Evaluate(selectAllScript, nil).Do(ctx)
|
||||
}
|
||||
|
||||
// ReplacesTextOnInput reports that InputText replaces existing content via
|
||||
// select-all, so the runner skips its pre-erase.
|
||||
func (d *Driver) ReplacesTextOnInput() bool {
|
||||
@@ -282,11 +299,7 @@ func (d *Driver) EraseText(callerCtx context.Context, _ int) error {
|
||||
defer cancel()
|
||||
return chromedp.Run(runCtx,
|
||||
chromedp.ActionFunc(func(ctx context.Context) error {
|
||||
if err := chromedp.Evaluate(`
|
||||
(function() {
|
||||
const el = document.activeElement;
|
||||
if (el && typeof el.select === 'function') el.select();
|
||||
})()`, nil).Do(ctx); err != nil {
|
||||
if err := selectFocusedText(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
return input.InsertText("").Do(ctx)
|
||||
@@ -368,7 +381,11 @@ func (d *Driver) Hierarchy(ctx context.Context) (string, error) {
|
||||
defer cancel()
|
||||
script := `
|
||||
(function() {
|
||||
const route = window.location.hash.replace(/^#/, '').split('?')[0] || '/';
|
||||
// Hash first (a HashRouter names the screen there), then the pathname, which
|
||||
// is where a path-routed SPA keeps it. Reporting '/' for every step of a
|
||||
// BrowserRouter app made every screen look like the same screen.
|
||||
const route = window.location.hash.replace(/^#/, '').split('?')[0] ||
|
||||
window.location.pathname || '/';
|
||||
// clickable and editable are resolved through the SAME selector sets
|
||||
// pkg/spec/src/web-runtime.ts uses, so the goja host (which reads this dump)
|
||||
// and the V8 host (which reads the DOM directly) cannot mean different things
|
||||
@@ -376,6 +393,14 @@ func (d *Driver) Hierarchy(ctx context.Context) (string, error) {
|
||||
// root a full-viewport tap target here and nowhere else.
|
||||
const NON_TEXT_INPUT_TYPES =
|
||||
['button','submit','checkbox','radio','range','color','file','image','reset'];
|
||||
// The disabled property belongs to real form controls only, so it reads
|
||||
// undefined on the role-based controls the tappable set now covers, and every
|
||||
// one of them looked enabled however plainly it was marked otherwise.
|
||||
// isEnabled in pkg/spec/src/web-runtime.ts answers the same two ways.
|
||||
function isEnabled(el) {
|
||||
if (el.disabled) return false;
|
||||
return el.getAttribute('aria-disabled') !== 'true';
|
||||
}
|
||||
function isEditableElement(el) {
|
||||
if (el.isContentEditable) return true;
|
||||
const tag = el.tagName.toLowerCase();
|
||||
@@ -383,10 +408,28 @@ func (d *Driver) Hierarchy(ctx context.Context) (string, error) {
|
||||
if (tag === 'input') return !NON_TEXT_INPUT_TYPES.includes((el.type || '').toLowerCase());
|
||||
return false;
|
||||
}
|
||||
const clickableSet = new Set(document.querySelectorAll(
|
||||
'a, button, input, select, textarea, [role="button"], [onclick]'));
|
||||
const editableSet = new Set(Array.from(
|
||||
document.querySelectorAll('input, textarea, [contenteditable]')).filter(isEditableElement));
|
||||
// Shadow roots are part of the page a user sees, so they are part of the page
|
||||
// we enumerate. Compose for Web mounts its canvas AND its accessibility tree
|
||||
// inside a shadow root on the mount element, so a light-DOM-only walk reports
|
||||
// four nodes for a whole app and offers no action on any of them.
|
||||
function deepQuery(sel) {
|
||||
const out = [];
|
||||
const visit = (root) => {
|
||||
for (const el of root.querySelectorAll(sel)) out.push(el);
|
||||
for (const el of root.querySelectorAll('*')) if (el.shadowRoot) visit(el.shadowRoot);
|
||||
};
|
||||
visit(document);
|
||||
return out;
|
||||
}
|
||||
const TAPPABLE_ROLES = [
|
||||
'button', 'link', 'checkbox', 'radio', 'switch', 'tab', 'option',
|
||||
'menuitem', 'menuitemcheckbox', 'menuitemradio', 'treeitem'];
|
||||
const clickableSet = new Set(deepQuery(
|
||||
'a, button, input, select, textarea, ' +
|
||||
TAPPABLE_ROLES.map(role => '[role="' + role + '"]').join(', ') +
|
||||
', [onclick]'));
|
||||
const editableSet = new Set(deepQuery(
|
||||
'input, textarea, [contenteditable]').filter(isEditableElement));
|
||||
function buildTree(el, isRoot) {
|
||||
const rect = el.getBoundingClientRect();
|
||||
const attrs = {};
|
||||
@@ -414,6 +457,14 @@ func (d *Driver) Hierarchy(ctx context.Context) (string, error) {
|
||||
const isClickable = clickableSet.has(el);
|
||||
const isEditable = editableSet.has(el);
|
||||
const children = [];
|
||||
// Shadow content first, then light children: the shadow tree is what the
|
||||
// host actually renders, and targetElements in web-runtime.ts walks the same
|
||||
// order, which is the order the two enumerations are compared in.
|
||||
if (el.shadowRoot) {
|
||||
for (const child of el.shadowRoot.children) {
|
||||
children.push(buildTree(child, false));
|
||||
}
|
||||
}
|
||||
for (const child of el.children) {
|
||||
if (child.tagName === 'HEAD') continue;
|
||||
children.push(buildTree(child, false));
|
||||
@@ -422,7 +473,7 @@ func (d *Driver) Hierarchy(ctx context.Context) (string, error) {
|
||||
attributes: attrs,
|
||||
children: children,
|
||||
clickable: isClickable || null,
|
||||
enabled: (!el.disabled) || null,
|
||||
enabled: isEnabled(el) || null,
|
||||
focused: document.activeElement === el || null,
|
||||
checked: el.checked || null,
|
||||
selected: el.selected || null,
|
||||
@@ -498,10 +549,138 @@ func (d *Driver) RecentLogs(_ context.Context, since time.Time, minLevel string)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (d *Driver) WaitForIdle(ctx context.Context, _ time.Duration) error {
|
||||
// domQuietPeriod is how long the DOM must stop changing before the page counts
|
||||
// as settled. Compose for Web syncs its accessibility DOM off the frame loop:
|
||||
// measured at ~136 ms behind an InputText on the folio wasm build, so waiting
|
||||
// for frames alone (~16 ms each) returns while the app still reports the old
|
||||
// text, and the next step types into a field it believes is still empty.
|
||||
const domQuietPeriod = 150 * time.Millisecond
|
||||
|
||||
// transitionSettlePeriod is how much longer the settle waits for a route
|
||||
// transition to finish once the DOM has gone quiet. A canvas app's cross-fade
|
||||
// is invisible to a mutation observer: Compose splices the incoming screen's
|
||||
// accessibility nodes in when the animation STARTS and removes the outgoing
|
||||
// screen's when it ends, and nothing in between touches the DOM, so the tree
|
||||
// sits byte-identical (and quiet) with both routes live for the whole
|
||||
// animation. Settling on quiet alone returns there, and the next step then
|
||||
// verifies a tree that names the screen the app is leaving: on the folio wasm
|
||||
// build a submit that landed on Home was recorded as still being on the
|
||||
// transaction screen, so a property gated on where the action landed read the
|
||||
// wrong route and went vacuous. The wait is bounded so a page that genuinely
|
||||
// shows two *Screen ids at rest costs this much per step and no more.
|
||||
const transitionSettlePeriod = 800 * time.Millisecond
|
||||
|
||||
// settleReturnMargin is what WaitForIdle holds back from the caller's timeout,
|
||||
// so returning late by our own doing surfaces as a settled page rather than a
|
||||
// context cancellation.
|
||||
const settleReturnMargin = 100 * time.Millisecond
|
||||
|
||||
// settleScanMargin covers the in-page work the two waits do not themselves
|
||||
// account for: liveScreens() walks the document and every shadow root on each
|
||||
// 16 ms poll, and the whole script costs one CDP round trip.
|
||||
const settleScanMargin = 250 * time.Millisecond
|
||||
|
||||
// MinIdleTimeout is the shortest timeout WaitForIdle can be handed and still
|
||||
// spend the waits it is built from: the DOM quiet period, the route-transition
|
||||
// window that only opens once that quiet period has elapsed, and the second
|
||||
// quiet period the transition's own closing mutation starts. A caller that
|
||||
// passes less caps the settle below its own budget, and the step then samples a
|
||||
// page that is still mid-transition - which is the exact failure the transition
|
||||
// wait exists to prevent. internal/runner raises a shorter caller timeout to
|
||||
// this value.
|
||||
func (d *Driver) MinIdleTimeout() time.Duration {
|
||||
return 2*domQuietPeriod + transitionSettlePeriod +
|
||||
settleScanMargin + settleReturnMargin
|
||||
}
|
||||
|
||||
func (d *Driver) WaitForIdle(ctx context.Context, timeout time.Duration) error {
|
||||
runCtx, cancel := d.runCtx(ctx)
|
||||
defer cancel()
|
||||
return chromedp.Run(runCtx, chromedp.WaitReady("body", chromedp.ByQuery))
|
||||
// Leave the caller's deadline some room: returning late by our own doing
|
||||
// would surface as a context cancellation instead of a settled page.
|
||||
budget := max(timeout-settleReturnMargin, domQuietPeriod)
|
||||
script := fmt.Sprintf(settleScript,
|
||||
domQuietPeriod.Milliseconds(),
|
||||
budget.Milliseconds(),
|
||||
transitionSettlePeriod.Milliseconds(),
|
||||
)
|
||||
return chromedp.Run(runCtx,
|
||||
chromedp.WaitReady("body", chromedp.ByQuery),
|
||||
chromedp.Evaluate(script, nil, awaitPromise),
|
||||
)
|
||||
}
|
||||
|
||||
// liveScreensFunction defines liveScreens(), the page-side count of live ids
|
||||
// ending in "Screen". More than one is a route transition in flight: the same
|
||||
// rule the tree parser applies (Transitional in internal/hierarchy), so the
|
||||
// driver and the runner agree on what a settled route looks like. It descends
|
||||
// shadow roots because a canvas app keeps its whole accessibility tree inside
|
||||
// one.
|
||||
const liveScreensFunction = `
|
||||
const liveScreens = () => {
|
||||
let count = 0;
|
||||
const visit = (root) => {
|
||||
count += root.querySelectorAll('[id$="Screen"]').length;
|
||||
for (const element of root.querySelectorAll('*')) {
|
||||
if (element.shadowRoot) visit(element.shadowRoot);
|
||||
}
|
||||
};
|
||||
visit(document);
|
||||
return count;
|
||||
};`
|
||||
|
||||
// settleScript resolves once the document has gone quiet for %d ms and is not
|
||||
// mid route transition, or after %d ms whatever happens; the transition wait
|
||||
// itself gives up after %d ms. Shadow roots get their own observer: a canvas
|
||||
// app keeps its whole accessibility tree inside one, and mutations there do not
|
||||
// reach an observer on the document.
|
||||
//
|
||||
// The transition window opens when the quiet period ends, not when the script
|
||||
// starts. Anchored at the start it is already spent by the time the check can
|
||||
// first run on any page that keeps mutating for longer than the window, so the
|
||||
// wait resolves immediately with both routes still live - the mid-transition
|
||||
// return this whole wait exists to prevent. Each mutation reopens it, and the
|
||||
// budget above bounds the total either way.
|
||||
const settleScript = `
|
||||
new Promise(resolve => {
|
||||
const quietMillis = %d, budgetMillis = %d, transitionMillis = %d;
|
||||
const observers = [];
|
||||
let transitionDeadline = 0;
|
||||
let timer = null;
|
||||
const finish = () => {
|
||||
clearTimeout(timer);
|
||||
for (const observer of observers) observer.disconnect();
|
||||
resolve();
|
||||
};
|
||||
` + liveScreensFunction + `
|
||||
const quiet = () => {
|
||||
if (transitionDeadline === 0) transitionDeadline = Date.now() + transitionMillis;
|
||||
if (liveScreens() > 1 && Date.now() < transitionDeadline) {
|
||||
timer = setTimeout(quiet, 16);
|
||||
return;
|
||||
}
|
||||
finish();
|
||||
};
|
||||
const restart = () => {
|
||||
clearTimeout(timer);
|
||||
transitionDeadline = 0;
|
||||
timer = setTimeout(quiet, quietMillis);
|
||||
};
|
||||
const watch = (root) => {
|
||||
const observer = new MutationObserver(restart);
|
||||
observer.observe(root, {subtree: true, childList: true, attributes: true, characterData: true});
|
||||
observers.push(observer);
|
||||
for (const element of root.querySelectorAll('*')) {
|
||||
if (element.shadowRoot) watch(element.shadowRoot);
|
||||
}
|
||||
};
|
||||
watch(document);
|
||||
setTimeout(finish, budgetMillis);
|
||||
restart();
|
||||
})`
|
||||
|
||||
func awaitPromise(params *runtime.EvaluateParams) *runtime.EvaluateParams {
|
||||
return params.WithAwaitPromise(true)
|
||||
}
|
||||
|
||||
func (d *Driver) Health(_ context.Context) (driver.Health, error) {
|
||||
@@ -596,12 +775,21 @@ func (d *Driver) InstallBundle(ctx context.Context, source []byte) error {
|
||||
|
||||
// EvaluateExtractors invokes the bundle-installed extractor table and returns
|
||||
// each extractor's JSON-encoded current value keyed by its registration index.
|
||||
//
|
||||
// The read waits out a route transition first, bounded by
|
||||
// transitionSettlePeriod. The hierarchy fetch already re-fetches a transitional
|
||||
// tree (fetchSyncedState in internal/runner); without the same rule here the
|
||||
// two halves of one step describe different moments, and the spec's own
|
||||
// extractors are the half that loses: on the folio wasm build the extractors
|
||||
// sampled mid cross-fade and reported the route the app was leaving, so a
|
||||
// property gated on where the action landed skipped the only step that action
|
||||
// could be judged on.
|
||||
func (d *Driver) EvaluateExtractors(ctx context.Context) (map[int]json.RawMessage, error) {
|
||||
const script = `JSON.stringify(window.__sanderlingExtractors__ ? window.__sanderlingExtractors__() : {})`
|
||||
script := fmt.Sprintf(extractorScript, transitionSettlePeriod.Milliseconds())
|
||||
var encoded string
|
||||
runCtx, cancel := d.runCtx(ctx)
|
||||
defer cancel()
|
||||
if err := chromedp.Run(runCtx, chromedp.Evaluate(script, &encoded)); err != nil {
|
||||
if err := chromedp.Run(runCtx, chromedp.Evaluate(script, &encoded, awaitPromise)); err != nil {
|
||||
return nil, fmt.Errorf("evaluate extractors: %w", err)
|
||||
}
|
||||
if encoded == "" || encoded == "{}" {
|
||||
@@ -612,16 +800,90 @@ func (d *Driver) EvaluateExtractors(ctx context.Context) (map[int]json.RawMessag
|
||||
return nil, fmt.Errorf("decode extractor map: %w", err)
|
||||
}
|
||||
result := make(map[int]json.RawMessage, len(stringMap))
|
||||
for key, value := range stringMap {
|
||||
for key, entry := range stringMap {
|
||||
index, err := strconv.Atoi(key)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("non-integer extractor key %q", key)
|
||||
}
|
||||
result[index] = value
|
||||
reading, err := extractorReading(entry)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("extractor %d: %w", index, err)
|
||||
}
|
||||
result[index] = reading
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// extractorReading unwraps one entry of the page's extractor table. The page
|
||||
// wraps every reading in a {"value": ...} envelope (evaluateExtractors in
|
||||
// pkg/spec/src/web-runtime.ts) because JSON has no undefined: an absent `value`
|
||||
// is the getter returning undefined, and returning it as an empty payload is
|
||||
// what makes the goja host record undefined too. Reading it as JSON null would
|
||||
// claim the getter returned null, so `x.current === undefined` would answer one
|
||||
// thing on native and another on web.
|
||||
func extractorReading(entry json.RawMessage) (json.RawMessage, error) {
|
||||
var envelope struct {
|
||||
Value json.RawMessage `json:"value"`
|
||||
}
|
||||
if err := json.Unmarshal(entry, &envelope); err != nil {
|
||||
return nil, fmt.Errorf(
|
||||
"reading %s is not a {\"value\"} envelope; the page and the host are "+
|
||||
"running different bundles: %w", entry, err)
|
||||
}
|
||||
return envelope.Value, nil
|
||||
}
|
||||
|
||||
// SetLastAction installs the previous step's action as state.lastAction inside
|
||||
// the page runtime. The page cannot derive it: only the runner knows which
|
||||
// action was actually applied. Without this call every web state.lastAction is
|
||||
// null, so a property gated on what the last action did is vacuously true and
|
||||
// reports a green run while checking nothing.
|
||||
//
|
||||
// The call is deliberately unguarded. A `setter && setter(...)` form evaluates
|
||||
// to undefined on a page whose runtime does not define the setter, and chromedp
|
||||
// reports that as success, so "the page cannot accept lastAction" would be
|
||||
// indistinguishable from "installed". That page is reachable: a run resolving
|
||||
// its web runtime from an older published @sanderling/spec would silently no-op
|
||||
// every step. Unguarded, the missing global throws and the run fails loudly.
|
||||
func (d *Driver) SetLastAction(ctx context.Context, encoded json.RawMessage) error {
|
||||
payload := strings.TrimSpace(string(encoded))
|
||||
if payload == "" {
|
||||
payload = "null"
|
||||
}
|
||||
script := fmt.Sprintf(`window.__sanderlingSetLastAction__(%s)`, payload)
|
||||
runCtx, cancel := d.runCtx(ctx)
|
||||
defer cancel()
|
||||
if err := chromedp.Run(runCtx, chromedp.Evaluate(script, nil)); err != nil {
|
||||
return fmt.Errorf("set last action: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// extractorScript resolves the extractor table once the page is not mid route
|
||||
// transition, giving up on that wait after %d ms.
|
||||
//
|
||||
// A missing table rejects rather than reporting {}, for the same reason
|
||||
// SetLastAction no longer guards its call: an empty override map is what a
|
||||
// spec with no extractors returns, so the guarded form made "this page has no
|
||||
// sanderling runtime" read as a normal step whose properties then ran on
|
||||
// goja's dump-derived values instead of the page's.
|
||||
const extractorScript = `
|
||||
new Promise((resolve, reject) => {
|
||||
const deadline = Date.now() + %d;` + liveScreensFunction + `
|
||||
const read = () => {
|
||||
if (liveScreens() > 1 && Date.now() < deadline) {
|
||||
setTimeout(read, 16);
|
||||
return;
|
||||
}
|
||||
if (typeof window.__sanderlingExtractors__ !== "function") {
|
||||
reject(new Error("__sanderlingExtractors__ is not installed in the page"));
|
||||
return;
|
||||
}
|
||||
resolve(JSON.stringify(window.__sanderlingExtractors__()));
|
||||
};
|
||||
read();
|
||||
})`
|
||||
|
||||
// NextActionFromV8 invokes the bundle-installed action generator and returns
|
||||
// the resulting Action JSON. Returns an empty json.RawMessage when the
|
||||
// generator declines to act this tick.
|
||||
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -474,3 +475,493 @@ func TestLaunch_KeepsBrowserAliveAfterCallerContextEnds(t *testing.T) {
|
||||
t.Fatalf("second Launch after the first caller context ended: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestInputText_ReplacesTextInsideAShadowRoot pins ReplacesTextOnInput's promise
|
||||
// on the shape a canvas app actually has. Compose for Web draws its text fields
|
||||
// on a canvas and routes typing through a hidden <input> INSIDE the shadow root
|
||||
// it mounts, and document.activeElement stops at a shadow boundary: it names the
|
||||
// host. The select-all therefore ran against a <div> with no select(), every
|
||||
// InputText appended to the last, and a fuzzer typing twice into one field built
|
||||
// up text it could never clear (observed on the folio wasm build as
|
||||
// "0.0000001" -> "0.0000001\t-1").
|
||||
func TestInputText_ReplacesTextInsideAShadowRoot(t *testing.T) {
|
||||
const page = `<body><div id="app"></div><script>
|
||||
const root = document.getElementById("app").attachShadow({mode: "open"});
|
||||
root.innerHTML = ` + "`" + `
|
||||
<style>
|
||||
#surface { position: absolute; left: 0; top: 0; }
|
||||
#a11y { position: absolute; left: 0; top: 0; pointer-events: none; }
|
||||
#proxy { position: absolute; left: -9999px; }
|
||||
</style>
|
||||
<canvas id="surface" width="300" height="200"></canvas>
|
||||
<div id="a11y"><div id="field">-</div></div>
|
||||
<input id="proxy" type="text">` + "`" + `;
|
||||
const proxy = root.getElementById("proxy");
|
||||
const field = root.getElementById("field");
|
||||
// The canvas owns the pointer (the a11y overlay is pointer-events: none)
|
||||
// and hands focus to the proxy, exactly as a canvas app does.
|
||||
root.getElementById("surface").addEventListener("click", function () { proxy.focus(); });
|
||||
proxy.addEventListener("input", function () { field.textContent = proxy.value; });
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
if err := d.Tap(ctx, 40, 40); err != nil {
|
||||
t.Fatalf("Tap: %v", err)
|
||||
}
|
||||
if err := d.InputText(ctx, "alpha"); err != nil {
|
||||
t.Fatalf("InputText: %v", err)
|
||||
}
|
||||
if err := d.InputText(ctx, "beta"); err != nil {
|
||||
t.Fatalf("InputText: %v", err)
|
||||
}
|
||||
|
||||
var shown string
|
||||
script := `document.getElementById("app").shadowRoot.getElementById("field").textContent`
|
||||
if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &shown)); err != nil {
|
||||
t.Fatalf("read field: %v", err)
|
||||
}
|
||||
if shown != "beta" {
|
||||
t.Errorf("field holds %q, want %q; the second InputText appended instead of replacing", shown, "beta")
|
||||
}
|
||||
|
||||
if err := d.EraseText(ctx, len("beta")); err != nil {
|
||||
t.Fatalf("EraseText: %v", err)
|
||||
}
|
||||
if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &shown)); err != nil {
|
||||
t.Fatalf("read field: %v", err)
|
||||
}
|
||||
if shown != "" {
|
||||
t.Errorf("field holds %q after EraseText, want empty", shown)
|
||||
}
|
||||
}
|
||||
|
||||
// TestHierarchy_ScreenFallsBackToThePathname pins the route the goja host reads
|
||||
// off the dump. Reading location.hash alone reported "/" on every step of a
|
||||
// path-routed SPA (react-router's BrowserRouter, which the replay UI itself
|
||||
// uses), so every screen looked like the same screen and no route-scoped
|
||||
// property or action could tell them apart.
|
||||
func TestHierarchy_ScreenFallsBackToThePathname(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(`<body><div id="app">app</div></body>`))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
for _, testCase := range []struct {
|
||||
name string
|
||||
path string
|
||||
want string
|
||||
}{
|
||||
{"path-routed", "/runs/20260101-120000/steps/7", "/runs/20260101-120000/steps/7"},
|
||||
{"hash wins when present", "/runs/1#/detail", "/detail"},
|
||||
{"root", "/", "/"},
|
||||
} {
|
||||
t.Run(testCase.name, func(t *testing.T) {
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL+testCase.path, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
dump, err := d.Hierarchy(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("Hierarchy: %v", err)
|
||||
}
|
||||
var root struct {
|
||||
Attributes map[string]string `json:"attributes"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(dump), &root); err != nil {
|
||||
t.Fatalf("unmarshal hierarchy: %v", err)
|
||||
}
|
||||
if got := root.Attributes["sanderling-screen"]; got != testCase.want {
|
||||
t.Errorf("sanderling-screen: got %q, want %q", got, testCase.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForIdle_WaitsForWorkTheActionKickedOff pins the settle the runner
|
||||
// relies on between acting and observing. WaitForIdle used to return the moment
|
||||
// <body> existed, which is true before the app has reacted at all: measured on
|
||||
// the folio wasm build, Compose's accessibility DOM lands ~136 ms after an
|
||||
// InputText, so the next step read the pre-action text and typed into a field
|
||||
// it believed was still empty.
|
||||
func TestWaitForIdle_WaitsForWorkTheActionKickedOff(t *testing.T) {
|
||||
const page = `<body><div id="app"></div><script>
|
||||
const root = document.getElementById("app").attachShadow({mode: "open"});
|
||||
root.innerHTML = '<button id="go" style="width:200px;height:80px">go</button><div id="out">pending</div>';
|
||||
root.getElementById("go").addEventListener("click", function () {
|
||||
setTimeout(function () { root.getElementById("out").textContent = "settled"; }, 100);
|
||||
});
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
if err := d.Tap(ctx, 40, 40); err != nil {
|
||||
t.Fatalf("Tap: %v", err)
|
||||
}
|
||||
if err := d.WaitForIdle(ctx, time.Second); err != nil {
|
||||
t.Fatalf("WaitForIdle: %v", err)
|
||||
}
|
||||
|
||||
var shown string
|
||||
script := `document.getElementById("app").shadowRoot.getElementById("out").textContent`
|
||||
if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &shown)); err != nil {
|
||||
t.Fatalf("read: %v", err)
|
||||
}
|
||||
if shown != "settled" {
|
||||
t.Errorf("observed %q; WaitForIdle returned before the tap's own work landed", shown)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForIdle_ReturnsOnABusyPage is the other half: a page that never stops
|
||||
// mutating (an animation, a polling widget) must not hold the step loop open.
|
||||
func TestWaitForIdle_ReturnsOnABusyPage(t *testing.T) {
|
||||
const page = `<body><div id="tick">0</div><script>
|
||||
let n = 0;
|
||||
setInterval(function () { document.getElementById("tick").textContent = String(++n); }, 15);
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
start := time.Now()
|
||||
if err := d.WaitForIdle(ctx, time.Second); err != nil {
|
||||
t.Fatalf("WaitForIdle: %v", err)
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed > 2*time.Second {
|
||||
t.Errorf("WaitForIdle took %s on a busy page; it must return inside its budget", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForIdle_WaitsOutARouteTransition covers the settle case a mutation
|
||||
// observer cannot see. A canvas app splices the incoming screen's
|
||||
// accessibility nodes in when its cross-fade STARTS and drops the outgoing
|
||||
// screen's when it ends; between those two mutations the DOM is quiet with both
|
||||
// routes live. Returning there hands the next step a tree naming the screen the
|
||||
// app is leaving, which on the folio wasm build recorded a submit that had
|
||||
// landed on Home as still being on the transaction screen: the route gate of an
|
||||
// action-gated property then skipped the very step the action landed on.
|
||||
//
|
||||
// The page keeps mutating for 300 ms after the route splice, and the settle
|
||||
// runs on the timeout production hands it (MinIdleTimeout). Both details are
|
||||
// load-bearing. A quiet page reaches the transition check immediately, so it
|
||||
// passes whether the transition window is anchored at the script start or at
|
||||
// the end of the quiet period; churn is what pushes the check past a
|
||||
// start-anchored deadline, which then finishes at once with two live screens.
|
||||
// And a caller timeout below MinIdleTimeout cuts the whole settle off before
|
||||
// the transition window can be spent, which is the same bug from the other end.
|
||||
func TestWaitForIdle_WaitsOutARouteTransition(t *testing.T) {
|
||||
const page = `<body><div id="app"></div><script>
|
||||
const root = document.getElementById("app").attachShadow({mode: "open"});
|
||||
root.innerHTML = '<button id="go" style="width:200px;height:80px">go</button>' +
|
||||
'<div id="LedgerScreen">ledger</div><div id="spinner">0</div>';
|
||||
root.getElementById("go").addEventListener("click", function () {
|
||||
const incoming = document.createElement("div");
|
||||
incoming.id = "HomeScreen";
|
||||
incoming.textContent = "home";
|
||||
root.appendChild(incoming);
|
||||
let frame = 0;
|
||||
const churn = setInterval(function () {
|
||||
root.getElementById("spinner").textContent = String(++frame);
|
||||
}, 30);
|
||||
setTimeout(function () { clearInterval(churn); }, 300);
|
||||
setTimeout(function () { root.getElementById("LedgerScreen").remove(); }, 1000);
|
||||
});
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
if err := d.Tap(ctx, 40, 40); err != nil {
|
||||
t.Fatalf("Tap: %v", err)
|
||||
}
|
||||
if err := d.WaitForIdle(ctx, d.MinIdleTimeout()); err != nil {
|
||||
t.Fatalf("WaitForIdle: %v", err)
|
||||
}
|
||||
|
||||
var live []string
|
||||
script := `Array.from(document.getElementById("app").shadowRoot
|
||||
.querySelectorAll('[id$="Screen"]')).map(e => e.id)`
|
||||
if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(script, &live)); err != nil {
|
||||
t.Fatalf("read: %v", err)
|
||||
}
|
||||
if len(live) != 1 || live[0] != "HomeScreen" {
|
||||
t.Errorf("live screens after the settle = %v, want [HomeScreen]; WaitForIdle "+
|
||||
"returned mid-transition, so the next step verifies the outgoing route", live)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForIdle_BoundsTheTransitionWait is the other half of the transition
|
||||
// wait: a page that shows two *Screen ids at rest is not mid-transition, it
|
||||
// just matches the heuristic, and it must cost one bounded wait rather than the
|
||||
// whole step budget on every step.
|
||||
func TestWaitForIdle_BoundsTheTransitionWait(t *testing.T) {
|
||||
const page = `<body><div id="HomeScreen">home</div><div id="LedgerScreen">ledger</div></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
start := time.Now()
|
||||
if err := d.WaitForIdle(ctx, 10*time.Second); err != nil {
|
||||
t.Fatalf("WaitForIdle: %v", err)
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed > transitionSettlePeriod+time.Second {
|
||||
t.Errorf("WaitForIdle took %s on a page with two resting screens; the "+
|
||||
"transition wait must be bounded by %s", elapsed, transitionSettlePeriod)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateExtractors_WaitsOutARouteTransition covers the other sampler. A
|
||||
// step reads the page twice: the hierarchy dump (which re-fetches while the
|
||||
// tree looks transitional) and the spec's own extractors in V8. Sampling the
|
||||
// extractors mid cross-fade reports the route the app is leaving, and an
|
||||
// action-gated property then skips the one step its action can be judged on:
|
||||
// on the folio wasm build a double-submit that landed on Home was extracted as
|
||||
// still being on the transaction screen.
|
||||
func TestEvaluateExtractors_WaitsOutARouteTransition(t *testing.T) {
|
||||
const page = `<body><div id="app"></div><script>
|
||||
const root = document.getElementById("app").attachShadow({mode: "open"});
|
||||
root.innerHTML = '<div id="LedgerScreen">ledger</div>';
|
||||
window.__sanderlingExtractors__ = function () {
|
||||
return {0: {value: Array.from(root.querySelectorAll('[id$="Screen"]')).map(e => e.id).join(",")}};
|
||||
};
|
||||
window.startTransition = function () {
|
||||
const incoming = document.createElement("div");
|
||||
incoming.id = "HomeScreen";
|
||||
root.appendChild(incoming);
|
||||
setTimeout(function () { root.getElementById("LedgerScreen").remove(); }, 400);
|
||||
};
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
if err := chromedp.Run(d.tabCtx, chromedp.Evaluate(`window.startTransition()`, nil)); err != nil {
|
||||
t.Fatalf("start transition: %v", err)
|
||||
}
|
||||
|
||||
values, err := d.EvaluateExtractors(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("EvaluateExtractors: %v", err)
|
||||
}
|
||||
if got := string(values[0]); got != `"HomeScreen"` {
|
||||
t.Errorf("extractor read %s, want \"HomeScreen\"; the extractors sampled "+
|
||||
"mid-transition, so the spec sees the route the app is leaving", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSetLastAction_ReportsAPageThatCannotTakeIt covers the install the whole
|
||||
// web path's action-gated properties hang off. A page without the setter is
|
||||
// reachable: internal/testrun resolves the web runtime from
|
||||
// node_modules/@sanderling/spec when no sibling checkout is present, and an
|
||||
// older published runtime does not define it. Guarded as
|
||||
// `setter && setter(...)`, that page returns undefined and chromedp reports
|
||||
// success, so every step silently no-ops and every property gated on the last
|
||||
// action goes vacuously true - a green run that checked nothing.
|
||||
func TestSetLastAction_ReportsAPageThatCannotTakeIt(t *testing.T) {
|
||||
const withSetter = `<body><script>
|
||||
window.__lastActionSeen = null;
|
||||
window.__sanderlingSetLastAction__ = function (value) { window.__lastActionSeen = value; };
|
||||
</script></body>`
|
||||
const withoutSetter = `<body><div id="app">no sanderling runtime here</div></body>`
|
||||
pages := map[string]string{"/with": withSetter, "/without": withoutSetter}
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(pages[r.URL.Path]))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
|
||||
if err := d.Launch(ctx, server.URL+"/with", false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
action := json.RawMessage(`{"kind":"Tap","on":"id:TxnSubmit"}`)
|
||||
if err := d.SetLastAction(ctx, action); err != nil {
|
||||
t.Fatalf("SetLastAction on a page that defines the setter: %v", err)
|
||||
}
|
||||
var seen map[string]string
|
||||
if err := chromedp.Run(d.tabCtx,
|
||||
chromedp.Evaluate(`window.__lastActionSeen`, &seen)); err != nil {
|
||||
t.Fatalf("read installed action: %v", err)
|
||||
}
|
||||
if seen["on"] != "id:TxnSubmit" {
|
||||
t.Errorf("the page received %v, want the action the runner applied", seen)
|
||||
}
|
||||
|
||||
if err := d.Launch(ctx, server.URL+"/without", false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
if err := d.SetLastAction(ctx, action); err == nil {
|
||||
t.Error("SetLastAction reported success on a page with no setter; " +
|
||||
"a runtime that cannot take lastAction is indistinguishable from one that did")
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateExtractors_ReportsAMissingTable is the same failure on the other
|
||||
// sampler. An empty override map is what a spec with no extractors returns, so
|
||||
// treating a missing table as {} makes "this page has no sanderling runtime"
|
||||
// read as an ordinary step - and the verifier then judges the run on goja's
|
||||
// dump-derived values while believing they came from the page.
|
||||
func TestEvaluateExtractors_ReportsAMissingTable(t *testing.T) {
|
||||
const page = `<body><div id="app">no sanderling runtime here</div></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
values, err := d.EvaluateExtractors(ctx)
|
||||
if err == nil {
|
||||
t.Errorf("EvaluateExtractors returned %v and no error on a page with no "+
|
||||
"extractor table; a page that cannot be read must not read as empty", values)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateExtractors_KeepsUndefinedApartFromNull covers the wire the page's
|
||||
// readings cross. JSON has no undefined, so the web runtime wraps each reading
|
||||
// in a {value} envelope: written straight into the table, an extractor that
|
||||
// returned undefined lost its whole index to JSON.stringify and the host kept
|
||||
// goja's dump-derived reading for it while the rest held the page's. An absent
|
||||
// value has to arrive as an empty payload, which is what makes the verifier
|
||||
// record undefined (the value the native host records for the same getter);
|
||||
// arriving as JSON null would claim the getter returned null.
|
||||
func TestEvaluateExtractors_KeepsUndefinedApartFromNull(t *testing.T) {
|
||||
const page = `<body><script>
|
||||
window.__sanderlingExtractors__ = function () {
|
||||
return {0: {}, 1: {value: null}, 2: {value: {balance: 7}}};
|
||||
};
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
|
||||
values, err := d.EvaluateExtractors(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("EvaluateExtractors: %v", err)
|
||||
}
|
||||
if len(values) != 3 {
|
||||
t.Fatalf("the page reported 3 readings, %d survived the wire: %v", len(values), values)
|
||||
}
|
||||
if got := values[0]; len(got) != 0 {
|
||||
t.Errorf("the undefined reading arrived as %s, want an empty payload; "+
|
||||
"the verifier records anything else as a value the getter never returned", got)
|
||||
}
|
||||
if got := string(values[1]); got != "null" {
|
||||
t.Errorf("the null reading arrived as %s, want null", got)
|
||||
}
|
||||
if got := string(values[2]); got != `{"balance":7}` {
|
||||
t.Errorf("the object reading arrived as %s, want {\"balance\":7}", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateExtractors_RejectsAnUnenvelopedReading is the loud failure a page
|
||||
// running an older @sanderling/spec produces. Its readings are bare values, and
|
||||
// a bare value is indistinguishable from a reading whose getter returned that
|
||||
// value, so accepting them silently puts the two engines on different bundles.
|
||||
func TestEvaluateExtractors_RejectsAnUnenvelopedReading(t *testing.T) {
|
||||
const page = `<body><script>
|
||||
window.__sanderlingExtractors__ = function () { return {0: "home"}; };
|
||||
</script></body>`
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
_, _ = w.Write([]byte(page))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
|
||||
values, err := d.EvaluateExtractors(ctx)
|
||||
if err == nil {
|
||||
t.Fatalf("EvaluateExtractors accepted %v from a page whose readings are not "+
|
||||
"enveloped; the page and the host are running different bundles", values)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "different bundles") {
|
||||
t.Errorf("EvaluateExtractors failed with %q, want it to name the bundle mismatch", err)
|
||||
}
|
||||
}
|
||||
@@ -60,15 +60,24 @@ type factRow struct {
|
||||
facts elementFacts
|
||||
}
|
||||
|
||||
// parityPages are the pages both producers are compared on. Each one must
|
||||
// exercise every fact both ways on its own (requireBothPolarities), so adding a
|
||||
// page never weakens the comparison. fact-parity-shadow.html is shaped like a
|
||||
// Compose for Web app: the whole UI lives inside a shadow root, which both
|
||||
// producers have to descend into or they enumerate one node for an entire app.
|
||||
var parityPages = []string{"fact-parity.html", "fact-parity-shadow.html"}
|
||||
|
||||
func TestHierarchy_DerivesTheSameFactsAsTheWebRuntime(t *testing.T) {
|
||||
server := httptest.NewServer(http.FileServer(http.Dir("testdata")))
|
||||
defer server.Close()
|
||||
|
||||
for _, page := range parityPages {
|
||||
t.Run(page, func(t *testing.T) {
|
||||
d := New()
|
||||
defer d.Terminate(context.Background())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
if err := d.Launch(ctx, server.URL+"/fact-parity.html", false, nil); err != nil {
|
||||
if err := d.Launch(ctx, server.URL+"/"+page, false, nil); err != nil {
|
||||
t.Fatalf("Launch: %v", err)
|
||||
}
|
||||
|
||||
@@ -84,6 +93,8 @@ func TestHierarchy_DerivesTheSameFactsAsTheWebRuntime(t *testing.T) {
|
||||
requireBothPolarities(t, fromWebRuntime)
|
||||
compareEnumeratedElements(t, fromDump, fromWebRuntime)
|
||||
compareDerivedFacts(t, fromDump, fromWebRuntime)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// factsFromHierarchyDump reads the dump the way the goja host does: parse it,
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
<!doctype html>
|
||||
<html id="page">
|
||||
<head id="page-head">
|
||||
<meta id="page-charset" charset="utf-8" />
|
||||
<title id="page-title">fact parity: shadow dom</title>
|
||||
</head>
|
||||
<body id="page-body">
|
||||
<!-- Shaped like Compose for Web: one mount element owning a shadow root
|
||||
that holds the canvas plus an accessibility overlay whose nodes are
|
||||
pointer-events: none, so taps fall through to the canvas. A producer
|
||||
that stops at the shadow boundary reports this whole app as one node. -->
|
||||
<div id="host">
|
||||
<div id="host-light-child">light child, no slot to render into</div>
|
||||
</div>
|
||||
<script id="page-script">
|
||||
const root = document.getElementById("host").attachShadow({ mode: "open" });
|
||||
root.innerHTML = `
|
||||
<style id="shadow-style">
|
||||
#shadow-surface { position: absolute; inset: 0; }
|
||||
#shadow-overlay { position: absolute; inset: 0; pointer-events: none; }
|
||||
#shadow-scroller { width: 320px; height: 120px; overflow: auto; }
|
||||
#shadow-scroller-content { height: 600px; }
|
||||
</style>
|
||||
<canvas id="shadow-surface" width="420" height="640"></canvas>
|
||||
<div id="shadow-overlay">
|
||||
<button id="shadow-save">save</button>
|
||||
<button id="shadow-cancel" disabled>cancel</button>
|
||||
<input id="shadow-amount" type="text" value="10" />
|
||||
<div id="shadow-plain">plain</div>
|
||||
</div>
|
||||
<div id="shadow-scroller"><div id="shadow-scroller-content"></div></div>`;
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -56,6 +56,23 @@
|
||||
<textarea id="notes"></textarea>
|
||||
<div id="bio" contenteditable="true">bio</div>
|
||||
<div id="menu" role="button">menu</div>
|
||||
<!-- One element per ARIA role both producers resolve as clickable. A role
|
||||
covered on one side only makes that control reachable for one host,
|
||||
which is how the whole set is kept honest: role="option" is the shape
|
||||
the replay UI gives its step rows. -->
|
||||
<div id="role-link" role="link">link</div>
|
||||
<div id="role-checkbox" role="checkbox">checkbox</div>
|
||||
<div id="role-radio" role="radio">radio</div>
|
||||
<div id="role-switch" role="switch">switch</div>
|
||||
<div id="role-tab" role="tab">tab</div>
|
||||
<div id="role-menuitem" role="menuitem">menu item</div>
|
||||
<div id="role-menuitemcheckbox" role="menuitemcheckbox">menu item checkbox</div>
|
||||
<div id="role-menuitemradio" role="menuitemradio">menu item radio</div>
|
||||
<div id="role-treeitem" role="treeitem">tree item</div>
|
||||
<ul id="role-listbox" role="listbox">
|
||||
<li id="role-option" role="option">option</li>
|
||||
<li id="role-option-disabled" role="option" aria-disabled="true">disabled option</li>
|
||||
</ul>
|
||||
<div id="attribute-click" onclick="void 0">attribute click</div>
|
||||
<div id="delegating-root">delegating root</div>
|
||||
<div id="plain">plain</div>
|
||||
|
||||
@@ -109,6 +109,13 @@ func NewDevice(ctx context.Context, options DeviceOptions) (*Driver, error) {
|
||||
d.restart = d.respawnDevice
|
||||
d.processContext, d.processCancel = context.WithCancel(ctx)
|
||||
|
||||
lock, err := acquireDeviceLock(d.udid)
|
||||
if err != nil {
|
||||
d.processCancel()
|
||||
return nil, err
|
||||
}
|
||||
d.deviceLock = lock
|
||||
|
||||
if err := d.bringUpDevice(ctx); err != nil {
|
||||
d.Close()
|
||||
return nil, err
|
||||
|
||||
@@ -42,6 +42,17 @@ const runnerStartupTimeout = 120 * time.Second
|
||||
// before it is killed. A variable so the kill-escalation test can shrink it.
|
||||
var shutdownGrace = 15 * time.Second
|
||||
|
||||
// launchTimeout bounds a single app lifecycle RPC. The runner serves lifecycle
|
||||
// inside its XCTest session, and a launch the simulator rejects sends that
|
||||
// session down a recovery chain (a 120s accessibility wait, a spindump, then an
|
||||
// idle wait) that answers minutes late or never. Callers reach Launch with an
|
||||
// undeadlined context, since it runs before the run's duration clock starts, so
|
||||
// the bound has to come from here or a wedged session hangs the run with no
|
||||
// trace, no error, and no end. Kept under runnerStartupTimeout: launching an
|
||||
// app inside a live session must cost less than cold-starting that session.
|
||||
// A variable so the timeout test can shrink it.
|
||||
var launchTimeout = 90 * time.Second
|
||||
|
||||
// longPressHoldMilliseconds is how long LongPress holds the finger down.
|
||||
const longPressHoldMilliseconds = 600
|
||||
|
||||
@@ -141,6 +152,34 @@ type Driver struct {
|
||||
// the moment startup finishes.
|
||||
processContext context.Context
|
||||
processCancel context.CancelFunc
|
||||
|
||||
// deviceLock is the exclusive claim on the target, held for the driver's
|
||||
// whole life and released by Close.
|
||||
deviceLock io.Closer
|
||||
}
|
||||
|
||||
// acquireDeviceLock takes an exclusive advisory lock on the target so only one
|
||||
// run drives it at a time. Two runs on one device interleave app lifecycle: the
|
||||
// second run's uninstall and reinstall land under the first's live automation
|
||||
// session, leaving its app proxies bound to a bundle the simulator no longer
|
||||
// knows, and every later snapshot and launch on that session stalls. Failing
|
||||
// fast beats recovering silently, since the other run owns the device and would
|
||||
// be corrupted either way. The lock lives on the file descriptor, so a crashed
|
||||
// run's claim is released by the kernel and never strands the device.
|
||||
func acquireDeviceLock(udid string) (io.Closer, error) {
|
||||
path := filepath.Join(os.TempDir(), "sanderling-ios-"+udid+".lock")
|
||||
file, err := os.OpenFile(path, os.O_CREATE|os.O_RDWR, 0o644)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open device lock %s: %w", path, err)
|
||||
}
|
||||
if err := syscall.Flock(int(file.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil {
|
||||
file.Close()
|
||||
return nil, fmt.Errorf(
|
||||
"ios target %s is already driven by another sanderling run (lock %s); "+
|
||||
"wait for that run to finish or point this one at a different device with --ios-device",
|
||||
udid, path)
|
||||
}
|
||||
return file, nil
|
||||
}
|
||||
|
||||
// New extracts the embedded companion, spawns it against the configured
|
||||
@@ -199,7 +238,15 @@ func New(ctx context.Context, options Options) (*Driver, error) {
|
||||
driverInstance.grantPaste = driverInstance.grantPasteboardAccess
|
||||
driverInstance.processContext, driverInstance.processCancel = context.WithCancel(ctx)
|
||||
|
||||
lock, err := acquireDeviceLock(driverInstance.udid)
|
||||
if err != nil {
|
||||
driverInstance.processCancel()
|
||||
return nil, err
|
||||
}
|
||||
driverInstance.deviceLock = lock
|
||||
|
||||
if err := driverInstance.bringUp(ctx); err != nil {
|
||||
driverInstance.Close()
|
||||
return nil, err
|
||||
}
|
||||
if driverInstance.hybrid {
|
||||
@@ -408,7 +455,9 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e
|
||||
|
||||
// Terminate first so the launch is a clean cold start regardless of the
|
||||
// app's prior state. A not-running app is not an error here.
|
||||
_ = d.withRecovery(ctx, func() error { return d.lifecycleCompanion().Terminate(ctx, d.bundleID) })
|
||||
_ = d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error {
|
||||
return companion.Terminate(callCtx, d.bundleID)
|
||||
})
|
||||
|
||||
if clearState {
|
||||
if err := d.clearAppState(ctx); err != nil {
|
||||
@@ -428,14 +477,26 @@ func (d *Driver) Launch(ctx context.Context, bundleID string, clearState bool, e
|
||||
}
|
||||
}
|
||||
|
||||
if err := d.withRecovery(ctx, func() error {
|
||||
return d.lifecycleCompanion().Launch(ctx, d.bundleID, true)
|
||||
if err := d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error {
|
||||
return companion.Launch(callCtx, d.bundleID, true)
|
||||
}); err != nil {
|
||||
return fmt.Errorf("launch %s: %w", d.bundleID, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// lifecycleCall runs an app lifecycle RPC against lifecycleCompanion under a
|
||||
// launchTimeout-bounded context, with the usual one-restart recovery. The
|
||||
// companion is resolved inside the retry so a restart's replacement client
|
||||
// serves the second attempt.
|
||||
func (d *Driver) lifecycleCall(ctx context.Context, call func(context.Context, transport.Companion) error) error {
|
||||
boundedCtx, cancel := context.WithTimeout(ctx, launchTimeout)
|
||||
defer cancel()
|
||||
return d.withRecovery(boundedCtx, func() error {
|
||||
return call(boundedCtx, d.lifecycleCompanion())
|
||||
})
|
||||
}
|
||||
|
||||
// lifecycleCompanion is the transport that owns app launch and terminate: the
|
||||
// in-simulator runner when the hybrid is active, otherwise the legacy
|
||||
// companion. Lifecycle performed outside the runner's automation session
|
||||
@@ -528,7 +589,9 @@ func (d *Driver) resetDataContainer(ctx context.Context) error {
|
||||
}
|
||||
|
||||
func (d *Driver) Terminate(ctx context.Context) error {
|
||||
return d.withRecovery(ctx, func() error { return d.lifecycleCompanion().Terminate(ctx, d.bundleID) })
|
||||
return d.lifecycleCall(ctx, func(callCtx context.Context, companion transport.Companion) error {
|
||||
return companion.Terminate(callCtx, d.bundleID)
|
||||
})
|
||||
}
|
||||
|
||||
func (d *Driver) Tap(ctx context.Context, x, y int) error {
|
||||
@@ -999,6 +1062,10 @@ func (d *Driver) Close() {
|
||||
if d.processCancel != nil {
|
||||
d.processCancel()
|
||||
}
|
||||
if d.deviceLock != nil {
|
||||
_ = d.deviceLock.Close()
|
||||
d.deviceLock = nil
|
||||
}
|
||||
}
|
||||
|
||||
// stopTunnel closes the in-process usbmux forwarder on the device path. Closing
|
||||
|
||||
@@ -851,3 +851,175 @@ func (s *sequencedDumpCompanion) AccessibilityInfo(context.Context) (string, err
|
||||
*s.reads++
|
||||
return s.dumps[index], nil
|
||||
}
|
||||
|
||||
// wedgedLifecycleCompanion never answers a lifecycle RPC, standing in for a
|
||||
// runner whose XCTest session is stuck inside a rejected launch.
|
||||
type wedgedLifecycleCompanion struct {
|
||||
fakeCompanion
|
||||
release chan struct{}
|
||||
}
|
||||
|
||||
func (w *wedgedLifecycleCompanion) block(ctx context.Context) error {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-w.release:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
func (w *wedgedLifecycleCompanion) Launch(ctx context.Context, _ string, _ bool) error {
|
||||
return w.block(ctx)
|
||||
}
|
||||
|
||||
func (w *wedgedLifecycleCompanion) Terminate(ctx context.Context, _ string) error {
|
||||
return w.block(ctx)
|
||||
}
|
||||
|
||||
// TestLaunchBoundsWedgedLifecycleRPC proves the launch path carries its own
|
||||
// deadline. Callers hand Launch an undeadlined context, so without one a runner
|
||||
// that never answers hangs the run forever with nothing printed.
|
||||
func TestLaunchBoundsWedgedLifecycleRPC(t *testing.T) {
|
||||
previous := launchTimeout
|
||||
launchTimeout = 100 * time.Millisecond
|
||||
defer func() { launchTimeout = previous }()
|
||||
|
||||
companion := &wedgedLifecycleCompanion{release: make(chan struct{})}
|
||||
defer close(companion.release)
|
||||
d := newTestDriver(companion)
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- d.Launch(context.Background(), "com.example.app", false, nil) }()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
if err == nil {
|
||||
t.Fatal("wedged launch returned nil; a stuck runner must surface an error")
|
||||
}
|
||||
if !errors.Is(err, context.DeadlineExceeded) {
|
||||
t.Fatalf("err = %v, want a deadline-exceeded error", err)
|
||||
}
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("Launch never returned: the lifecycle RPC is unbounded, so a stuck runner hangs the run forever")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTerminateBoundsWedgedLifecycleRPC covers the same bound on the standalone
|
||||
// terminate, which the runner calls mid-run on an equally stuck session.
|
||||
func TestTerminateBoundsWedgedLifecycleRPC(t *testing.T) {
|
||||
previous := launchTimeout
|
||||
launchTimeout = 100 * time.Millisecond
|
||||
defer func() { launchTimeout = previous }()
|
||||
|
||||
companion := &wedgedLifecycleCompanion{release: make(chan struct{})}
|
||||
defer close(companion.release)
|
||||
d := newTestDriver(companion)
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- d.Terminate(context.Background()) }()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
if !errors.Is(err, context.DeadlineExceeded) {
|
||||
t.Fatalf("err = %v, want a deadline-exceeded error", err)
|
||||
}
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("Terminate never returned: the lifecycle RPC is unbounded")
|
||||
}
|
||||
}
|
||||
|
||||
// TestLaunchLeavesATighterCallerDeadlineAlone confirms the bound narrows the
|
||||
// caller's context and never widens it, so a caller that wants to give up
|
||||
// sooner still does.
|
||||
func TestLaunchLeavesATighterCallerDeadlineAlone(t *testing.T) {
|
||||
previous := launchTimeout
|
||||
launchTimeout = 30 * time.Second
|
||||
defer func() { launchTimeout = previous }()
|
||||
|
||||
companion := &wedgedLifecycleCompanion{release: make(chan struct{})}
|
||||
defer close(companion.release)
|
||||
d := newTestDriver(companion)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond)
|
||||
defer cancel()
|
||||
start := time.Now()
|
||||
if err := d.Launch(ctx, "com.example.app", false, nil); !errors.Is(err, context.DeadlineExceeded) {
|
||||
t.Fatalf("err = %v, want a deadline-exceeded error", err)
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed > 5*time.Second {
|
||||
t.Fatalf("Launch took %v; the driver's bound overrode the caller's tighter deadline", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
// newLockTestOptions builds New options that dial a seamed companion, so the
|
||||
// device-lock tests exercise New without spawning anything.
|
||||
func newLockTestOptions(t *testing.T, udid string) Options {
|
||||
t.Helper()
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { listener.Close() })
|
||||
go func() {
|
||||
for {
|
||||
connection, acceptErr := listener.Accept()
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
_ = connection.Close()
|
||||
}
|
||||
}()
|
||||
return Options{
|
||||
UniqueDeviceIdentifier: udid,
|
||||
pickAddress: func() (string, error) { return listener.Addr().String(), nil },
|
||||
spawnChild: func(context.Context, string) (*exec.Cmd, error) { return &exec.Cmd{}, nil },
|
||||
dialCompanion: func(string) (transport.Companion, error) {
|
||||
return &fakeCompanion{accessibilityJSON: "[]"}, nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestNewRejectsConcurrentRunOnSameDevice proves a second run cannot claim a
|
||||
// device the first is driving. Two runs interleave app lifecycle on one
|
||||
// simulator: the second's reinstall lands under the first's automation session
|
||||
// and wedges it. Failing fast names the contended device; the claim is released
|
||||
// on Close so the next run is not locked out.
|
||||
func TestNewRejectsConcurrentRunOnSameDevice(t *testing.T) {
|
||||
t.Setenv("SANDERLING_SIMULATOR_COMPANION", "legacy")
|
||||
udid := "LOCK-TEST-" + t.Name()
|
||||
|
||||
first, err := New(context.Background(), newLockTestOptions(t, udid))
|
||||
if err != nil {
|
||||
t.Fatalf("first New: %v", err)
|
||||
}
|
||||
|
||||
_, err = New(context.Background(), newLockTestOptions(t, udid))
|
||||
if err == nil {
|
||||
t.Fatal("second run claimed a device the first still drives; concurrent runs corrupt each other's session")
|
||||
}
|
||||
if !strings.Contains(err.Error(), udid) {
|
||||
t.Fatalf("err = %v, want it to name the contended device %s", err, udid)
|
||||
}
|
||||
|
||||
first.Close()
|
||||
third, err := New(context.Background(), newLockTestOptions(t, udid))
|
||||
if err != nil {
|
||||
t.Fatalf("device stayed locked after Close: %v", err)
|
||||
}
|
||||
third.Close()
|
||||
}
|
||||
|
||||
// TestAcquireDeviceLockKeepsDistinctDevicesIndependent guards against a lock
|
||||
// path that ignores the udid and serializes unrelated runs.
|
||||
func TestAcquireDeviceLockKeepsDistinctDevicesIndependent(t *testing.T) {
|
||||
first, err := acquireDeviceLock("LOCK-TEST-DEVICE-A")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer first.Close()
|
||||
second, err := acquireDeviceLock("LOCK-TEST-DEVICE-B")
|
||||
if err != nil {
|
||||
t.Fatalf("a second device was refused while another was locked: %v", err)
|
||||
}
|
||||
defer second.Close()
|
||||
}
|
||||
@@ -129,6 +129,9 @@ func mapElement(element *rawElement) (treeNode, bool) {
|
||||
attributes["hintText"] = label
|
||||
} else if label != "" {
|
||||
attributes["accessibilityText"] = label
|
||||
if labelIsDisplayedText(element.Type) {
|
||||
attributes["text"] = label
|
||||
}
|
||||
}
|
||||
|
||||
enabled := element.Enabled
|
||||
@@ -150,6 +153,22 @@ func isEditable(elementType string) bool {
|
||||
return elementType == "TextArea" || elementType == "TextField"
|
||||
}
|
||||
|
||||
// labelIsDisplayedText reports whether an element type's AXLabel is the string
|
||||
// drawn on screen rather than an accessibility annotation about it. Only
|
||||
// StaticText qualifies: a text element's label IS what it renders, so it
|
||||
// belongs in `text`, matching a TextView on Android and a text node on web.
|
||||
//
|
||||
// Buttons and images are deliberately excluded even though a titled button's
|
||||
// label is also its visible title. The snapshot cannot tell that button apart
|
||||
// from an icon-only one whose label exists purely for VoiceOver, nor from a
|
||||
// container whose label is a comma-joined reading of its children ("CH,
|
||||
// Checking, $0.00, 0 transactions"). Inventing `text` for those would put
|
||||
// strings in `text` that no user can read, and would diverge from Android,
|
||||
// which leaves `text` empty and reports a contentDescription as `description`.
|
||||
func labelIsDisplayedText(elementType string) bool {
|
||||
return elementType == "StaticText"
|
||||
}
|
||||
|
||||
func stringValue(pointer *string) string {
|
||||
if pointer == nil {
|
||||
return ""
|
||||
|
||||
@@ -132,6 +132,79 @@ func TestNonEmptyValueMapsToText(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestStaticTextLabelMapsToText(t *testing.T) {
|
||||
dump := `[{"type":"StaticText","frame":{"x":0,"y":0,"width":50,"height":18},
|
||||
"AXLabel":"$239.00","AXValue":null,"enabled":true}]`
|
||||
element := parseSingle(t, dump)
|
||||
// A StaticText renders its label, so a spec reading .text must see it.
|
||||
if element.Text != "$239.00" {
|
||||
t.Fatalf("text = %q, want $239.00", element.Text)
|
||||
}
|
||||
// The raw label stays available: desc:/label:/content-desc: selectors and
|
||||
// the settle hash read it on the iOS path.
|
||||
if element.Description != "$239.00" {
|
||||
t.Fatalf("description = %q, want $239.00", element.Description)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonTextLabelStaysDescriptionOnly(t *testing.T) {
|
||||
// An icon-only button's label is a VoiceOver annotation, and a container's
|
||||
// is a comma-joined reading of its children. Neither is on screen, so
|
||||
// neither may become text; Android reports both as description too.
|
||||
cases := []struct{ elementType, label string }{
|
||||
{"Button", "Log out"},
|
||||
{"Image", "Avatar"},
|
||||
{"Other", "CH, Checking, $0.00, 0 transactions"},
|
||||
}
|
||||
for _, testCase := range cases {
|
||||
dump := `[{"type":"` + testCase.elementType + `","frame":{"x":0,"y":0,"width":10,"height":10},
|
||||
"AXLabel":"` + testCase.label + `","AXValue":null,"enabled":true}]`
|
||||
element := parseSingle(t, dump)
|
||||
if element.Text != "" {
|
||||
t.Errorf("%s text = %q, want empty", testCase.elementType, element.Text)
|
||||
}
|
||||
if element.Description != testCase.label {
|
||||
t.Errorf("%s description = %q, want %q", testCase.elementType, element.Description, testCase.label)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Mirrors the folio Home screen as the companion actually dumps it: each
|
||||
// AccountCard is a Button carrying a merged VoiceOver label, with the name and
|
||||
// balance as StaticText siblings that the flat tree re-parents by containment.
|
||||
// Reading a card's balance is what folio's totalBalance extractor does, and it
|
||||
// read nothing on iOS until StaticText labels became text.
|
||||
func TestGoldenHomeCardBalancesAreReadableAsText(t *testing.T) {
|
||||
tree := mapAndParse(t, readDump(t, "home-cards-describe.json"), 402, 874)
|
||||
|
||||
cards := tree.FindAllNodes("id:HomeScreen > id:AccountCard")
|
||||
if len(cards) != 2 {
|
||||
t.Fatalf("account cards = %d, want 2", len(cards))
|
||||
}
|
||||
want := []struct{ name, balance string }{{"Checking", "$12.34"}, {"Savings", "$500.00"}}
|
||||
for i, card := range cards {
|
||||
balance := card.Find("id:AccountBalance")
|
||||
if balance == nil {
|
||||
t.Fatalf("card %d: id:AccountBalance did not resolve", i)
|
||||
}
|
||||
if balance.Text != want[i].balance {
|
||||
t.Errorf("card %d: balance text = %q, want %q", i, balance.Text, want[i].balance)
|
||||
}
|
||||
name := card.Find("id:AccountName")
|
||||
if name == nil {
|
||||
t.Fatalf("card %d: id:AccountName did not resolve", i)
|
||||
}
|
||||
if name.Text != want[i].name {
|
||||
t.Errorf("card %d: name text = %q, want %q", i, name.Text, want[i].name)
|
||||
}
|
||||
// The card's own merged label is not visible text; the web fallback in
|
||||
// the folio spec parses cardText and must not see a VoiceOver reading.
|
||||
if card.Element.Text != "" {
|
||||
t.Errorf("card %d: card text = %q, want empty", i, card.Element.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEditableAndClickableFlags(t *testing.T) {
|
||||
cases := []struct {
|
||||
elementType string
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
[
|
||||
{"type":"Application","frame":{"x":0,"y":0,"width":402,"height":874},"enabled":true,"AXLabel":"Folio","AXValue":null,"AXUniqueId":null},
|
||||
{"type":"Other","frame":{"x":0,"y":62,"width":402,"height":778},"enabled":true,"AXLabel":null,"AXValue":null,"AXUniqueId":"HomeScreen"},
|
||||
{"type":"StaticText","frame":{"x":20,"y":76,"width":97,"height":24},"enabled":true,"AXLabel":"Accounts","AXValue":null,"AXUniqueId":null},
|
||||
{"type":"Button","frame":{"x":340,"y":71,"width":48,"height":48},"enabled":true,"AXLabel":"Log out","AXValue":null,"AXUniqueId":"LogoutButton"},
|
||||
{"type":"Button","frame":{"x":20,"y":130,"width":362,"height":72},"enabled":true,"AXLabel":"CH, Checking, $12.34, 0 transactions","AXValue":null,"AXUniqueId":"AccountCard"},
|
||||
{"type":"StaticText","frame":{"x":90,"y":150,"width":74,"height":18},"enabled":true,"AXLabel":"Checking","AXValue":null,"AXUniqueId":"AccountName"},
|
||||
{"type":"StaticText","frame":{"x":313,"y":157,"width":53,"height":18},"enabled":true,"AXLabel":"$12.34","AXValue":null,"AXUniqueId":"AccountBalance"},
|
||||
{"type":"Button","frame":{"x":20,"y":210,"width":362,"height":72},"enabled":true,"AXLabel":"SA, Savings, $500.00, 2 transactions","AXValue":null,"AXUniqueId":"AccountCard"},
|
||||
{"type":"StaticText","frame":{"x":90,"y":230,"width":66,"height":18},"enabled":true,"AXLabel":"Savings","AXValue":null,"AXUniqueId":"AccountName"},
|
||||
{"type":"StaticText","frame":{"x":306,"y":237,"width":60,"height":18},"enabled":true,"AXLabel":"$500.00","AXValue":null,"AXUniqueId":"AccountBalance"},
|
||||
{"type":"StaticText","frame":{"x":20,"y":716,"width":106,"height":14},"enabled":true,"AXLabel":"TOTAL BALANCE","AXValue":null,"AXUniqueId":null},
|
||||
{"type":"StaticText","frame":{"x":20,"y":731,"width":85,"height":33},"enabled":true,"AXLabel":"$512.34","AXValue":null,"AXUniqueId":null},
|
||||
{"type":"Button","frame":{"x":20,"y":777,"width":362,"height":48},"enabled":true,"AXLabel":"+ Add account","AXValue":null,"AXUniqueId":"AddAccountButton"},
|
||||
{"type":"StaticText","frame":{"x":141,"y":792,"width":121,"height":18},"enabled":true,"AXLabel":"+ Add account","AXValue":null,"AXUniqueId":null}
|
||||
]
|
||||
+81
-19
@@ -31,6 +31,11 @@ type Options struct {
|
||||
// positive value stops the loop once that many steps have run.
|
||||
MaxSteps int
|
||||
|
||||
// StopOnViolation ends the step loop as soon as a step records a
|
||||
// violation, so a run that exists to find one bug stops at the evidence
|
||||
// instead of spending the rest of its budget past it.
|
||||
StopOnViolation bool
|
||||
|
||||
BundleID string
|
||||
Driver driver.DeviceDriver
|
||||
Verifier *verifier.Verifier
|
||||
@@ -74,6 +79,7 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
if logger == nil {
|
||||
logger = slog.Default()
|
||||
}
|
||||
options.IdleTimeout = resolveIdleTimeout(options)
|
||||
|
||||
// Gate on the app actually being on top before acting, so the first
|
||||
// action never fires against a leftover screen or a system dialog. Done
|
||||
@@ -87,6 +93,7 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
if err != nil {
|
||||
return Summary{}, err
|
||||
}
|
||||
_, pageExtractors := extractorSource.(webSource)
|
||||
|
||||
summary := Summary{StartTime: time.Now()}
|
||||
deadline := summary.StartTime.Add(options.Duration)
|
||||
@@ -122,9 +129,8 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
var logs []verifier.LogEntry
|
||||
|
||||
// gctx is bound to the errgroup so a returned error (or outer
|
||||
// cancellation) propagates to siblings - notably the V8 extractor
|
||||
// goroutine, whose CDP round-trip can otherwise outrun the step
|
||||
// budget on a hung tab.
|
||||
// cancellation) propagates to every sibling read rather than leaving
|
||||
// one blocked on a hung device.
|
||||
g, gctx := errgroup.WithContext(ctx)
|
||||
si := stepIndex
|
||||
// fetchSyncedState issues a single Snapshot RPC so hierarchy and
|
||||
@@ -143,16 +149,6 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
logs = collectLogs(gctx, options.Driver, logSince)
|
||||
return nil
|
||||
})
|
||||
var v8Overrides map[int]json.RawMessage
|
||||
g.Go(func() error {
|
||||
overrides, err := extractorSource.ExtractorOverrides(gctx)
|
||||
if err != nil {
|
||||
logger.Warn("v8 extractor evaluation failed", "step", si, "err", err)
|
||||
return nil
|
||||
}
|
||||
v8Overrides = overrides
|
||||
return nil
|
||||
})
|
||||
// All goroutines write to local variables and return nil, so the Wait
|
||||
// error is always nil; ignored intentionally.
|
||||
_ = g.Wait()
|
||||
@@ -195,6 +191,29 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
var witnesses map[string]trace.Witness
|
||||
skippedVerification := false
|
||||
if !transitional {
|
||||
// The page-side extractors evaluate only on steps the verifier will
|
||||
// accept, which is why this read waits for the tree instead of
|
||||
// racing it. A spec's extractor getters carry state across steps
|
||||
// (folio's last-seen Home total, its submit counters) and that state
|
||||
// advances every time they run: evaluating them on a step whose
|
||||
// values are then thrown away leaves the page one window ahead of
|
||||
// the verifier, so the next accepted pair brackets two committed
|
||||
// transactions while having counted one submit, and the property
|
||||
// convicts a healthy app. It costs the latency the read used to hide
|
||||
// behind the hierarchy fetch; the fetch is what decides whether this
|
||||
// step counts at all, so it has to go first.
|
||||
//
|
||||
// lastAction is the same value PushSnapshot hands the goja state
|
||||
// below: the two engines evaluate this step against one action.
|
||||
v8Overrides, overridesErr := extractorSource.ExtractorOverrides(ctx, lastAction)
|
||||
if overridesErr != nil {
|
||||
// Not a warning. Without the page's values this step's
|
||||
// extractors keep goja's dump-derived readings while the
|
||||
// previous step holds the page's, and a delta property then
|
||||
// compares two producers and fires on an app that did nothing
|
||||
// wrong.
|
||||
return summary, fmt.Errorf("step %d extractor overrides: %w", stepIndex, overridesErr)
|
||||
}
|
||||
if err := options.Verifier.PushSnapshot(verifier.SnapshotInput{
|
||||
Tree: tree,
|
||||
ScreenshotPNG: screenshotPNG,
|
||||
@@ -206,13 +225,26 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
}); err != nil {
|
||||
return summary, fmt.Errorf("step %d push: %w", stepIndex, err)
|
||||
}
|
||||
// Every failure below leaves some extractors holding the page's
|
||||
// value and the rest holding goja's reading of the dump, and a
|
||||
// property comparing previous to current across that split fires
|
||||
// on a healthy app. Each also means the two engines loaded
|
||||
// different bundles, which nothing downstream can reconcile.
|
||||
if pageExtractors && len(v8Overrides) != options.Verifier.ExtractorCount() {
|
||||
return summary, fmt.Errorf(
|
||||
"step %d: the page reported values for %d of the spec's %d extractors; "+
|
||||
"the page and the host are running different bundles",
|
||||
stepIndex, len(v8Overrides), options.Verifier.ExtractorCount())
|
||||
}
|
||||
skipped, overrideErr := options.Verifier.OverrideExtractorValues(v8Overrides)
|
||||
if overrideErr != nil {
|
||||
logger.Warn("v8 override apply failed", "step", stepIndex, "err", overrideErr)
|
||||
return summary, fmt.Errorf("step %d apply extractor overrides: %w", stepIndex, overrideErr)
|
||||
}
|
||||
if skipped > 0 {
|
||||
logger.Warn("v8 override skipped out-of-range entries",
|
||||
"step", stepIndex, "skipped", skipped, "have", len(v8Overrides))
|
||||
return summary, fmt.Errorf(
|
||||
"step %d: %d of %d extractor overrides fell outside the spec's extractor list; "+
|
||||
"the page and the host are running different bundles",
|
||||
stepIndex, skipped, len(v8Overrides))
|
||||
}
|
||||
options.Verifier.EvaluateProperties()
|
||||
violations = options.Verifier.NewlyViolatedProperties()
|
||||
@@ -317,6 +349,12 @@ func Run(ctx context.Context, options Options) (Summary, error) {
|
||||
summary.Steps = stepIndex
|
||||
if len(violations) > 0 {
|
||||
summary.Violations = append(summary.Violations, violationRecords(violations, witnesses, stepIndex)...)
|
||||
// The step is already written, so the trace ends on the state that
|
||||
// produced the violation. Finalize below still runs, so pending
|
||||
// liveness obligations are reported alongside it.
|
||||
if options.StopOnViolation {
|
||||
break
|
||||
}
|
||||
}
|
||||
// Wait actions are themselves a settling: skip the idle poll. Actions
|
||||
// that mutate the UI fall through to WaitForIdle so the next step's
|
||||
@@ -391,12 +429,36 @@ func validate(options Options) error {
|
||||
if options.Duration <= 0 {
|
||||
return errors.New("runner: Duration must be positive")
|
||||
}
|
||||
if options.IdleTimeout <= 0 {
|
||||
options.IdleTimeout = 2 * time.Second
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// defaultIdleTimeout is the settle budget a caller that names none gets.
|
||||
const defaultIdleTimeout = 2 * time.Second
|
||||
|
||||
// idleTimeoutFloor is a driver that knows how long its own settle can take.
|
||||
// Declared here rather than in the driver package (like lastActionInstaller in
|
||||
// source.go) so the mobile drivers stay untouched.
|
||||
type idleTimeoutFloor interface {
|
||||
MinIdleTimeout() time.Duration
|
||||
}
|
||||
|
||||
// resolveIdleTimeout settles the per-step settle budget: the caller's value,
|
||||
// defaulted when unset, and raised to whatever the driver says its own settle
|
||||
// needs. The chrome driver's settle waits for the DOM to go quiet and only then
|
||||
// opens its route-transition window; handed less than their sum it is cut off
|
||||
// mid-transition, and the step samples the screen the app is leaving. A driver
|
||||
// that reports no floor keeps the caller's value exactly.
|
||||
func resolveIdleTimeout(options Options) time.Duration {
|
||||
timeout := options.IdleTimeout
|
||||
if timeout <= 0 {
|
||||
timeout = defaultIdleTimeout
|
||||
}
|
||||
if floor, ok := options.Driver.(idleTimeoutFloor); ok {
|
||||
timeout = max(timeout, floor.MinIdleTimeout())
|
||||
}
|
||||
return timeout
|
||||
}
|
||||
|
||||
// ensureForeground keeps the app under test in the foreground. When the driver
|
||||
// can report the foreground app and it no longer matches the bundle under test,
|
||||
// the app is relaunched. Returns true when a relaunch happened so the caller
|
||||
|
||||
@@ -2343,3 +2343,99 @@ func TestRunner_SkipsActionWhenOverlayStealsFocusAtApplyTime(t *testing.T) {
|
||||
t.Errorf("no step recorded action_skipped=%q, so the undispatched action looks executed", actionSkippedForeground)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunner_StopOnViolationEndsAtTheFirstViolation pins the gate CI runs on:
|
||||
// a step budget of 8 against a spec that only violates on the third step must
|
||||
// end on step 3 and write nothing after it, so the trace's last state is the
|
||||
// one that produced the violation.
|
||||
func TestRunner_StopOnViolationEndsAtTheFirstViolation(t *testing.T) {
|
||||
const thirdStepViolationSpec = `
|
||||
import { actions, always, extract } from "@sanderling/spec";
|
||||
let observed = 0;
|
||||
const tick = extract(() => ++observed);
|
||||
globalThis.properties = {
|
||||
staysUnderThree: always(() => tick.current < 3),
|
||||
};
|
||||
globalThis.actions = actions(() => []);
|
||||
`
|
||||
state := newHarnessWithSpec(t, thirdStepViolationSpec)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
summary, err := Run(ctx, Options{
|
||||
Duration: time.Hour,
|
||||
IdleTimeout: 20 * time.Millisecond,
|
||||
MaxSteps: 8,
|
||||
StopOnViolation: true,
|
||||
Driver: state.mock,
|
||||
Verifier: state.verifier,
|
||||
TraceWriter: state.writer,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Run: %v", err)
|
||||
}
|
||||
if !containsProperty(summary.Violations, "staysUnderThree") {
|
||||
t.Fatalf("expected staysUnderThree to fire, got %v", summary.Violations)
|
||||
}
|
||||
if summary.Steps != 3 {
|
||||
t.Errorf("steps: got %d, want 3 (the run must stop at the violating step, not run the 8-step budget)",
|
||||
summary.Steps)
|
||||
}
|
||||
for _, step := range traceStepIndices(t, state.writer.Directory()) {
|
||||
if step > summary.Steps {
|
||||
t.Errorf("trace kept stepping after the violation: found step %d past step %d",
|
||||
step, summary.Steps)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunner_WithoutStopOnViolationRunsTheWholeBudget is the other half: the
|
||||
// default must stay a full-budget fuzz run, so turning the flag on is the only
|
||||
// thing that shortens a run.
|
||||
func TestRunner_WithoutStopOnViolationRunsTheWholeBudget(t *testing.T) {
|
||||
state := newHarnessWithSpec(t, violationSpec)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
summary, err := Run(ctx, Options{
|
||||
Duration: time.Hour,
|
||||
IdleTimeout: 20 * time.Millisecond,
|
||||
MaxSteps: 4,
|
||||
Driver: state.mock,
|
||||
Verifier: state.verifier,
|
||||
TraceWriter: state.writer,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Run: %v", err)
|
||||
}
|
||||
if summary.Steps != 4 {
|
||||
t.Errorf("steps: got %d, want 4; a violation must not shorten a default run", summary.Steps)
|
||||
}
|
||||
}
|
||||
|
||||
// traceStepIndices reads every step index the trace recorded, so a test can
|
||||
// assert on what the run actually wrote rather than on the summary alone.
|
||||
func traceStepIndices(t *testing.T, directory string) []int {
|
||||
t.Helper()
|
||||
file, err := os.Open(filepath.Join(directory, "trace.jsonl"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer file.Close()
|
||||
var steps []int
|
||||
scanner := bufio.NewScanner(file)
|
||||
scanner.Buffer(make([]byte, 0, 64*1024), 8*1024*1024)
|
||||
for scanner.Scan() {
|
||||
var line struct {
|
||||
Step int `json:"step"`
|
||||
}
|
||||
if err := json.Unmarshal(scanner.Bytes(), &line); err != nil {
|
||||
t.Fatalf("trace line decode: %v", err)
|
||||
}
|
||||
steps = append(steps, line.Step)
|
||||
}
|
||||
if err := scanner.Err(); err != nil {
|
||||
t.Fatalf("scan trace: %v", err)
|
||||
}
|
||||
return steps
|
||||
}
|
||||
@@ -25,8 +25,23 @@ type ActionSource interface {
|
||||
// ExtractorSource yields per-step extractor overrides the runner applies after
|
||||
// PushSnapshot. The mobile path has none (returns nil); the web path returns the
|
||||
// values its extractors computed in V8 against the real DOM.
|
||||
//
|
||||
// lastAction is the action the previous step actually applied, the same value
|
||||
// PushSnapshot hands the goja state. The web path has to install it in the page
|
||||
// before its extractors run: a spec extractor reading state.lastAction runs in
|
||||
// V8 there, and V8 has no way to know what the runner dispatched.
|
||||
type ExtractorSource interface {
|
||||
ExtractorOverrides(ctx context.Context) (map[int]json.RawMessage, error)
|
||||
ExtractorOverrides(
|
||||
ctx context.Context,
|
||||
lastAction *verifier.Action,
|
||||
) (map[int]json.RawMessage, error)
|
||||
}
|
||||
|
||||
// lastActionInstaller is the web driver's channel for the previous step's
|
||||
// action. It is declared here rather than folded into driver.WebDriver so the
|
||||
// mobile drivers stay untouched; every web driver must implement it.
|
||||
type lastActionInstaller interface {
|
||||
SetLastAction(ctx context.Context, encoded json.RawMessage) error
|
||||
}
|
||||
|
||||
// gojaSource drives both action selection and (trivially) extractor overrides
|
||||
@@ -40,7 +55,10 @@ func (s gojaSource) NextAction(context.Context, int) (verifier.Action, error) {
|
||||
return s.verifier.NextAction()
|
||||
}
|
||||
|
||||
func (gojaSource) ExtractorOverrides(context.Context) (map[int]json.RawMessage, error) {
|
||||
func (gojaSource) ExtractorOverrides(
|
||||
context.Context,
|
||||
*verifier.Action,
|
||||
) (map[int]json.RawMessage, error) {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
@@ -61,7 +79,24 @@ func (s webSource) NextAction(ctx context.Context, _ int) (verifier.Action, erro
|
||||
return verifier.DecodeAction(raw)
|
||||
}
|
||||
|
||||
func (s webSource) ExtractorOverrides(ctx context.Context) (map[int]json.RawMessage, error) {
|
||||
// ExtractorOverrides installs the previous step's action in the page, then
|
||||
// reads back what the spec's extractors computed against the live DOM. The
|
||||
// install is not best-effort: a web driver that cannot take it leaves
|
||||
// state.lastAction null in V8, which silently turns every action-gated
|
||||
// property vacuously true, so it is reported as an error instead.
|
||||
func (s webSource) ExtractorOverrides(
|
||||
ctx context.Context,
|
||||
lastAction *verifier.Action,
|
||||
) (map[int]json.RawMessage, error) {
|
||||
installer, ok := s.web.(lastActionInstaller)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf(
|
||||
"web driver %T cannot install state.lastAction; every property gated "+
|
||||
"on the last action would be vacuously true", s.web)
|
||||
}
|
||||
if err := installer.SetLastAction(ctx, verifier.EncodeLastAction(lastAction)); err != nil {
|
||||
return nil, fmt.Errorf("install last action: %w", err)
|
||||
}
|
||||
return s.web.EvaluateExtractors(ctx)
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
package runner
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/priyanshujain/sanderling/internal/driver"
|
||||
mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock"
|
||||
"github.com/priyanshujain/sanderling/internal/trace"
|
||||
)
|
||||
|
||||
// carrierSpec registers one extractor whose value the page supplies. It stands
|
||||
// in for every spec whose getters carry state across steps (folio's last-seen
|
||||
// Home total, its submit counters): what matters is that ASKING the page for
|
||||
// the value is what advances it.
|
||||
const carrierSpec = `
|
||||
import { actions, extract } from "@sanderling/spec";
|
||||
const carrier = extract("carrier", () => 0);
|
||||
globalThis.properties = {};
|
||||
globalThis.actions = actions(() => []);
|
||||
`
|
||||
|
||||
// carrierWebDriver is a web target that alternates between a cross-fading
|
||||
// hierarchy (which the runner discards as transitional) and a settled one, and
|
||||
// whose page-side extractor advances a counter on every evaluation - exactly
|
||||
// what a spec-authored carrier does in V8.
|
||||
type carrierWebDriver struct {
|
||||
*mockdriver.Driver
|
||||
transitional bool
|
||||
snapshots int
|
||||
reads int
|
||||
}
|
||||
|
||||
func (d *carrierWebDriver) Snapshot(ctx context.Context) (string, driver.Image, error) {
|
||||
_, image, err := d.Driver.Snapshot(ctx)
|
||||
d.snapshots++
|
||||
if !d.transitional {
|
||||
return `{"attributes":{"resource-id":"HomeScreen"},"children":[]}`, image, err
|
||||
}
|
||||
// A genuine cross-fade: two live routes, and a tree that keeps changing
|
||||
// between retries so the runner spends its whole retry budget on it.
|
||||
return fmt.Sprintf(`{"attributes":{"resource-id":"root"},"children":[
|
||||
{"attributes":{"resource-id":"HomeScreen","text":"frame-%d"},"children":[]},
|
||||
{"attributes":{"resource-id":"LedgerScreen"},"children":[]}
|
||||
]}`, d.snapshots), image, err
|
||||
}
|
||||
|
||||
func (d *carrierWebDriver) InstallBundle(context.Context, []byte) error { return nil }
|
||||
|
||||
func (d *carrierWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) {
|
||||
d.reads++
|
||||
return map[int]json.RawMessage{0: json.RawMessage(strconv.Itoa(d.reads))}, nil
|
||||
}
|
||||
|
||||
// NextActionFromV8 runs once per step, after the hierarchy fetch, so flipping
|
||||
// here makes every other step a cross-fade.
|
||||
func (d *carrierWebDriver) NextActionFromV8(context.Context) (json.RawMessage, error) {
|
||||
d.transitional = !d.transitional
|
||||
return json.RawMessage(`{"kind":"Tap","x":5,"y":5}`), nil
|
||||
}
|
||||
|
||||
func (d *carrierWebDriver) SetLastAction(context.Context, json.RawMessage) error { return nil }
|
||||
|
||||
// TestRunner_TransitionalStepNeverAdvancesThePageCarrier pins the ordering the
|
||||
// web path depends on. The page-side extractors must run only on steps the
|
||||
// verifier accepts: their getters advance spec state every time they evaluate,
|
||||
// so evaluating them on a step whose values are then discarded leaves the page
|
||||
// one window ahead of the verifier. The next accepted pair then brackets two
|
||||
// committed transactions while having counted one submit, and the property
|
||||
// convicts an app that did nothing wrong.
|
||||
func TestRunner_TransitionalStepNeverAdvancesThePageCarrier(t *testing.T) {
|
||||
state := newHarnessWithSpec(t, carrierSpec)
|
||||
web := &carrierWebDriver{Driver: state.mock}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
summary, err := Run(ctx, Options{
|
||||
Duration: 30 * time.Second,
|
||||
IdleTimeout: 20 * time.Millisecond,
|
||||
MaxSteps: 5,
|
||||
Driver: web,
|
||||
Verifier: state.verifier,
|
||||
TraceWriter: state.writer,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Run: %v", err)
|
||||
}
|
||||
if summary.Steps != 5 {
|
||||
t.Fatalf("steps = %d, want 5", summary.Steps)
|
||||
}
|
||||
|
||||
type traceLine struct {
|
||||
Step int `json:"step"`
|
||||
Transitional bool `json:"transitional"`
|
||||
ExtractorChanges map[string]trace.ExtractorChange `json:"extractor_changes"`
|
||||
}
|
||||
body, err := os.ReadFile(filepath.Join(state.writer.Directory(), "trace.jsonl"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
verified, transitional := 0, 0
|
||||
previous := 0
|
||||
for _, raw := range bytes.Split(bytes.TrimSpace(body), []byte("\n")) {
|
||||
var line traceLine
|
||||
if err := json.Unmarshal(raw, &line); err != nil {
|
||||
t.Fatalf("decode trace line: %v", err)
|
||||
}
|
||||
if line.Transitional {
|
||||
transitional++
|
||||
continue
|
||||
}
|
||||
verified++
|
||||
change, ok := line.ExtractorChanges["carrier"]
|
||||
if !ok {
|
||||
t.Fatalf("step %d: no carrier value reached the verifier", line.Step)
|
||||
}
|
||||
current, convErr := strconv.Atoi(string(change.Curr))
|
||||
if convErr != nil {
|
||||
t.Fatalf("step %d: carrier value %s: %v", line.Step, change.Curr, convErr)
|
||||
}
|
||||
if current != previous+1 {
|
||||
t.Errorf("step %d: carrier went %d -> %d; the page advanced it on a "+
|
||||
"step the verifier discarded, so the verifier's window is wider "+
|
||||
"than the one the spec counted actions over",
|
||||
line.Step, previous, current)
|
||||
}
|
||||
previous = current
|
||||
}
|
||||
if verified == 0 || transitional == 0 {
|
||||
t.Fatalf("need both kinds of step to prove anything: %d verified, %d transitional",
|
||||
verified, transitional)
|
||||
}
|
||||
if web.reads != verified {
|
||||
t.Errorf("the page evaluated its extractors %d time(s) across %d verified step(s); "+
|
||||
"every evaluation the verifier does not use still advances spec state",
|
||||
web.reads, verified)
|
||||
}
|
||||
}
|
||||
|
||||
// installFailsWebDriver is a web target whose page cannot take the runner's
|
||||
// lastAction: an older published @sanderling/spec runtime, a bundle that never
|
||||
// installed, a tab that navigated away from it.
|
||||
type installFailsWebDriver struct {
|
||||
*mockdriver.Driver
|
||||
}
|
||||
|
||||
func (d *installFailsWebDriver) InstallBundle(context.Context, []byte) error { return nil }
|
||||
|
||||
func (d *installFailsWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) {
|
||||
return map[int]json.RawMessage{0: json.RawMessage(`1`)}, nil
|
||||
}
|
||||
|
||||
func (d *installFailsWebDriver) NextActionFromV8(context.Context) (json.RawMessage, error) {
|
||||
return json.RawMessage(`{"kind":"Tap","x":5,"y":5}`), nil
|
||||
}
|
||||
|
||||
func (d *installFailsWebDriver) SetLastAction(context.Context, json.RawMessage) error {
|
||||
return errors.New("__sanderlingSetLastAction__ is not a function")
|
||||
}
|
||||
|
||||
// TestRunner_LastActionInstallFailureFailsTheRun covers the other half of the
|
||||
// same trust boundary. A run that cannot install lastAction in the page cannot
|
||||
// apply the page's extractor values either, so the step keeps goja's
|
||||
// dump-derived readings while the step before it holds the page's, and a delta
|
||||
// property compares two producers and fires. Downgraded to a warning that is a
|
||||
// green run reporting a violation nobody can reproduce.
|
||||
func TestRunner_LastActionInstallFailureFailsTheRun(t *testing.T) {
|
||||
state := newHarnessWithSpec(t, carrierSpec)
|
||||
web := &installFailsWebDriver{Driver: state.mock}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
_, err := Run(ctx, Options{
|
||||
Duration: 2 * time.Second,
|
||||
IdleTimeout: 20 * time.Millisecond,
|
||||
MaxSteps: 3,
|
||||
Driver: web,
|
||||
Verifier: state.verifier,
|
||||
TraceWriter: state.writer,
|
||||
})
|
||||
if err == nil {
|
||||
t.Fatal("Run succeeded with a page that cannot take lastAction; the run " +
|
||||
"reported green while its extractor values came from two engines")
|
||||
}
|
||||
if !bytes.Contains([]byte(err.Error()), []byte("install last action")) {
|
||||
t.Errorf("Run error = %v, want it to name the failed lastAction install", err)
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -44,6 +45,8 @@ func (d *webMockDriver) NextActionFromV8(context.Context) (json.RawMessage, erro
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func (d *webMockDriver) SetLastAction(context.Context, json.RawMessage) error { return nil }
|
||||
|
||||
// TestRunner_TraceRecordsTheValueTheVerdictUsed fails if the trace and the
|
||||
// verdict disagree about an extractor. A witness is only an explanation of a
|
||||
// violation if it holds the state the violated property was evaluated against.
|
||||
@@ -116,3 +119,48 @@ func TestRunner_TraceRecordsTheValueTheVerdictUsed(t *testing.T) {
|
||||
t.Error("no witness reached the trace; nothing was compared")
|
||||
}
|
||||
}
|
||||
|
||||
// splitTableSpec registers two extractors whose goja bodies both answer "goja".
|
||||
// The page below reports only the first, so index 1 keeps goja's dump-derived
|
||||
// reading while index 0 holds the page's.
|
||||
const splitTableSpec = `
|
||||
import { actions, extract } from "@sanderling/spec";
|
||||
extract("first", () => "goja");
|
||||
extract("second", () => "goja");
|
||||
globalThis.properties = {};
|
||||
globalThis.actions = actions(() => []);
|
||||
`
|
||||
|
||||
// TestRunner_PartialExtractorTableIsFatal pins the failure the runner used to
|
||||
// let through. JSON.stringify drops an undefined-valued key, so a page whose
|
||||
// extractors are mostly undefined off their own screen reported a table with
|
||||
// holes in it, and the run completed with half the extractors reading from V8
|
||||
// and half from goja. A delta property spanning that split convicts an app that
|
||||
// did nothing wrong, which is worse than a crash: it is a green report of a bug
|
||||
// that is not there, or a red one for a bug nobody can reproduce.
|
||||
func TestRunner_PartialExtractorTableIsFatal(t *testing.T) {
|
||||
state := newHarnessWithSpec(t, splitTableSpec)
|
||||
web := &webMockDriver{
|
||||
Driver: state.mock,
|
||||
overrides: map[int]json.RawMessage{0: json.RawMessage(`"v8"`)},
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
_, err := Run(ctx, Options{
|
||||
Duration: time.Hour,
|
||||
IdleTimeout: 20 * time.Millisecond,
|
||||
MaxSteps: 2,
|
||||
Driver: web,
|
||||
Verifier: state.verifier,
|
||||
TraceWriter: state.writer,
|
||||
})
|
||||
if err == nil {
|
||||
t.Fatal("the run completed on a page that reported 1 of 2 extractors; " +
|
||||
"the second extractor silently kept goja's value")
|
||||
}
|
||||
const want = "the page reported values for 1 of the spec's 2 extractors"
|
||||
if !strings.Contains(err.Error(), want) {
|
||||
t.Errorf("Run failed with %q, want it to name the split: %q", err, want)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
package runner
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock"
|
||||
)
|
||||
|
||||
// On web the spec's extractors run in the page, so state.lastAction has to be
|
||||
// installed there by the runner. It used to be hardcoded null in
|
||||
// pkg/spec/src/web-runtime.ts, which made every property gated on the last
|
||||
// action (folio's submitMovesBalanceByTypedAmount, for one) vacuously true on
|
||||
// web: no failure, no warning, just a green run that proved nothing.
|
||||
|
||||
const lastActionSpec = `
|
||||
import { actions } from "@sanderling/spec";
|
||||
globalThis.actions = actions(() => []);
|
||||
globalThis.properties = {};
|
||||
`
|
||||
|
||||
// tappingWebDriver is a web target whose V8 picker always taps one named
|
||||
// control, so the runner has a real applied action to report on the next step.
|
||||
type tappingWebDriver struct {
|
||||
*mockdriver.Driver
|
||||
installed []string
|
||||
}
|
||||
|
||||
func (d *tappingWebDriver) InstallBundle(context.Context, []byte) error { return nil }
|
||||
|
||||
func (d *tappingWebDriver) EvaluateExtractors(context.Context) (map[int]json.RawMessage, error) {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func (d *tappingWebDriver) NextActionFromV8(context.Context) (json.RawMessage, error) {
|
||||
return json.RawMessage(`{"kind":"Tap","x":12,"y":34,"selector":"id:TxnSubmit"}`), nil
|
||||
}
|
||||
|
||||
func (d *tappingWebDriver) SetLastAction(_ context.Context, encoded json.RawMessage) error {
|
||||
d.installed = append(d.installed, string(encoded))
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestRunner_WebInstallsLastActionInThePage(t *testing.T) {
|
||||
state := newHarnessWithSpec(t, lastActionSpec)
|
||||
web := &tappingWebDriver{Driver: state.mock}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
if _, err := Run(ctx, Options{
|
||||
Duration: time.Hour,
|
||||
IdleTimeout: 20 * time.Millisecond,
|
||||
MaxSteps: 3,
|
||||
Driver: web,
|
||||
Verifier: state.verifier,
|
||||
TraceWriter: state.writer,
|
||||
}); err != nil {
|
||||
t.Fatalf("Run: %v", err)
|
||||
}
|
||||
|
||||
if len(web.installed) < 2 {
|
||||
t.Fatalf("the page was handed lastAction %d time(s); the web path never installed it",
|
||||
len(web.installed))
|
||||
}
|
||||
// Step 1 has no previous action, exactly as the goja host reports it.
|
||||
if web.installed[0] != "null" {
|
||||
t.Errorf("step 1 installed %s, want null", web.installed[0])
|
||||
}
|
||||
// Every later step carries what the runner actually applied. The shape is
|
||||
// the goja host's (internal/verifier/marshal.go lastActionFields), pinned
|
||||
// against it by TestLastAction_WebJSONMatchesTheGojaObject.
|
||||
const want = `{"kind":"Tap","on":"id:TxnSubmit"}`
|
||||
if web.installed[1] != want {
|
||||
t.Errorf("step 2 installed %s, want %s", web.installed[1], want)
|
||||
}
|
||||
}
|
||||
@@ -20,6 +20,26 @@ import (
|
||||
|
||||
const sidecarStartupTimeout = 30 * time.Second
|
||||
|
||||
// launchTimeout bounds the pre-run app launch. It happens before the runner
|
||||
// starts, so --duration does not cover it, and Execute's context is the bare
|
||||
// signal-aware root with no deadline of its own: a driver wedged here would
|
||||
// hang the run forever having printed nothing and written no trace. Generous
|
||||
// enough to sit above every driver's own launch bound (the iOS clear-state path
|
||||
// reinstalls the app first) so a driver-level error is what a user usually
|
||||
// sees, and this stays the backstop. A variable so the timeout test can shrink
|
||||
// it.
|
||||
var launchTimeout = 3 * time.Minute
|
||||
|
||||
// launchApp starts the app under test under a bounded context.
|
||||
func launchApp(ctx context.Context, activeDriver driver.DeviceDriver, options Options) error {
|
||||
launchCtx, cancel := context.WithTimeout(ctx, launchTimeout)
|
||||
defer cancel()
|
||||
if err := activeDriver.Launch(launchCtx, options.BundleID, options.ClearData, nil); err != nil {
|
||||
return fmt.Errorf("launch app: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Options are the parameters for a single test pipeline run.
|
||||
type Options struct {
|
||||
Spec string
|
||||
@@ -38,6 +58,10 @@ type Options struct {
|
||||
// Arm labels the experiment cell this run belongs to and is recorded in
|
||||
// meta.json so a directory of runs can be attributed to a cell.
|
||||
Arm string
|
||||
// ExitOnViolation stops the run at the first violation and reports the
|
||||
// recorded violations as a ViolationsError, so a caller (CI) can tell
|
||||
// "the run found the bug" from "the run finished clean".
|
||||
ExitOnViolation bool
|
||||
// Generator selects the action picker: "llm" or the default seeded picker.
|
||||
Generator string
|
||||
// LabelSource selects how candidates are named to the model picker, and is
|
||||
@@ -145,8 +169,8 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error {
|
||||
}
|
||||
defer cleanup()
|
||||
|
||||
if err := activeDriver.Launch(ctx, options.BundleID, options.ClearData, nil); err != nil {
|
||||
return fmt.Errorf("launch app: %w", err)
|
||||
if err := launchApp(ctx, activeDriver, options); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if web, ok := activeDriver.(driver.WebDriver); ok && len(webBundle.JavaScript) > 0 {
|
||||
@@ -203,6 +227,7 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error {
|
||||
Logger: newProgressLogger(stdout),
|
||||
Generator: options.Generator,
|
||||
LabelSource: options.LabelSource,
|
||||
StopOnViolation: options.ExitOnViolation,
|
||||
})
|
||||
|
||||
terminateCtx, terminateCancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
@@ -215,9 +240,32 @@ func Execute(ctx context.Context, options Options, stdout io.Writer) error {
|
||||
|
||||
fmt.Fprintf(stdout, "\nelapsed: %s\n", summary.EndTime.Sub(summary.StartTime).Round(time.Millisecond))
|
||||
runner.RenderSummary(stdout, summary, options.Platform)
|
||||
return runOutcome(options, summary)
|
||||
}
|
||||
|
||||
// runOutcome turns a finished run into the pipeline's result. Without
|
||||
// --exit-on-violation a run that found violations is still a successful run
|
||||
// (the summary reports them), which is the behaviour every existing caller
|
||||
// depends on.
|
||||
func runOutcome(options Options, summary runner.Summary) error {
|
||||
if options.ExitOnViolation && len(summary.Violations) > 0 {
|
||||
return ViolationsError{Count: len(summary.Violations)}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ViolationsError reports a run that recorded violations under
|
||||
// --exit-on-violation. It is deliberately distinct from every other error the
|
||||
// pipeline returns: those mean the harness broke, this one means the run did
|
||||
// its job and found something.
|
||||
type ViolationsError struct {
|
||||
Count int
|
||||
}
|
||||
|
||||
func (e ViolationsError) Error() string {
|
||||
return fmt.Sprintf("%d violation record(s)", e.Count)
|
||||
}
|
||||
|
||||
// bundleInputs holds the pre-driver assembly: alias map, seed, esbuild defines,
|
||||
// and the resolved spec-API/goja-runtime paths the bundler consumes.
|
||||
type bundleInputs struct {
|
||||
|
||||
@@ -1,12 +1,16 @@
|
||||
package testrun
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/priyanshujain/sanderling/internal/driver"
|
||||
"github.com/priyanshujain/sanderling/internal/runner"
|
||||
"github.com/priyanshujain/sanderling/internal/verifier"
|
||||
)
|
||||
|
||||
@@ -254,3 +258,72 @@ func TestBuildRunMeta_OmitsModelWhenSpecDeclaresNoLLMGenerator(t *testing.T) {
|
||||
t.Errorf("model recorded without a spec-declared llm generator: %q", meta.Model)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunOutcome_ReportsViolationsOnlyUnderTheFlag pins the CI contract: the
|
||||
// typed error is what makes `sanderling test` exit 2, and it must appear only
|
||||
// when the caller asked for it. A run that finds violations without the flag
|
||||
// stays a successful run, which is what every existing invocation expects.
|
||||
func TestRunOutcome_ReportsViolationsOnlyUnderTheFlag(t *testing.T) {
|
||||
violated := runner.Summary{
|
||||
Steps: 7,
|
||||
Violations: []runner.ViolationRecord{{StepIndex: 3, Properties: []string{"balanceMoves"}}},
|
||||
}
|
||||
clean := runner.Summary{Steps: 7}
|
||||
|
||||
if err := runOutcome(Options{}, violated); err != nil {
|
||||
t.Errorf("without --exit-on-violation a violated run must succeed, got %v", err)
|
||||
}
|
||||
if err := runOutcome(Options{ExitOnViolation: true}, clean); err != nil {
|
||||
t.Errorf("a clean run must succeed under --exit-on-violation, got %v", err)
|
||||
}
|
||||
|
||||
err := runOutcome(Options{ExitOnViolation: true}, violated)
|
||||
var violations ViolationsError
|
||||
if !errors.As(err, &violations) {
|
||||
t.Fatalf("expected a ViolationsError, got %v", err)
|
||||
}
|
||||
if violations.Count != 1 {
|
||||
t.Errorf("count: got %d, want 1", violations.Count)
|
||||
}
|
||||
}
|
||||
|
||||
// wedgedLaunchDriver never returns from Launch, standing in for a driver whose
|
||||
// device-side session is stuck.
|
||||
type wedgedLaunchDriver struct {
|
||||
driver.DeviceDriver
|
||||
release chan struct{}
|
||||
}
|
||||
|
||||
func (w *wedgedLaunchDriver) Launch(ctx context.Context, _ string, _ bool, _ map[string]string) error {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-w.release:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// TestLaunchAppBoundsWedgedDriver proves the pre-run launch carries a deadline.
|
||||
// It runs before the runner starts, so --duration does not cover it and
|
||||
// Execute's root context has no deadline: unbounded, a wedged driver hangs the
|
||||
// run forever with no trace directory and no error.
|
||||
func TestLaunchAppBoundsWedgedDriver(t *testing.T) {
|
||||
previous := launchTimeout
|
||||
launchTimeout = 100 * time.Millisecond
|
||||
defer func() { launchTimeout = previous }()
|
||||
|
||||
wedged := &wedgedLaunchDriver{release: make(chan struct{})}
|
||||
defer close(wedged.release)
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- launchApp(context.Background(), wedged, Options{BundleID: "com.example.app"}) }()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
if !errors.Is(err, context.DeadlineExceeded) {
|
||||
t.Fatalf("err = %v, want a deadline-exceeded error", err)
|
||||
}
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("launchApp never returned: the pre-run launch is unbounded, so a wedged driver hangs the run forever")
|
||||
}
|
||||
}
|
||||
+103
-33
@@ -1,6 +1,7 @@
|
||||
package verifier
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"slices"
|
||||
@@ -358,49 +359,118 @@ func selectorObjectToString(runtime *goja.Runtime, arg goja.Value) string {
|
||||
return strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
// actionField is one property of the lastAction object, in the order the
|
||||
// object is built. Value is a string, an int, or a nested []actionField for
|
||||
// the from/to points.
|
||||
type actionField struct {
|
||||
key string
|
||||
value any
|
||||
}
|
||||
|
||||
// lastActionFields is the ONE description of the lastAction shape. The goja
|
||||
// host turns it into a JS object (lastActionObject); the web host receives the
|
||||
// same fields as JSON (EncodeLastAction) and installs them as state.lastAction
|
||||
// in the page. Both hosts therefore expose identical field names, casing,
|
||||
// presence and order, so a property reading state.lastAction cannot mean one
|
||||
// thing on native and another on web.
|
||||
func lastActionFields(action *Action) []actionField {
|
||||
point := func(x, y int) []actionField {
|
||||
return []actionField{{key: "x", value: x}, {key: "y", value: y}}
|
||||
}
|
||||
fields := []actionField{{key: "kind", value: string(action.Kind)}}
|
||||
if action.On != "" {
|
||||
fields = append(fields, actionField{key: "on", value: action.On})
|
||||
}
|
||||
if action.Text != "" {
|
||||
fields = append(fields, actionField{key: "text", value: action.Text})
|
||||
}
|
||||
switch action.Kind {
|
||||
case ActionKindSwipe:
|
||||
fields = append(fields,
|
||||
actionField{key: "from", value: point(action.FromX, action.FromY)},
|
||||
actionField{key: "to", value: point(action.ToX, action.ToY)})
|
||||
if action.DurationMillis > 0 {
|
||||
fields = append(fields,
|
||||
actionField{key: "durationMillis", value: action.DurationMillis})
|
||||
}
|
||||
case ActionKindScroll:
|
||||
fields = append(fields,
|
||||
actionField{key: "direction", value: action.Direction},
|
||||
actionField{key: "from", value: point(action.FromX, action.FromY)},
|
||||
actionField{key: "to", value: point(action.ToX, action.ToY)})
|
||||
case ActionKindPressKey:
|
||||
fields = append(fields, actionField{key: "key", value: action.Key})
|
||||
case ActionKindWait:
|
||||
fields = append(fields,
|
||||
actionField{key: "durationMillis", value: action.DurationMillis})
|
||||
}
|
||||
return fields
|
||||
}
|
||||
|
||||
func lastActionObject(runtime *goja.Runtime, action *Action) goja.Value {
|
||||
if action == nil {
|
||||
return goja.Null()
|
||||
}
|
||||
return objectFromFields(runtime, lastActionFields(action))
|
||||
}
|
||||
|
||||
func objectFromFields(runtime *goja.Runtime, fields []actionField) *goja.Object {
|
||||
object := runtime.NewObject()
|
||||
_ = object.Set("kind", string(action.Kind))
|
||||
if action.On != "" {
|
||||
_ = object.Set("on", action.On)
|
||||
for _, field := range fields {
|
||||
if nested, ok := field.value.([]actionField); ok {
|
||||
_ = object.Set(field.key, objectFromFields(runtime, nested))
|
||||
continue
|
||||
}
|
||||
if action.Text != "" {
|
||||
_ = object.Set("text", action.Text)
|
||||
}
|
||||
switch action.Kind {
|
||||
case ActionKindSwipe:
|
||||
from := runtime.NewObject()
|
||||
_ = from.Set("x", action.FromX)
|
||||
_ = from.Set("y", action.FromY)
|
||||
to := runtime.NewObject()
|
||||
_ = to.Set("x", action.ToX)
|
||||
_ = to.Set("y", action.ToY)
|
||||
_ = object.Set("from", from)
|
||||
_ = object.Set("to", to)
|
||||
if action.DurationMillis > 0 {
|
||||
_ = object.Set("durationMillis", action.DurationMillis)
|
||||
}
|
||||
case ActionKindScroll:
|
||||
_ = object.Set("direction", action.Direction)
|
||||
from := runtime.NewObject()
|
||||
_ = from.Set("x", action.FromX)
|
||||
_ = from.Set("y", action.FromY)
|
||||
to := runtime.NewObject()
|
||||
_ = to.Set("x", action.ToX)
|
||||
_ = to.Set("y", action.ToY)
|
||||
_ = object.Set("from", from)
|
||||
_ = object.Set("to", to)
|
||||
case ActionKindPressKey:
|
||||
_ = object.Set("key", action.Key)
|
||||
case ActionKindWait:
|
||||
_ = object.Set("durationMillis", action.DurationMillis)
|
||||
_ = object.Set(field.key, field.value)
|
||||
}
|
||||
return object
|
||||
}
|
||||
|
||||
// EncodeLastAction renders the previous step's action for the web host, which
|
||||
// has no Go-side state object to read: the runner pushes this JSON into the
|
||||
// page before each extractor evaluation. A nil action encodes as JSON null,
|
||||
// the same value the goja host reports on the first step of a run and after a
|
||||
// step whose action was never applied.
|
||||
func EncodeLastAction(action *Action) json.RawMessage {
|
||||
if action == nil {
|
||||
return json.RawMessage("null")
|
||||
}
|
||||
return encodeFields(lastActionFields(action))
|
||||
}
|
||||
|
||||
func encodeFields(fields []actionField) json.RawMessage {
|
||||
var buffer bytes.Buffer
|
||||
buffer.WriteByte('{')
|
||||
for index, field := range fields {
|
||||
if index > 0 {
|
||||
buffer.WriteByte(',')
|
||||
}
|
||||
buffer.Write(encodeJSValue(field.key))
|
||||
buffer.WriteByte(':')
|
||||
if nested, ok := field.value.([]actionField); ok {
|
||||
buffer.Write(encodeFields(nested))
|
||||
continue
|
||||
}
|
||||
buffer.Write(encodeJSValue(field.value))
|
||||
}
|
||||
buffer.WriteByte('}')
|
||||
return buffer.Bytes()
|
||||
}
|
||||
|
||||
// encodeJSValue encodes one value the way JS JSON.stringify would, so the JSON
|
||||
// the web host parses is byte-identical to what the goja object stringifies to.
|
||||
// Go escapes <, > and & by default, which JSON.stringify does not, and that
|
||||
// alone would make the two hosts encode the same selector differently.
|
||||
func encodeJSValue(value any) []byte {
|
||||
var buffer bytes.Buffer
|
||||
encoder := json.NewEncoder(&buffer)
|
||||
encoder.SetEscapeHTML(false)
|
||||
if err := encoder.Encode(value); err != nil {
|
||||
return []byte("null")
|
||||
}
|
||||
return bytes.TrimRight(buffer.Bytes(), "\n")
|
||||
}
|
||||
|
||||
func runtimeMillis(stepTime, runStart time.Time) int64 {
|
||||
if stepTime.IsZero() || runStart.IsZero() {
|
||||
return 0
|
||||
|
||||
@@ -154,3 +154,49 @@ func TestLastActionObject_ExposesKindSpecificFields(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestLastAction_WebJSONMatchesTheGojaObject pins the two hosts to ONE shape.
|
||||
// The goja host builds state.lastAction as a JS object; the web host receives
|
||||
// EncodeLastAction's JSON and installs the parsed value as state.lastAction in
|
||||
// the page. A field this side renames, drops or cases differently would leave a
|
||||
// spec reading state.lastAction working on native and silently mismatching on
|
||||
// web, which is the failure this whole path exists to prevent. Comparing
|
||||
// goja's own JSON.stringify against the encoder is the strongest available
|
||||
// statement that the two are the same object.
|
||||
func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) {
|
||||
verifier := newVerifier(t)
|
||||
mustLoad(t, verifier, `
|
||||
globalThis.last = __sanderling__.extract(state => JSON.stringify(state.lastAction));
|
||||
`)
|
||||
|
||||
for _, testCase := range []struct {
|
||||
name string
|
||||
action *Action
|
||||
}{
|
||||
{"nil", nil},
|
||||
{"Tap", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", X: 12, Y: 34}},
|
||||
{"TapWithoutSelector", &Action{Kind: ActionKindTap, X: 12, Y: 34}},
|
||||
{"DoubleTap", &Action{Kind: ActionKindDoubleTap, On: `desc:say "hi" <b>`}},
|
||||
{"InputText", &Action{Kind: ActionKindInputText, On: "id:field", Text: "50"}},
|
||||
{"Swipe", &Action{Kind: ActionKindSwipe, FromX: 1, FromY: 2, ToX: 3, ToY: 4, DurationMillis: 250}},
|
||||
{"SwipeNoDuration", &Action{Kind: ActionKindSwipe, FromX: 1, FromY: 2, ToX: 3, ToY: 4}},
|
||||
{"Scroll", &Action{Kind: ActionKindScroll, Direction: "down", FromX: 5, FromY: 6, ToX: 5, ToY: 1}},
|
||||
{"PressKey", &Action{Kind: ActionKindPressKey, Key: "enter"}},
|
||||
{"Wait", &Action{Kind: ActionKindWait, DurationMillis: 500}},
|
||||
} {
|
||||
t.Run(testCase.name, func(t *testing.T) {
|
||||
if err := verifier.PushSnapshot(SnapshotInput{
|
||||
Snapshots: Snapshots{},
|
||||
LastAction: testCase.action,
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
handle := verifier.runtime.GlobalObject().Get("last").ToObject(verifier.runtime)
|
||||
goja := handle.Get("current").String()
|
||||
web := string(EncodeLastAction(testCase.action))
|
||||
if goja != web {
|
||||
t.Errorf("the two hosts disagree on state.lastAction\n goja: %s\n web: %s", goja, web)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -1382,3 +1382,54 @@ func TestWithPlatform_IOSReachesPicker(t *testing.T) {
|
||||
t.Errorf("key = %q, want back (native press-key pool)", action.Key)
|
||||
}
|
||||
}
|
||||
|
||||
// TestOverrideExtractorValues_EmptyPayloadIsUndefined pins the cross-host
|
||||
// meaning of a page reading with no value. JSON has no undefined, so the web
|
||||
// runtime wraps every reading in a {value} envelope and the chrome driver hands
|
||||
// an absent value through as an empty payload. It has to land here as undefined,
|
||||
// because that is what PushSnapshot records for a getter that returned undefined
|
||||
// on native: decoding it as null instead would make `x.current === undefined`
|
||||
// answer one thing on the native host and another on web, for one spec.
|
||||
func TestOverrideExtractorValues_EmptyPayloadIsUndefined(t *testing.T) {
|
||||
verifier := newVerifier(t)
|
||||
mustLoad(t, verifier, helloSpec)
|
||||
if err := verifier.PushSnapshot(SnapshotInput{Snapshots: Snapshots{"ledger.balance": json.RawMessage(`42`)}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := verifier.OverrideExtractorValues(map[int]json.RawMessage{
|
||||
0: nil,
|
||||
1: json.RawMessage(`null`),
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for _, want := range []struct {
|
||||
expression string
|
||||
result bool
|
||||
}{
|
||||
{"screen.current === undefined", true},
|
||||
{"screen.current === null", false},
|
||||
{"balance.current === null", true},
|
||||
{"balance.current === undefined", false},
|
||||
} {
|
||||
value, err := verifier.runtime.RunString(want.expression)
|
||||
if err != nil {
|
||||
t.Fatalf("evaluate %s: %v", want.expression, err)
|
||||
}
|
||||
if value.ToBoolean() != want.result {
|
||||
t.Errorf("a spec reading %s gets %v, want %v", want.expression, value, want.result)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestExtractorCount_ReportsEveryRegisteredExtractor keeps the web path's
|
||||
// completeness check honest: it compares the page's reading count against this
|
||||
// number, so a count that ignored an extractor would let a partial table
|
||||
// through.
|
||||
func TestExtractorCount_ReportsEveryRegisteredExtractor(t *testing.T) {
|
||||
verifier := newVerifier(t)
|
||||
mustLoad(t, verifier, helloSpec)
|
||||
if got := verifier.ExtractorCount(); got != 2 {
|
||||
t.Errorf("ExtractorCount() = %d, want 2 (helloSpec registers screen and balance)", got)
|
||||
}
|
||||
}
|
||||
@@ -411,6 +411,15 @@ func (v *Verifier) ChangedExtractors() map[string]ExtractorChange {
|
||||
return changes
|
||||
}
|
||||
|
||||
// ExtractorCount reports how many extractors the spec registered. The web path
|
||||
// compares it against the number of readings the page sent: a page reporting
|
||||
// fewer leaves the rest holding goja's dump-derived value while the others hold
|
||||
// the page's, and a property comparing previous to current across that split
|
||||
// fires on a healthy app.
|
||||
func (v *Verifier) ExtractorCount() int {
|
||||
return len(v.extractors)
|
||||
}
|
||||
|
||||
// OverrideExtractorValues replaces each extractor's `current` slot with a
|
||||
// caller-supplied value, keyed by registration index. Used by the web tick
|
||||
// path so extractor bodies that ran in V8 (against the real DOM) drive the
|
||||
|
||||
+210
-26
@@ -282,13 +282,32 @@ function selectorFromString(selector: string): { css?: string; xpath?: string }
|
||||
return { css: cssPart(kind, value) };
|
||||
}
|
||||
|
||||
// deepQueryAll resolves a CSS selector against a root AND every shadow root
|
||||
// beneath it. querySelectorAll stops dead at a shadow boundary, and a canvas app
|
||||
// (Compose for Web mounts its canvas and its whole accessibility tree inside a
|
||||
// shadow root on the mount element) keeps its entire UI on the far side of one:
|
||||
// without this a spec sees four nodes and can neither enumerate a target nor
|
||||
// resolve a testTag. Light-DOM matches come first, then shadow content in walk
|
||||
// order. XPath has no equivalent, so `text:` selectors stop at the boundary.
|
||||
function deepQueryAll(selector: string, root: ParentNode): Element[] {
|
||||
const found: Element[] = [];
|
||||
const visit = (scope: ParentNode): void => {
|
||||
for (const element of Array.from(scope.querySelectorAll(selector))) found.push(element);
|
||||
for (const element of Array.from(scope.querySelectorAll<HTMLElement>("*"))) {
|
||||
if (element.shadowRoot) visit(element.shadowRoot);
|
||||
}
|
||||
};
|
||||
visit(root);
|
||||
return found;
|
||||
}
|
||||
|
||||
function queryElement(
|
||||
root: ParentNode,
|
||||
selector: unknown,
|
||||
): Element | null {
|
||||
if (typeof selector === "string") {
|
||||
const { css, xpath } = selectorFromString(selector);
|
||||
if (css) return root.querySelector(css);
|
||||
if (css) return deepQueryAll(css, root)[0] ?? null;
|
||||
if (xpath) {
|
||||
const result = document.evaluate(
|
||||
xpath,
|
||||
@@ -313,7 +332,7 @@ function queryElement(
|
||||
}
|
||||
if (selector && typeof selector === "object") {
|
||||
const { css, xpath } = selectorFromObject(selector as Record<string, string | boolean | undefined>);
|
||||
if (css) return root.querySelector(css);
|
||||
if (css) return deepQueryAll(css, root)[0] ?? null;
|
||||
if (xpath) {
|
||||
const result = document.evaluate(
|
||||
xpath,
|
||||
@@ -331,13 +350,27 @@ function queryElement(
|
||||
function queryAllElements(root: ParentNode, selector: unknown): Element[] {
|
||||
if (typeof selector === "string") {
|
||||
const { css, xpath } = selectorFromString(selector);
|
||||
if (css) return Array.from(root.querySelectorAll(css));
|
||||
if (css) return deepQueryAll(css, root);
|
||||
if (xpath) return evaluateXPathAll(xpath, root as Node);
|
||||
return [];
|
||||
}
|
||||
// A selector path: every match of the first segment is searched for the rest,
|
||||
// concatenated in walk order, mirroring FindAllBySelectorPath in
|
||||
// internal/hierarchy. Falling through to the object branch (as this did)
|
||||
// returned NOTHING for a path on web while native returned matches, so a spec
|
||||
// reading state.ax.findAll([{screen}, {row}]) saw an empty list on web and
|
||||
// every property over it passed by having nothing to check.
|
||||
if (Array.isArray(selector)) {
|
||||
const head = selector[0];
|
||||
if (head === undefined) return [];
|
||||
const heads = queryAllElements(root, head);
|
||||
if (selector.length === 1) return heads;
|
||||
const rest = selector.slice(1);
|
||||
return heads.flatMap((element) => queryAllElements(element, rest));
|
||||
}
|
||||
if (selector && typeof selector === "object" && !Array.isArray(selector)) {
|
||||
const { css, xpath } = selectorFromObject(selector as Record<string, string | boolean | undefined>);
|
||||
if (css) return Array.from(root.querySelectorAll(css));
|
||||
if (css) return deepQueryAll(css, root);
|
||||
if (xpath) return evaluateXPathAll(xpath, root as Node);
|
||||
}
|
||||
return [];
|
||||
@@ -391,7 +424,47 @@ function fieldHint(element: Element): string {
|
||||
return element.getAttribute("name") ?? "";
|
||||
}
|
||||
|
||||
function elementHandle(element: Element): Record<string, unknown> {
|
||||
// SELECTOR_TAG is the key an ax element carries the selector it was found by,
|
||||
// the same key the goja host writes (internal/verifier/bindings.go tagSelector).
|
||||
// The shared serializer (runtime-entry.ts pointOf) reads it off an author
|
||||
// target, so a spec's Tap({ on: state.ax.find(...) }) reaches the runner naming
|
||||
// the control it acted on instead of a bare pair of coordinates. Without it
|
||||
// `lastAction.on` is empty on web for exactly the actions a spec authored.
|
||||
const SELECTOR_TAG = "__sanderlingSelector";
|
||||
|
||||
// selectorTag renders a selector argument in the canonical "k:v" grammar the
|
||||
// hierarchy package parses, chains joined by " > ". It mirrors
|
||||
// selectorStringFromJS in internal/verifier/marshal.go, so an element found by
|
||||
// the same selector is labelled with the SAME string on both hosts.
|
||||
function selectorTag(selector: unknown): string {
|
||||
if (typeof selector === "string") return selector;
|
||||
if (Array.isArray(selector)) {
|
||||
return selector
|
||||
.map(selectorTag)
|
||||
.filter((segment) => segment !== "")
|
||||
.join(" > ");
|
||||
}
|
||||
if (selector && typeof selector === "object") {
|
||||
const source = selector as Record<string, unknown>;
|
||||
return Object.keys(source)
|
||||
.filter((key) => key !== SELECTOR_TAG && source[key] !== undefined && source[key] !== null)
|
||||
.map((key) => `${key}:${String(source[key])}`)
|
||||
.join(" ");
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
// isEnabled answers the `enabled` fact. `.disabled` is a property only real form
|
||||
// controls have, so it reads undefined on the role-based controls the tappable
|
||||
// set now covers, and every one of them looked enabled however plainly it was
|
||||
// marked otherwise. internal/driver/chrome/driver.go answers the same two ways
|
||||
// for the dump the goja host reads.
|
||||
function isEnabled(element: Element): boolean {
|
||||
if ((element as HTMLButtonElement).disabled) return false;
|
||||
return element.getAttribute("aria-disabled") !== "true";
|
||||
}
|
||||
|
||||
function elementHandle(element: Element, selector: unknown): Record<string, unknown> {
|
||||
const rect = element.getBoundingClientRect();
|
||||
const x = Math.round(rect.left + rect.width / 2);
|
||||
const y = Math.round(rect.top + rect.height / 2);
|
||||
@@ -416,7 +489,7 @@ function elementHandle(element: Element): Record<string, unknown> {
|
||||
desc: ariaLabel,
|
||||
class: (element as HTMLElement).className ?? "",
|
||||
clickable: true,
|
||||
enabled: !(element as HTMLButtonElement).disabled,
|
||||
enabled: isEnabled(element),
|
||||
editable: isEditableElement(element as HTMLElement),
|
||||
focused: document.activeElement === element,
|
||||
x,
|
||||
@@ -429,12 +502,15 @@ function elementHandle(element: Element): Record<string, unknown> {
|
||||
},
|
||||
attrs,
|
||||
dataset: datasetCopy,
|
||||
find(selector: unknown): unknown {
|
||||
const child = queryElement(element, selector);
|
||||
return child ? elementHandle(child) : undefined;
|
||||
[SELECTOR_TAG]: selectorTag(selector),
|
||||
find(childSelector: unknown): unknown {
|
||||
const child = queryElement(element, childSelector);
|
||||
return child ? elementHandle(child, childSelector) : undefined;
|
||||
},
|
||||
findAll(selector: unknown): unknown[] {
|
||||
return queryAllElements(element, selector).map(elementHandle);
|
||||
findAll(childSelector: unknown): unknown[] {
|
||||
return queryAllElements(element, childSelector).map((child) =>
|
||||
elementHandle(child, childSelector),
|
||||
);
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -443,10 +519,12 @@ function buildAx(): unknown {
|
||||
return {
|
||||
find(selector: unknown): unknown {
|
||||
const element = queryElement(document, selector);
|
||||
return element ? elementHandle(element) : undefined;
|
||||
return element ? elementHandle(element, selector) : undefined;
|
||||
},
|
||||
findAll(selector: unknown): unknown[] {
|
||||
return queryAllElements(document, selector).map(elementHandle);
|
||||
return queryAllElements(document, selector).map((element) =>
|
||||
elementHandle(element, selector),
|
||||
);
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -483,13 +561,22 @@ if (typeof globalThis.addEventListener === "function") {
|
||||
});
|
||||
}
|
||||
|
||||
// lastAction is what the previous step actually did, pushed in by the Go runner
|
||||
// (internal/runner, via __sanderlingSetLastAction__) before each extractor
|
||||
// evaluation, in the shape internal/verifier/marshal.go builds for goja. The
|
||||
// page cannot derive it: only the runner knows whether the action it picked was
|
||||
// really applied, and under --generator llm the action is not picked here at
|
||||
// all. Hardcoding null here, as this file used to, makes every spec property
|
||||
// that reads state.lastAction vacuously true on web.
|
||||
let lastAction: unknown = null;
|
||||
|
||||
function buildState(): unknown {
|
||||
return {
|
||||
snapshots: {},
|
||||
ax: buildAx(),
|
||||
document,
|
||||
window,
|
||||
lastAction: null,
|
||||
lastAction,
|
||||
time: 0,
|
||||
logs: [],
|
||||
exceptions: capturedExceptions.slice(),
|
||||
@@ -535,6 +622,11 @@ const runtime = {
|
||||
// the host invoking the extractor/next-action callbacks.
|
||||
defineLockedGlobal("__sanderling__", runtime);
|
||||
|
||||
// The host calls this once per step, before __sanderlingExtractors__.
|
||||
defineLockedGlobal("__sanderlingSetLastAction__", (value: unknown) => {
|
||||
lastAction = value ?? null;
|
||||
});
|
||||
|
||||
// writable:false stops a page script from shadowing the runtime via plain
|
||||
// assignment (the realistic in-page threat). configurable:true is required so
|
||||
// unit tests sharing one process can reinstall a fake via defineProperty; a
|
||||
@@ -549,9 +641,18 @@ function defineLockedGlobal(name: string, value: unknown): void {
|
||||
});
|
||||
}
|
||||
|
||||
function evaluateExtractors(): Record<number, unknown> {
|
||||
// Each reading is wrapped in a {value} envelope because JSON has no undefined.
|
||||
// Written straight into the map, an extractor whose getter returned undefined
|
||||
// (folio's on(route, tag) off its own screen, which is most extractors on most
|
||||
// steps) had its whole INDEX dropped by JSON.stringify, and the host kept goja's
|
||||
// dump-derived reading for it while the rest held the page's. Inside the
|
||||
// envelope the same drop means "this getter returned undefined", which is what
|
||||
// the goja host records for the same getter; a JSON null would instead claim it
|
||||
// returned null, and `x.current === undefined` would answer differently on the
|
||||
// two hosts.
|
||||
function evaluateExtractors(): Record<number, { value?: unknown }> {
|
||||
const state = buildState();
|
||||
const result: Record<number, unknown> = {};
|
||||
const result: Record<number, { value?: unknown }> = {};
|
||||
for (let i = 0; i < extractors.length; i++) {
|
||||
const entry = extractors[i];
|
||||
if (!entry) continue;
|
||||
@@ -568,7 +669,7 @@ function evaluateExtractors(): Record<number, unknown> {
|
||||
extracting = false;
|
||||
}
|
||||
entry.currentValue = value;
|
||||
result[i] = sanitize(value);
|
||||
result[i] = { value: sanitize(value) };
|
||||
}
|
||||
return result;
|
||||
}
|
||||
@@ -608,7 +709,22 @@ function sanitizeAt(value: unknown, depth: number, seen: WeakSet<object>): unkno
|
||||
// only how the DOM answers "is this clickable" / "is this editable", the two
|
||||
// facts with no direct DOM equivalent of the accessibility attributes native
|
||||
// platforms expose.
|
||||
const TAPPABLE_SELECTOR = 'a, button, input, select, textarea, [role="button"], [onclick]';
|
||||
//
|
||||
// TAPPABLE_ROLES are the ARIA roles whose whole contract is that a user
|
||||
// activates the element. Covering only role="button" left every other one
|
||||
// invisible to the enumeration, however plain the control looked: the replay UI
|
||||
// builds its step rows as <li role="option">, and the spec dogfooding it had to
|
||||
// hand-write an action to reach them because no default verb could see a single
|
||||
// row. internal/driver/chrome/driver.go resolves the same set for the hierarchy
|
||||
// dump the goja host reads, and the two are compared element by element by
|
||||
// TestHierarchy_DerivesTheSameFactsAsTheWebRuntime.
|
||||
const TAPPABLE_ROLES = [
|
||||
"button", "link", "checkbox", "radio", "switch", "tab", "option",
|
||||
"menuitem", "menuitemcheckbox", "menuitemradio", "treeitem",
|
||||
];
|
||||
const TAPPABLE_SELECTOR = `a, button, input, select, textarea, ${
|
||||
TAPPABLE_ROLES.map((role) => `[role="${role}"]`).join(", ")
|
||||
}, [onclick]`;
|
||||
const EDITABLE_SELECTOR = "input, textarea, [contenteditable]";
|
||||
|
||||
const NON_TEXT_INPUT_TYPES = [
|
||||
@@ -651,12 +767,78 @@ function pointOf(element: Element): Candidate {
|
||||
const HEAD_SELECTOR = "head, head *";
|
||||
|
||||
// targetElements is the walk the target list is built from: the document in
|
||||
// pre-order, minus the head subtree.
|
||||
// pre-order, minus the head subtree, with each shadow host's content spliced in
|
||||
// directly after the host. That is buildTree's order in
|
||||
// internal/driver/chrome/driver.go, and the two producers are compared element
|
||||
// by element in enumeration order.
|
||||
function targetElements(): HTMLElement[] {
|
||||
const inHead = new Set<Element>(Array.from(document.querySelectorAll(HEAD_SELECTOR)));
|
||||
return Array.from(document.querySelectorAll<HTMLElement>("*")).filter(
|
||||
const walked: HTMLElement[] = [];
|
||||
expandShadowContent(
|
||||
Array.from(document.querySelectorAll<HTMLElement>("*")).filter(
|
||||
(element) => !inHead.has(element),
|
||||
),
|
||||
walked,
|
||||
);
|
||||
return walked;
|
||||
}
|
||||
|
||||
// expandShadowContent copies a tree-ordered element list into `into`, following
|
||||
// each host into its shadow root (and into nested hosts) as it goes.
|
||||
function expandShadowContent(elements: HTMLElement[], into: HTMLElement[]): void {
|
||||
for (const element of elements) {
|
||||
into.push(element);
|
||||
const shadow = element.shadowRoot;
|
||||
if (!shadow) continue;
|
||||
expandShadowContent(Array.from(shadow.querySelectorAll<HTMLElement>("*")), into);
|
||||
}
|
||||
}
|
||||
|
||||
// IDENTITY_KEYS is the ladder a target's selector is built from, mirroring
|
||||
// selectorForElement in internal/verifier/worker.go: the id first (where
|
||||
// Compose for Web lands a testTag), then data-testid, then the description.
|
||||
// Every key here is one the goja host's selector grammar already understands,
|
||||
// so the runner can re-resolve the target it names. `desc` reads the same three
|
||||
// attributes, in the same order, that the hierarchy dump folds into
|
||||
// content-desc (internal/driver/chrome/driver.go); reading fewer of them would
|
||||
// let a selector this side calls unique resolve to a different element on the
|
||||
// Go side, which re-routes the action to whatever the dump matched first.
|
||||
const IDENTITY_KEYS: ReadonlyArray<readonly [string, (element: HTMLElement) => string]> = [
|
||||
["id", (element) => element.id],
|
||||
["data-testid", (element) => element.dataset.testid ?? ""],
|
||||
[
|
||||
"desc",
|
||||
(element) =>
|
||||
element.getAttribute("aria-label") ||
|
||||
element.getAttribute("alt") ||
|
||||
element.getAttribute("title") ||
|
||||
"",
|
||||
],
|
||||
];
|
||||
|
||||
// selectorsFor names each enumerated element, or leaves it unnamed. A value is
|
||||
// only used when it occurs ONCE across the enumeration, so an action carrying
|
||||
// the selector can never be re-resolved onto a sibling that shares the value
|
||||
// (folio's Home screen has many AccountCards under one testTag). Unnamed
|
||||
// elements keep the coordinates-only behaviour the web host always had.
|
||||
function selectorsFor(elements: readonly HTMLElement[]): Array<string | undefined> {
|
||||
const counts = IDENTITY_KEYS.map(() => new Map<string, number>());
|
||||
for (const element of elements) {
|
||||
IDENTITY_KEYS.forEach(([, read], index) => {
|
||||
const value = read(element);
|
||||
if (!value) return;
|
||||
const seen = counts[index]!;
|
||||
seen.set(value, (seen.get(value) ?? 0) + 1);
|
||||
});
|
||||
}
|
||||
return elements.map((element) => {
|
||||
for (let index = 0; index < IDENTITY_KEYS.length; index++) {
|
||||
const [key, read] = IDENTITY_KEYS[index]!;
|
||||
const value = read(element);
|
||||
if (value && counts[index]!.get(value) === 1) return `${key}:${value}`;
|
||||
}
|
||||
return undefined;
|
||||
});
|
||||
}
|
||||
|
||||
// collectTargets walks the document ONCE and reports every element with the facts
|
||||
@@ -664,16 +846,17 @@ function targetElements(): HTMLElement[] {
|
||||
// resolved by selector first so the DOM's answer to "clickable" and "editable"
|
||||
// stays expressed in CSS, as it always was.
|
||||
function collectTargets(): TargetElement[] {
|
||||
const clickable = new Set<Element>(Array.from(document.querySelectorAll(TAPPABLE_SELECTOR)));
|
||||
const clickable = new Set<Element>(deepQueryAll(TAPPABLE_SELECTOR, document));
|
||||
const editable = new Set<Element>(
|
||||
Array.from(document.querySelectorAll<HTMLElement>(EDITABLE_SELECTOR)).filter(
|
||||
isEditableElement,
|
||||
),
|
||||
(deepQueryAll(EDITABLE_SELECTOR, document) as HTMLElement[]).filter(isEditableElement),
|
||||
);
|
||||
return targetElements().map((element) => ({
|
||||
const elements = targetElements();
|
||||
const selectors = selectorsFor(elements);
|
||||
return elements.map((element, index) => ({
|
||||
...pointOf(element),
|
||||
selector: selectors[index],
|
||||
clickable: clickable.has(element),
|
||||
enabled: !(element as HTMLButtonElement).disabled,
|
||||
enabled: isEnabled(element),
|
||||
editable: editable.has(element),
|
||||
scrollable: isScrollable(element),
|
||||
}));
|
||||
@@ -734,6 +917,7 @@ export const __testing__ = {
|
||||
selectorFromObject,
|
||||
SELECTOR_KEYS,
|
||||
unknownSelectorKeyMessage,
|
||||
selectorTag,
|
||||
xpathStringLiteral,
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import {
|
||||
cardAccountName,
|
||||
cardBalanceText,
|
||||
parseDollarCents,
|
||||
} from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
// Android and iOS expose AccountName/AccountBalance as their own nodes. Web
|
||||
// merges the card into one node whose text is initials + name +
|
||||
// "N transaction(s)" + balance, with no separator between the parts. Both
|
||||
// shapes have to land on the same balance, and the name has to stay usable as
|
||||
// an identity key.
|
||||
|
||||
const balanceOf = (cardText: string) =>
|
||||
parseDollarCents(cardBalanceText({ childText: undefined, cardText }));
|
||||
|
||||
// Renders a card the way HomeScreen.kt does, so a test states the account and
|
||||
// lets the fixture do the concatenating.
|
||||
const card = (initials: string, name: string, count: number, balance: string) =>
|
||||
`${initials}${name}${count === 1 ? "1 transaction" : `${count} transactions`}${balance}`;
|
||||
|
||||
test("structured child wins over the card text", () => {
|
||||
assert.equal(
|
||||
cardBalanceText({ childText: "$118.00", cardText: "SASavings1 transaction$118.00" }),
|
||||
"$118.00",
|
||||
);
|
||||
assert.equal(
|
||||
cardAccountName({ childText: "Savings", cardText: "SASavings1 transaction$118.00" }),
|
||||
"Savings",
|
||||
);
|
||||
});
|
||||
|
||||
test("merged card text: balance is the amount at the end, not scraped digits", () => {
|
||||
// The naive reading, text.replace(/[^0-9]/g, ""), absorbs the 12 of
|
||||
// "12 transactions" and returns 12258900.
|
||||
assert.equal(balanceOf("INInvestments12 transactions$2,589.00"), 258900);
|
||||
});
|
||||
|
||||
test("merged card text: singular transaction label", () => {
|
||||
assert.equal(balanceOf("SASavings1 transaction$118.00"), 11800);
|
||||
});
|
||||
|
||||
test("merged card text: negative balance keeps its sign", () => {
|
||||
assert.equal(cardBalanceText({ childText: undefined, cardText: "TRTravel3 transactions-$45.50" }), "-$45.50");
|
||||
assert.equal(balanceOf("TRTravel3 transactions-$45.50"), -4550);
|
||||
});
|
||||
|
||||
test("structured child: negative balance keeps its sign", () => {
|
||||
assert.equal(parseDollarCents(cardBalanceText({ childText: "-$45.50", cardText: undefined })), -4550);
|
||||
});
|
||||
|
||||
test("zero-balance card is 0, not unknown", () => {
|
||||
assert.equal(balanceOf("EFEmergency Fund0 transactions$0.00"), 0);
|
||||
assert.equal(balanceOf("Aa0 transactions$0.00"), 0);
|
||||
});
|
||||
|
||||
// The trap a lazy balance regex falls into: a name ending in digits runs
|
||||
// straight into the transaction count, so only anchoring the amount at the end
|
||||
// of the string gets it right.
|
||||
test("name ending in digits does not leak into the balance", () => {
|
||||
assert.equal(balanceOf(card("T2", "Travel 2024", 12, "$75.00")), 7500);
|
||||
assert.equal(balanceOf(card("T2", "Travel 2024", 0, "$0.00")), 0);
|
||||
assert.equal(balanceOf(card("20", "2024", 3, "-$1,234.56")), -123456);
|
||||
});
|
||||
|
||||
// The account key only has to be stable and per-account. newAccountBalanceIsZero
|
||||
// reads it as a set member: a key that drifted as an account's transaction
|
||||
// count grew would make an existing account look brand new, and the property
|
||||
// would fire on it for holding the balance it just earned.
|
||||
test("account key is stable as the transaction count and balance move", () => {
|
||||
const cards: [string, string][] = [
|
||||
["CH", "Checking"],
|
||||
["T2", "Travel 2024"],
|
||||
["A2", "Account 2"],
|
||||
["Aa", "a"],
|
||||
["?", ""],
|
||||
["5T", "5 transactions"],
|
||||
["X9", "x9"],
|
||||
];
|
||||
for (const [initials, name] of cards) {
|
||||
const keys = new Set<string>();
|
||||
for (let count = 0; count <= 130; count++) {
|
||||
keys.add(cardAccountName({
|
||||
childText: undefined,
|
||||
cardText: card(initials, name, count, `$${count * 7}.50`),
|
||||
}));
|
||||
}
|
||||
assert.equal(keys.size, 1, `key for ${JSON.stringify(name)} drifted: ${[...keys].join(", ")}`);
|
||||
}
|
||||
});
|
||||
|
||||
test("account keys are distinct across the accounts a run creates", () => {
|
||||
const names: [string, string][] = [
|
||||
["CH", "Checking"],
|
||||
["SA", "Savings"],
|
||||
["TR", "Travel"],
|
||||
["EF", "Emergency Fund"],
|
||||
["IN", "Investments"],
|
||||
["T2", "Travel 2024"],
|
||||
["A2", "Account 2"],
|
||||
["A1", "Account 12"],
|
||||
["Aa", "a"],
|
||||
];
|
||||
const keys = names.map(([initials, name]) =>
|
||||
cardAccountName({ childText: undefined, cardText: card(initials, name, 4, "$9.00") }));
|
||||
assert.equal(new Set(keys).size, names.length);
|
||||
});
|
||||
|
||||
// elementHandle in the web runtime truncates node text at 200 characters, so a
|
||||
// long account name (the input corpus types 4096 "a"s) pushes the balance off
|
||||
// the end of the string. That balance is unknown, and unknown must not read as
|
||||
// zero: newAccountBalanceIsZero passes an unknown balance rather than
|
||||
// convicting a card it could not read.
|
||||
test("card text truncated past the balance reads as unknown, not zero", () => {
|
||||
const cardText = "AA" + "a".repeat(198);
|
||||
assert.equal(cardBalanceText({ childText: undefined, cardText }), undefined);
|
||||
assert.equal(balanceOf(cardText), null);
|
||||
});
|
||||
|
||||
test("empty and missing text are unknown, not zero", () => {
|
||||
assert.equal(parseDollarCents(undefined), null);
|
||||
assert.equal(parseDollarCents(""), null);
|
||||
assert.equal(parseDollarCents("no digits here"), null);
|
||||
assert.equal(parseDollarCents("$12"), null);
|
||||
assert.equal(cardBalanceText({ childText: undefined, cardText: undefined }), undefined);
|
||||
assert.equal(cardAccountName({ childText: undefined, cardText: undefined }), "");
|
||||
});
|
||||
|
||||
// The two accessibility shapes have to read the same per-card balance, which is
|
||||
// what the accounts extractor compares. The Home total is no longer a sum of
|
||||
// these: it is the app's own TOTAL BALANCE node (see folio-total-balance.test.ts).
|
||||
test("merged card text and a structured child give the same balance", () => {
|
||||
const merged = ["INInvestments12 transactions$2,589.00", "Aa0 transactions$0.00"].map(cardText =>
|
||||
cardBalanceText({ childText: undefined, cardText }));
|
||||
assert.deepEqual(merged, ["$2,589.00", "$0.00"]);
|
||||
assert.deepEqual(merged.map(parseDollarCents), [258900, 0]);
|
||||
});
|
||||
@@ -0,0 +1,159 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import {
|
||||
committedTransactionsExceedSubmits,
|
||||
countSubmitsInWindow,
|
||||
homeAccountsOf,
|
||||
homeTxnCountsOf,
|
||||
readHomeCards,
|
||||
} from "../../../examples/folio/sanderling/predicates.ts";
|
||||
import type {
|
||||
CardReading,
|
||||
HomeCardReading,
|
||||
TxnCount,
|
||||
} from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
const card = (name: string, balance: number | null, count: TxnCount | undefined) => ({
|
||||
name,
|
||||
balance,
|
||||
count,
|
||||
});
|
||||
|
||||
test("a laid-out card list reads as an account list and a count map", () => {
|
||||
const cards = [card("Checking", 0, "0"), card("Travel", 2411200, "1")];
|
||||
assert.deepEqual(homeAccountsOf(cards), [
|
||||
{ name: "Checking", balance: 0 },
|
||||
{ name: "Travel", balance: 2411200 },
|
||||
]);
|
||||
assert.deepEqual(homeTxnCountsOf(cards), { Checking: "0", Travel: "1" });
|
||||
});
|
||||
|
||||
// Android draws Home's own node a frame or two before its list, so `findAll`
|
||||
// over the cards comes back empty while the screen already claims to be Home.
|
||||
// "No cards on screen" is not "no accounts".
|
||||
test("Home with nothing laid out yet is unknown, not empty", () => {
|
||||
assert.equal(homeAccountsOf([]), null);
|
||||
assert.equal(homeTxnCountsOf([]), null);
|
||||
});
|
||||
|
||||
test("a card with no readable name or count is left out of the map", () => {
|
||||
assert.deepEqual(
|
||||
homeTxnCountsOf([card("", 0, "3"), card("Checking", 0, undefined), card("Savings", 0, "2")]),
|
||||
{ Savings: "2" },
|
||||
);
|
||||
});
|
||||
|
||||
test("every card unreadable leaves nothing to compare, which is unknown", () => {
|
||||
assert.equal(homeTxnCountsOf([card("", 0, "3"), card("Checking", 0, undefined)]), null);
|
||||
});
|
||||
|
||||
test("off Home the carrier is reported unchanged", () => {
|
||||
const carried = { Checking: "3" };
|
||||
assert.deepEqual(readHomeCards({ route: "ledger", reading: null, previousCarrier: carried }), {
|
||||
value: carried,
|
||||
carrier: carried,
|
||||
fresh: false,
|
||||
});
|
||||
});
|
||||
|
||||
test("a readable list replaces the carrier and closes the window", () => {
|
||||
assert.deepEqual(
|
||||
readHomeCards({ route: "home", reading: { Checking: "5" }, previousCarrier: { Checking: "3" } }),
|
||||
{ value: { Checking: "5" }, carrier: { Checking: "5" }, fresh: true },
|
||||
);
|
||||
});
|
||||
|
||||
// The poisoned carrier, the same defect readHomeTotalBalance was fixed for and
|
||||
// this reading was not. An empty reading written into the carrier is handed
|
||||
// straight back on every later off-Home step, so one un-laid-out Home turns the
|
||||
// comparison into {} against {} for the rest of the run. Measured on android:
|
||||
// counts_prev was {} at EVERY evaluation point of all 17 runs, which is a
|
||||
// counting invariant that cannot fire at all.
|
||||
test("an un-laid-out Home reports unknown but leaves the carrier intact", () => {
|
||||
const carried = { Checking: "3", Savings: "1" };
|
||||
assert.deepEqual(readHomeCards({ route: "home", reading: null, previousCarrier: carried }), {
|
||||
value: null,
|
||||
carrier: carried,
|
||||
fresh: false,
|
||||
});
|
||||
});
|
||||
|
||||
// The trace the fix has to survive, stepped through the carrier and the window
|
||||
// the spec holds. A double-submit commits two rows against one action, the
|
||||
// Home it lands on has not drawn its list yet, and the counting invariant must
|
||||
// still be able to see the pair once a real Home comes back.
|
||||
function run(steps: { route: string | null; cards: CardReading[]; lastAction: unknown }[]) {
|
||||
let carrier: Record<string, TxnCount> | null = null;
|
||||
let submits = 0;
|
||||
const out: { counts: Record<string, TxnCount> | null; submits: number }[] = [];
|
||||
for (const step of steps) {
|
||||
const reading: HomeCardReading<Record<string, TxnCount>> = readHomeCards({
|
||||
route: step.route,
|
||||
reading: homeTxnCountsOf(step.cards),
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
carrier = reading.carrier;
|
||||
const window = countSubmitsInWindow({
|
||||
previousCount: submits,
|
||||
lastAction: step.lastAction as { kind?: string; on?: string } | null,
|
||||
fresh: reading.fresh,
|
||||
});
|
||||
submits = window.next;
|
||||
out.push({ counts: reading.value, submits: window.reported });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const idle = { kind: "Tap", on: "testTag:AccountCard" };
|
||||
const doubleSubmit = { kind: "DoubleTap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" };
|
||||
|
||||
test("an un-laid-out Home no longer kills the counting invariant", () => {
|
||||
const trace = run([
|
||||
{ route: "home", cards: [card("Checking", 0, "3")], lastAction: null },
|
||||
{ route: "ledger", cards: [], lastAction: idle },
|
||||
{ route: "home", cards: [], lastAction: doubleSubmit },
|
||||
{ route: "ledger", cards: [], lastAction: idle },
|
||||
{ route: "home", cards: [card("Checking", 0, "5")], lastAction: idle },
|
||||
]);
|
||||
|
||||
// The un-laid-out Home is unknown for its own step, and the two steps after
|
||||
// it get the last list anyone actually read rather than an empty one.
|
||||
assert.deepEqual(trace[2]?.counts, null);
|
||||
assert.deepEqual(trace[3]?.counts, { Checking: "3" });
|
||||
assert.deepEqual(trace[4]?.counts, { Checking: "5" });
|
||||
|
||||
// The window it did not close still holds the double-submit, so one action
|
||||
// against two committed rows is visible at the step that can compare them.
|
||||
assert.equal(trace[4]?.submits, 1);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: trace[3]?.counts ?? null,
|
||||
countsAfter: trace[4]?.counts ?? null,
|
||||
submitsInWindow: trace[4]?.submits ?? 0,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The other half of the pairing: the counts window has to close on the counts
|
||||
// reading, not on the total's. A Home frame can render its footer total while
|
||||
// its list is still empty, and a window that reset there would compare a pair of
|
||||
// readings spanning submits it had already forgotten.
|
||||
test("an un-laid-out Home does not close the counting window", () => {
|
||||
const trace = run([
|
||||
{ route: "home", cards: [card("Checking", 0, "3")], lastAction: null },
|
||||
{ route: "ledger", cards: [], lastAction: doubleSubmit },
|
||||
{ route: "home", cards: [], lastAction: doubleSubmit },
|
||||
{ route: "home", cards: [card("Checking", 0, "7")], lastAction: idle },
|
||||
]);
|
||||
assert.equal(trace[3]?.submits, 2);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "3" },
|
||||
countsAfter: trace[3]?.counts ?? null,
|
||||
submitsInWindow: trace[3]?.submits ?? 0,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,235 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import { createdAccountHasNonZeroBalance } from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
const created = { kind: "Tap", on: "testTag:AddAccountScreen > testTag:AddAccountSubmit" };
|
||||
const idle = { kind: "Tap", on: "testTag:HomeScreen > testTag:AccountCard" };
|
||||
|
||||
const account = (name: string, balance: number | null) => ({ name, balance });
|
||||
|
||||
// The property still has teeth: the account the fuzzer just asked for, holding
|
||||
// money on the step its creation landed on Home, is a real violation.
|
||||
test("an account created holding money is a violation", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000)],
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("a double-tapped create is judged the same way", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: { kind: "DoubleTap", on: "id:AddAccountSubmit" },
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000)],
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("an account created empty is what the app is supposed to do", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 0)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("a balance that could not be read is not evidence", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", null)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The false positives this replaces. Home lists the accounts that fit the
|
||||
// viewport, so an account arrives in a later reading for reasons that have
|
||||
// nothing to do with being created: the list scrolled, a clipped card finished
|
||||
// laying out, or (before the route fix) the earlier reading came off a
|
||||
// half-rendered Home mid-transition. Measured on android: an existing Travel
|
||||
// holding $24,112.00 and an existing Savings holding $429,585.00, convicted for
|
||||
// coming into view.
|
||||
test("an account scrolling into view is not an account being created", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: idle,
|
||||
typedName: "Travel",
|
||||
before: [account("Emergency Fund", 461012300), account("Checking", 0)],
|
||||
after: [
|
||||
account("Emergency Fund", 461012300),
|
||||
account("Checking", 0),
|
||||
account("Travel", 2411200),
|
||||
account("Savings", 0),
|
||||
],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("nor is one that appears with no action at all behind it", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: null,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 2411200)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// Even on the creation step, the only card judged is the one that answers to
|
||||
// the name the fuzzer typed. A card that came into view alongside it is still
|
||||
// just a card that came into view.
|
||||
test("a funded account arriving beside the created one is not judged", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Savings",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Savings", 0), account("Travel", 2411200)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("off Home there is no reading to judge", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "ledger",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("a transition frame names no route, so nothing is judged there either", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: null,
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("an unknown reading on either side is not evidence", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: null,
|
||||
after: [account("Travel", 5000)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: null,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// defaultActions types edge-case text into the name field, and an empty name is
|
||||
// not a name we can find a card by.
|
||||
test("an empty typed name attributes nothing", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: " ",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: undefined,
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// Web merges the card into one node whose text opens with the avatar initials,
|
||||
// so the identity key carries them: "TRTravel" is the card for "Travel".
|
||||
test("the merged web key still matches the name that was typed", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("CHChecking", 0)],
|
||||
after: [account("CHChecking", 0), account("TRTravel", 5000)],
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// Two cards answering to one typed name leave the appearance unattributable:
|
||||
// the fuzzer creates duplicates from a five-name list, and the tree has been
|
||||
// seen exposing the same card twice on a transition frame.
|
||||
test("two cards matching the typed name are not attributable to the creation", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0)],
|
||||
after: [account("Checking", 0), account("Travel", 5000), account("MyTravel", 900)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("a card that was already there is not a card that was just created", () => {
|
||||
assert.equal(
|
||||
createdAccountHasNonZeroBalance({
|
||||
route: "home",
|
||||
lastAction: created,
|
||||
typedName: "Travel",
|
||||
before: [account("Checking", 0), account("Travel", 2411200)],
|
||||
after: [account("Checking", 0), account("Travel", 2411200)],
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,64 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import { oncePerFrame, routeOfFrame } from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
// oncePerFrame is what stops the folio spec re-walking the accessibility tree
|
||||
// once per extractor, and the whole of its safety is that the state object is a
|
||||
// fresh one every step (goja's stateObject, web's buildState). These tests pin
|
||||
// both halves: the same frame is read once, a different frame is a different
|
||||
// answer. A cache that outlived its frame would freeze every reading the spec
|
||||
// takes and the properties over them would go quietly vacuous.
|
||||
const SCREENS = { login: "LoginScreen", home: "HomeScreen" } as const;
|
||||
|
||||
interface Frame {
|
||||
present: readonly string[];
|
||||
finds: number;
|
||||
}
|
||||
|
||||
const frameShowing = (...present: readonly string[]): Frame => ({ present, finds: 0 });
|
||||
|
||||
const routeOf = oncePerFrame((frame: Frame) =>
|
||||
routeOfFrame(SCREENS, tag => {
|
||||
frame.finds++;
|
||||
return frame.present.includes(tag);
|
||||
}),
|
||||
);
|
||||
|
||||
test("one frame is walked once, however many readings ask", () => {
|
||||
const home = frameShowing("HomeScreen");
|
||||
assert.equal(routeOf(home), "home");
|
||||
assert.equal(routeOf(home), "home");
|
||||
assert.equal(routeOf(home), "home");
|
||||
assert.equal(home.finds, 2);
|
||||
});
|
||||
|
||||
test("a new frame is a new answer", () => {
|
||||
const home = frameShowing("HomeScreen");
|
||||
const login = frameShowing("LoginScreen");
|
||||
assert.equal(routeOf(home), "home");
|
||||
assert.equal(routeOf(login), "login");
|
||||
assert.equal(login.finds, 2);
|
||||
});
|
||||
|
||||
test("a transition frame is not answered off the frame before it", () => {
|
||||
assert.equal(routeOf(frameShowing("HomeScreen")), "home");
|
||||
assert.equal(routeOf(frameShowing("HomeScreen", "LoginScreen")), null);
|
||||
});
|
||||
|
||||
test("returning to an earlier frame re-reads it", () => {
|
||||
const home = frameShowing("HomeScreen");
|
||||
routeOf(home);
|
||||
routeOf(frameShowing("LoginScreen"));
|
||||
assert.equal(routeOf(home), "home");
|
||||
assert.equal(home.finds, 4);
|
||||
});
|
||||
|
||||
// What makes memoizing the card list worth more than memoizing the route: the
|
||||
// three readings taken off it share one parse instead of three.
|
||||
test("a frame's reading is handed back by identity", () => {
|
||||
const cardsOf = oncePerFrame((frame: Frame) => frame.present.map(tag => ({ tag })));
|
||||
const home = frameShowing("HomeScreen");
|
||||
assert.equal(cardsOf(home), cardsOf(home));
|
||||
assert.notEqual(cardsOf(home), cardsOf(frameShowing("HomeScreen")));
|
||||
});
|
||||
@@ -13,6 +13,7 @@ test("single submit: delta matches typed amount", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 1500,
|
||||
@@ -26,6 +27,7 @@ test("double submit: delta is twice the typed amount, fires", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 2000,
|
||||
@@ -39,6 +41,7 @@ test("DoubleTap kind also caught when delta exceeds typed amount", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "DoubleTap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 1000,
|
||||
@@ -52,6 +55,7 @@ test("wrong action kind: vacuous true even with mismatch", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "InputText", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 1000,
|
||||
@@ -65,6 +69,7 @@ test("wrong target: vacuous true even with mismatch", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: "testTag:LoginScreen > testTag:LoginSubmit" },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 1000,
|
||||
@@ -78,6 +83,7 @@ test("null lastAction: vacuous true", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: null,
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 2000,
|
||||
@@ -91,6 +97,7 @@ test("zero typedAmount: vacuous true", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 0,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 1500,
|
||||
@@ -104,6 +111,7 @@ test("selector as object: coerced safely and TxnSubmit detected", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: { testTag: "TxnSubmit" } },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 1000,
|
||||
@@ -117,6 +125,7 @@ test("selector as object without TxnSubmit: vacuous true", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: { testTag: "LoginSubmit" } },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 1000,
|
||||
@@ -130,6 +139,7 @@ test("raw whole-dollar input: single submit clears", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: parseTypedAmount("50"),
|
||||
prevTotalBalance: 5000,
|
||||
currTotalBalance: 10000,
|
||||
@@ -143,6 +153,7 @@ test("raw whole-dollar input: double submit fires", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: parseTypedAmount("50"),
|
||||
prevTotalBalance: 5000,
|
||||
currTotalBalance: 15000,
|
||||
@@ -156,6 +167,7 @@ test("decimal input from empty prior balance clears", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: parseTypedAmount("5.50"),
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 550,
|
||||
@@ -169,6 +181,7 @@ test("DoubleTap kind with raw whole-dollar input fires", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "DoubleTap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: parseTypedAmount("100"),
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 20000,
|
||||
@@ -182,6 +195,7 @@ test("route gate: ledger landing with stale carrier is skipped", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "ledger",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 5000,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 0,
|
||||
@@ -195,6 +209,7 @@ test("route gate: add-transaction landing with double-submit delta is skipped",
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "add-transaction",
|
||||
lastAction: { kind: "DoubleTap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 5000,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 10000,
|
||||
@@ -208,6 +223,7 @@ test("route gate: null route is skipped", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: null,
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 5000,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 0,
|
||||
@@ -221,6 +237,7 @@ test("route gate: home landing with matching delta passes", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 5000,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 5000,
|
||||
@@ -234,6 +251,7 @@ test("route gate: home landing with double-insert delta fires", () => {
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 5000,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 10000,
|
||||
@@ -241,3 +259,246 @@ test("route gate: home landing with double-insert delta fires", () => {
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// Precision. Cents are integers in float64 here, so the equality only means
|
||||
// something while every number involved is exactly representable. The app takes
|
||||
// any amount that fits a Kotlin Long, and an iOS run reached a balance around
|
||||
// 1e18 cents, where representable values sit 128 apart: the delta of a
|
||||
// perfectly healthy single submit no longer reads back as the typed amount.
|
||||
const HUGE_BALANCE = 999999999999999900;
|
||||
|
||||
test("above 2^53 the arithmetic itself is wrong, which is why the guard exists", () => {
|
||||
assert.notEqual(Math.abs(HUGE_BALANCE + 1600 - HUGE_BALANCE), 1600);
|
||||
});
|
||||
|
||||
test("above 2^53 a healthy single submit is not reported", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 1600,
|
||||
prevTotalBalance: HUGE_BALANCE,
|
||||
currTotalBalance: HUGE_BALANCE + 1600,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("above 2^53 a double-submit delta is not reported either", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 1600,
|
||||
prevTotalBalance: HUGE_BALANCE,
|
||||
currTotalBalance: HUGE_BALANCE + 3200,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("an unreadable previous balance above 2^53 is not evidence", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 1600,
|
||||
prevTotalBalance: HUGE_BALANCE,
|
||||
currTotalBalance: 5000,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// A typed amount past the safe range cannot be compared either. parseTypedAmount
|
||||
// returns 0 for those now, but the predicate takes the number from its caller
|
||||
// and must not convict on one it cannot hold.
|
||||
test("typed amount above 2^53 is not evidence", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 1e23,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 0,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The boundary, from both sides. MAX_SAFE_INTEGER still gets judged; one cent
|
||||
// more is where counting stops being exact.
|
||||
test("boundary: a double submit landing exactly on MAX_SAFE_INTEGER still fires", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 4503599627370495,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 9007199254740990,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("boundary: a single submit landing exactly on MAX_SAFE_INTEGER passes", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 9007199254740991,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 9007199254740991,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("boundary: one cent past MAX_SAFE_INTEGER stops being evidence", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 4503599627370496,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 9007199254740992,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The guard covers the balances and the typed amount, not their difference: two
|
||||
// safe balances subtract exactly whenever the result could have matched a safe
|
||||
// typed amount, so a mismatch here is real and must still be reported.
|
||||
test("a large but exact difference between safe balances still fires", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: -9007199254740991,
|
||||
currTotalBalance: 9007199254740991,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The 21-digit corpus amount end to end: the app refuses it, so nothing moves,
|
||||
// and the property must stay quiet rather than demand a 1e23-cent move.
|
||||
test("21-digit typed amount with an unmoved balance is not a violation", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: parseTypedAmount("999999999999999999999"),
|
||||
prevTotalBalance: 220900,
|
||||
currTotalBalance: 220900,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// Freshness. prevTotalBalance is the last total we READ, so the window between
|
||||
// it and now can hold more than one submit's transactions. A delta measured
|
||||
// over such a window is not evidence about the amount typed into any one of
|
||||
// them, and the android run that produced a 13000 delta against a typed 19600
|
||||
// is what that looks like: the window held a double-submit's two 19600 debits
|
||||
// and an unrelated 26200 credit.
|
||||
test("freshness: two submits in the window is vacuous, not a conviction", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "DoubleTap", on: submitOn },
|
||||
submitsInWindow: 2,
|
||||
typedAmount: 19600,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: -13000,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("freshness: two submits cannot convict even on a clean 2x delta", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 2,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 2000,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The boundary of the rule, from both sides. One submit is the only window the
|
||||
// property judges: zero means the total moved without a submit landing in it
|
||||
// (nothing to attribute the move to), and two or more means the move is shared.
|
||||
test("freshness boundary: exactly one submit is the window that convicts", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "DoubleTap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 19600,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: -39200,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("freshness boundary: one submit with a healthy 1x delta still passes", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 19600,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: -19600,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("freshness boundary: three submits is vacuous", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 3,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 0,
|
||||
currTotalBalance: 2500,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// A zero count would mean the step's own action was not counted as a submit,
|
||||
// which contradicts the action gate above it. Guard it anyway: a window with no
|
||||
// submit in it explains no balance move.
|
||||
test("freshness boundary: a window with no submit in it is vacuous", () => {
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: { kind: "Tap", on: submitOn },
|
||||
submitsInWindow: 0,
|
||||
typedAmount: 500,
|
||||
prevTotalBalance: 1000,
|
||||
currTotalBalance: 2000,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,151 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import {
|
||||
countSubmitsInWindow,
|
||||
isTxnSubmitTap,
|
||||
readHomeTotalBalance,
|
||||
} from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
const submitOn = "testTag:AddTransactionScreen > testTag:TxnSubmit";
|
||||
|
||||
test("a tap on TxnSubmit is a commit", () => {
|
||||
assert.equal(isTxnSubmitTap({ kind: "Tap", on: submitOn }), true);
|
||||
});
|
||||
|
||||
test("a double-tap on TxnSubmit is ONE commit action, not two", () => {
|
||||
const window = countSubmitsInWindow({
|
||||
previousCount: 0,
|
||||
lastAction: { kind: "DoubleTap", on: submitOn },
|
||||
fresh: true,
|
||||
});
|
||||
assert.equal(window.reported, 1);
|
||||
});
|
||||
|
||||
test("a selector object naming TxnSubmit is a commit", () => {
|
||||
assert.equal(isTxnSubmitTap({ kind: "Tap", on: { testTag: "TxnSubmit" } }), true);
|
||||
});
|
||||
|
||||
test("typing into the amount field is not a commit", () => {
|
||||
assert.equal(isTxnSubmitTap({ kind: "InputText", on: submitOn }), false);
|
||||
});
|
||||
|
||||
test("tapping some other button is not a commit", () => {
|
||||
assert.equal(isTxnSubmitTap({ kind: "Tap", on: "testTag:AddAccountSubmit" }), false);
|
||||
});
|
||||
|
||||
test("no action at all is not a commit", () => {
|
||||
assert.equal(isTxnSubmitTap(null), false);
|
||||
});
|
||||
|
||||
test("a fresh Home reading closes the window and the next one starts empty", () => {
|
||||
assert.deepEqual(
|
||||
countSubmitsInWindow({ previousCount: 0, lastAction: { kind: "Tap", on: submitOn }, fresh: true }),
|
||||
{ reported: 1, next: 0 },
|
||||
);
|
||||
});
|
||||
|
||||
test("landing off Home keeps the submit in the window for the next step", () => {
|
||||
assert.deepEqual(
|
||||
countSubmitsInWindow({ previousCount: 0, lastAction: { kind: "Tap", on: submitOn }, fresh: false }),
|
||||
{ reported: 1, next: 1 },
|
||||
);
|
||||
});
|
||||
|
||||
test("a non-submit step neither adds to nor forgets the window", () => {
|
||||
assert.deepEqual(
|
||||
countSubmitsInWindow({ previousCount: 1, lastAction: { kind: "Tap", on: "testTag:AccountCard" }, fresh: false }),
|
||||
{ reported: 1, next: 1 },
|
||||
);
|
||||
});
|
||||
|
||||
test("a second submit with no Home reading between them counts two", () => {
|
||||
assert.deepEqual(
|
||||
countSubmitsInWindow({ previousCount: 1, lastAction: { kind: "DoubleTap", on: submitOn }, fresh: true }),
|
||||
{ reported: 2, next: 0 },
|
||||
);
|
||||
});
|
||||
|
||||
// The two traces the freshness rule exists to tell apart, driven step by step
|
||||
// through the same pair of carriers the spec holds.
|
||||
function run(steps: { route: string | null; totalText?: string; lastAction: unknown }[]) {
|
||||
let carrier: number | null = null;
|
||||
let submits = 0;
|
||||
const out: { total: number | null; submits: number }[] = [];
|
||||
for (const step of steps) {
|
||||
const reading = readHomeTotalBalance({
|
||||
route: step.route,
|
||||
totalText: step.totalText,
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
carrier = reading.carrier;
|
||||
const window = countSubmitsInWindow({
|
||||
previousCount: submits,
|
||||
lastAction: step.lastAction as { kind?: string; on?: string } | null,
|
||||
fresh: reading.fresh,
|
||||
});
|
||||
submits = window.next;
|
||||
out.push({ total: reading.value, submits: window.reported });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const idle = { kind: "Tap", on: "testTag:AccountCard" };
|
||||
const submit = { kind: "Tap", on: submitOn };
|
||||
const doubleSubmit = { kind: "DoubleTap", on: submitOn };
|
||||
|
||||
// A double-submit pops the back stack twice (each Submit calls
|
||||
// navigator.back), so unlike a healthy single submit it lands back on Home,
|
||||
// which is why the property can see it at all.
|
||||
test("clean double submit: one action in the window, delta is 2x", () => {
|
||||
const trace = run([
|
||||
{ route: "home", totalText: "$0.00", lastAction: null },
|
||||
{ route: "ledger", lastAction: idle },
|
||||
{ route: "ledger", lastAction: idle },
|
||||
{ route: "home", totalText: "$100.00", lastAction: doubleSubmit },
|
||||
]);
|
||||
assert.equal(trace[3]?.submits, 1);
|
||||
assert.equal(trace[0]?.total, 0);
|
||||
assert.equal(trace[3]?.total, 10000);
|
||||
});
|
||||
|
||||
// The contaminated window from the android run: an unrelated submit committed
|
||||
// while we were off Home, then the double-submit landed. The delta spans three
|
||||
// transactions, so it is not evidence about either typed amount.
|
||||
test("two submits between Home visits: the window is not evidence", () => {
|
||||
const trace = run([
|
||||
{ route: "home", totalText: "$0.00", lastAction: null },
|
||||
{ route: "ledger", lastAction: idle },
|
||||
{ route: "ledger", lastAction: submit },
|
||||
{ route: "ledger", lastAction: idle },
|
||||
{ route: "home", totalText: "-$130.00", lastAction: doubleSubmit },
|
||||
]);
|
||||
assert.equal(trace[4]?.submits, 2);
|
||||
});
|
||||
|
||||
// Freshness is restored by seeing Home, not by time passing.
|
||||
test("a Home visit between two submits restores a one-action window", () => {
|
||||
const trace = run([
|
||||
{ route: "home", totalText: "$0.00", lastAction: null },
|
||||
{ route: "ledger", lastAction: submit },
|
||||
{ route: "home", totalText: "$262.00", lastAction: idle },
|
||||
{ route: "ledger", lastAction: idle },
|
||||
{ route: "home", totalText: "$66.00", lastAction: doubleSubmit },
|
||||
]);
|
||||
assert.equal(trace[1]?.submits, 1);
|
||||
assert.equal(trace[2]?.submits, 1);
|
||||
assert.equal(trace[4]?.submits, 1);
|
||||
});
|
||||
|
||||
// An unreadable Home is not a Home reading: it must not close the window, or
|
||||
// the count would go back to zero against a total nobody read.
|
||||
test("an unreadable Home does not close the window", () => {
|
||||
const trace = run([
|
||||
{ route: "home", totalText: "$0.00", lastAction: null },
|
||||
{ route: "ledger", lastAction: submit },
|
||||
{ route: "home", totalText: undefined, lastAction: idle },
|
||||
{ route: "home", totalText: "$66.00", lastAction: doubleSubmit },
|
||||
]);
|
||||
assert.equal(trace[2]?.total, null);
|
||||
assert.equal(trace[3]?.submits, 2);
|
||||
});
|
||||
@@ -1,97 +1,88 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import { computeHomeTotalBalance } from "../../../examples/folio/sanderling/predicates.ts";
|
||||
import { readHomeTotalBalance } from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
test("on Home with two cards ($10, $20): returns $30", () => {
|
||||
assert.equal(
|
||||
computeHomeTotalBalance({
|
||||
cardBalanceTexts: ["$10.00", "$20.00"],
|
||||
test("on Home the app's own total is the reading, the carrier and fresh", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "home", totalText: "$30.00", previousCarrier: 0 }),
|
||||
{ value: 3000, carrier: 3000, fresh: true },
|
||||
);
|
||||
});
|
||||
|
||||
test("off Home there is nothing to read, so the carrier is reported unchanged", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "ledger", totalText: undefined, previousCarrier: 3000 }),
|
||||
{ value: 3000, carrier: 3000, fresh: false },
|
||||
);
|
||||
});
|
||||
|
||||
test("off Home before any Home visit reports the null carrier, still not fresh", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "ledger", totalText: undefined, previousCarrier: null }),
|
||||
{ value: null, carrier: null, fresh: false },
|
||||
);
|
||||
});
|
||||
|
||||
test("a negative total parses with its sign", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "home", totalText: "-$1,234.56", previousCarrier: 0 }),
|
||||
{ value: -123456, carrier: -123456, fresh: true },
|
||||
);
|
||||
});
|
||||
|
||||
test("a fresh Home total overrides whatever the carrier held", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "home", totalText: "$7.50", previousCarrier: 9999 }),
|
||||
{ value: 750, carrier: 750, fresh: true },
|
||||
);
|
||||
});
|
||||
|
||||
// The poisoned carrier. An unreadable Home total is UNKNOWN for that step, so
|
||||
// null is reported and the property goes vacuous, but the carrier must keep the
|
||||
// last total we actually read. Writing null into the carrier is what used to end
|
||||
// the run: off-Home steps hand the carrier straight back, so a single
|
||||
// unreadable Home left every later step null.
|
||||
test("an unreadable Home total reports null but leaves the carrier intact", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "home", totalText: undefined, previousCarrier: 3000 }),
|
||||
{ value: null, carrier: 3000, fresh: false },
|
||||
);
|
||||
});
|
||||
|
||||
test("a garbled Home total is unknown, not zero", () => {
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route: "home", totalText: "$", previousCarrier: 3000 }),
|
||||
{ value: null, carrier: 3000, fresh: false },
|
||||
);
|
||||
});
|
||||
|
||||
test("an unreadable Home no longer poisons the steps after it", () => {
|
||||
let carrier: number | null = null;
|
||||
const seen: (number | null)[] = [];
|
||||
const step = (route: string | null, totalText: string | undefined) => {
|
||||
const reading = readHomeTotalBalance({ route, totalText, previousCarrier: carrier });
|
||||
carrier = reading.carrier;
|
||||
seen.push(reading.value);
|
||||
};
|
||||
|
||||
step("home", "$30.00");
|
||||
step("home", undefined);
|
||||
step("ledger", undefined);
|
||||
step("ledger", undefined);
|
||||
step("home", "$50.00");
|
||||
|
||||
assert.deepEqual(seen, [3000, null, 3000, 3000, 5000]);
|
||||
});
|
||||
|
||||
// The clipped fifth account card that started this: it is not a card reading
|
||||
// any more, and the footer total the app renders is unaffected by which cards
|
||||
// the viewport happens to fit.
|
||||
test("Home total is one node, so an off-screen account cannot change it", () => {
|
||||
const withFiveCards = readHomeTotalBalance({
|
||||
route: "home",
|
||||
totalText: "$2,589.00",
|
||||
previousCarrier: 0,
|
||||
}),
|
||||
3000,
|
||||
);
|
||||
});
|
||||
|
||||
test("off Home (no cards) after a Home visit of $30: returns carrier $30", () => {
|
||||
assert.equal(
|
||||
computeHomeTotalBalance({
|
||||
cardBalanceTexts: [],
|
||||
previousCarrier: 3000,
|
||||
}),
|
||||
3000,
|
||||
);
|
||||
});
|
||||
|
||||
test("off Home (no cards) with carrier still 0: returns 0", () => {
|
||||
assert.equal(
|
||||
computeHomeTotalBalance({
|
||||
cardBalanceTexts: [],
|
||||
previousCarrier: 0,
|
||||
}),
|
||||
0,
|
||||
);
|
||||
});
|
||||
|
||||
test("sequence: Home $30, off-Home, Home $50 tracks new Home totals", () => {
|
||||
let carrier = 0;
|
||||
carrier = computeHomeTotalBalance({
|
||||
cardBalanceTexts: ["$10.00", "$20.00"],
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
assert.equal(carrier, 3000);
|
||||
carrier = computeHomeTotalBalance({
|
||||
cardBalanceTexts: [],
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
assert.equal(carrier, 3000);
|
||||
carrier = computeHomeTotalBalance({
|
||||
cardBalanceTexts: ["$20.00", "$30.00"],
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
assert.equal(carrier, 5000);
|
||||
});
|
||||
|
||||
test("Ledger step (no Home cards) holds the carrier, ignores Ledger balance", () => {
|
||||
let carrier = 0;
|
||||
carrier = computeHomeTotalBalance({
|
||||
cardBalanceTexts: ["$10.00", "$20.00"],
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
assert.equal(carrier, 3000);
|
||||
carrier = computeHomeTotalBalance({
|
||||
cardBalanceTexts: [],
|
||||
previousCarrier: carrier,
|
||||
});
|
||||
assert.equal(carrier, 3000);
|
||||
});
|
||||
|
||||
test("negative card balance parses with sign and sums correctly", () => {
|
||||
assert.equal(
|
||||
computeHomeTotalBalance({
|
||||
cardBalanceTexts: ["-$5.00", "$10.00"],
|
||||
previousCarrier: 0,
|
||||
}),
|
||||
500,
|
||||
);
|
||||
});
|
||||
|
||||
test("single card on Home overrides any previous carrier", () => {
|
||||
assert.equal(
|
||||
computeHomeTotalBalance({
|
||||
cardBalanceTexts: ["$7.50"],
|
||||
previousCarrier: 9999,
|
||||
}),
|
||||
750,
|
||||
);
|
||||
});
|
||||
|
||||
test("undefined card balance text is treated as 0", () => {
|
||||
assert.equal(
|
||||
computeHomeTotalBalance({
|
||||
cardBalanceTexts: [undefined, "$10.00"],
|
||||
previousCarrier: 0,
|
||||
}),
|
||||
1000,
|
||||
);
|
||||
assert.deepEqual(withFiveCards, { value: 258900, carrier: 258900, fresh: true });
|
||||
});
|
||||
@@ -0,0 +1,142 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import {
|
||||
countSubmitsInWindow,
|
||||
readHomeCards,
|
||||
readHomeTotalBalance,
|
||||
routeOfFrame,
|
||||
submitChangesBalanceByTypedAmount,
|
||||
} from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
// The spec's own screen table. A frame is the set of markers its accessibility
|
||||
// tree carries, which is all routeOfFrame is allowed to look at.
|
||||
const SCREENS = {
|
||||
login: "LoginScreen",
|
||||
"add-account": "AddAccountScreen",
|
||||
"add-transaction": "AddTransactionScreen",
|
||||
ledger: "LedgerScreen",
|
||||
home: "HomeScreen",
|
||||
};
|
||||
|
||||
const frame =
|
||||
(...tags: string[]) =>
|
||||
(tag: string) =>
|
||||
tags.includes(tag);
|
||||
|
||||
test("a frame showing one screen names its route", () => {
|
||||
assert.equal(routeOfFrame(SCREENS, frame("LoginScreen")), "login");
|
||||
assert.equal(routeOfFrame(SCREENS, frame("AddAccountScreen")), "add-account");
|
||||
assert.equal(routeOfFrame(SCREENS, frame("AddTransactionScreen")), "add-transaction");
|
||||
assert.equal(routeOfFrame(SCREENS, frame("LedgerScreen")), "ledger");
|
||||
assert.equal(routeOfFrame(SCREENS, frame("HomeScreen")), "home");
|
||||
});
|
||||
|
||||
test("a frame showing no screen at all is unknown", () => {
|
||||
assert.equal(routeOfFrame(SCREENS, frame()), null);
|
||||
assert.equal(routeOfFrame(SCREENS, frame("SomethingElse")), null);
|
||||
});
|
||||
|
||||
// The android transition frame: 425 of 1879 steps across 17 measured runs carry
|
||||
// two screens, in every combination the navigation graph allows. Ranking the
|
||||
// markers and taking the first answers add-transaction for the first of these
|
||||
// while a second, unscoped look answers "on Home" -- and two answers for one
|
||||
// frame is the defect. There is one answer now, and on a transition frame it is
|
||||
// "I do not know".
|
||||
test("a transition frame showing two screens names neither", () => {
|
||||
assert.equal(routeOfFrame(SCREENS, frame("AddTransactionScreen", "HomeScreen")), null);
|
||||
assert.equal(routeOfFrame(SCREENS, frame("HomeScreen", "LedgerScreen")), null);
|
||||
assert.equal(routeOfFrame(SCREENS, frame("AddAccountScreen", "HomeScreen")), null);
|
||||
assert.equal(routeOfFrame(SCREENS, frame("HomeScreen", "LoginScreen")), null);
|
||||
assert.equal(routeOfFrame(SCREENS, frame("AddTransactionScreen", "LedgerScreen")), null);
|
||||
});
|
||||
|
||||
test("the three-screen frames android also emits name nothing", () => {
|
||||
assert.equal(
|
||||
routeOfFrame(SCREENS, frame("AddTransactionScreen", "HomeScreen", "LedgerScreen")),
|
||||
null,
|
||||
);
|
||||
});
|
||||
|
||||
// Everything read off Home takes the route as its only input, so a frame that
|
||||
// is not Home cannot be read as Home by anything.
|
||||
test("a transition frame's half-drawn Home total is not a reading", () => {
|
||||
const route = routeOfFrame(SCREENS, frame("AddTransactionScreen", "HomeScreen"));
|
||||
assert.deepEqual(
|
||||
readHomeTotalBalance({ route, totalText: "$86,911.00", previousCarrier: 8681600 }),
|
||||
{ value: 8681600, carrier: 8681600, fresh: false },
|
||||
);
|
||||
});
|
||||
|
||||
test("nor is its half-drawn card list", () => {
|
||||
const route = routeOfFrame(SCREENS, frame("AddAccountScreen", "HomeScreen"));
|
||||
const carried = { Travel: "8", Checking: "0" };
|
||||
const partial = { Travel: "8" };
|
||||
assert.deepEqual(readHomeCards<Record<string, string>>({ route, reading: partial, previousCarrier: carried }), {
|
||||
value: carried,
|
||||
carrier: carried,
|
||||
fresh: false,
|
||||
});
|
||||
});
|
||||
|
||||
// The false conviction itself, android seed 3, steps 98-102 of the recorded
|
||||
// trace. Five submits deep into the window the app sits on AddTransaction with
|
||||
// "339" typed; a double-tap on Back starts the trip Home; the frame that comes
|
||||
// back carries BOTH screens with Home's total already drawn behind the outgoing
|
||||
// one. The old spec read that total as a fresh Home reading, reset the window to
|
||||
// zero, and aimed the next tap at a TxnSubmit button that had stopped existing.
|
||||
// The tap landed on Home, committed nothing, and the property demanded 33900 of
|
||||
// movement for it. Nine of the eleven android convictions were this, all at
|
||||
// delta 0.0x, all a single Tap where the real bug is a DoubleTap.
|
||||
test("the measured android transition chain no longer convicts at delta 0", () => {
|
||||
let carrier: number | null = 8681600;
|
||||
let submits = 5;
|
||||
const step = (
|
||||
tags: string[],
|
||||
totalText: string | undefined,
|
||||
lastAction: { kind: string; on: string } | null,
|
||||
) => {
|
||||
const route = routeOfFrame(SCREENS, frame(...tags));
|
||||
const reading = readHomeTotalBalance({ route, totalText, previousCarrier: carrier });
|
||||
const window = countSubmitsInWindow({ previousCount: submits, lastAction, fresh: reading.fresh });
|
||||
carrier = reading.carrier;
|
||||
submits = window.next;
|
||||
return { route, total: reading.value, submits: window.reported };
|
||||
};
|
||||
|
||||
const back = { kind: "DoubleTap", on: "id:BackButton" };
|
||||
const phantomSubmit = { kind: "Tap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" };
|
||||
|
||||
const transition = step(["AddTransactionScreen", "HomeScreen"], "$86,911.00", back);
|
||||
assert.equal(transition.route, null);
|
||||
assert.equal(transition.total, 8681600);
|
||||
assert.equal(transition.submits, 5);
|
||||
|
||||
const landing = step(["HomeScreen"], "$86,911.00", phantomSubmit);
|
||||
assert.equal(landing.submits, 6);
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: landing.route,
|
||||
lastAction: phantomSubmit,
|
||||
submitsInWindow: landing.submits,
|
||||
typedAmount: 33900,
|
||||
prevTotalBalance: transition.total,
|
||||
currTotalBalance: landing.total,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
|
||||
// What the reset bought the old spec: the same landing, judged against a
|
||||
// window of one and a total the transition frame had already banked.
|
||||
assert.equal(
|
||||
submitChangesBalanceByTypedAmount({
|
||||
route: "home",
|
||||
lastAction: phantomSubmit,
|
||||
submitsInWindow: 1,
|
||||
typedAmount: 33900,
|
||||
prevTotalBalance: 8691100,
|
||||
currTotalBalance: 8691100,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,452 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { test } from "node:test";
|
||||
|
||||
import {
|
||||
cardTxnCount,
|
||||
committedTransactionsExceedSubmits,
|
||||
homeTxnCountsOf,
|
||||
} from "../../../examples/folio/sanderling/predicates.ts";
|
||||
|
||||
// Android and iOS give the count its own node; web merges the card into one
|
||||
// string, where the count sits between the name and the balance. The reading
|
||||
// carries which of the two it came from: a number is a count nothing else could
|
||||
// have leaked into, a string is a digit run that may have.
|
||||
test("a dedicated count node reads as a number, not a digit run", () => {
|
||||
assert.equal(
|
||||
cardTxnCount({ childText: "12 transactions", cardText: "INInvestments12 transactions$2,589.00" }),
|
||||
12,
|
||||
);
|
||||
});
|
||||
|
||||
test("merged card text: the count is taken from in front of the balance", () => {
|
||||
assert.equal(
|
||||
cardTxnCount({ childText: undefined, cardText: "INInvestments12 transactions$2,589.00" }),
|
||||
"12",
|
||||
);
|
||||
});
|
||||
|
||||
test("merged card text: the singular label parses too", () => {
|
||||
assert.equal(
|
||||
cardTxnCount({ childText: undefined, cardText: "SASavings1 transaction$118.00" }),
|
||||
"1",
|
||||
);
|
||||
});
|
||||
|
||||
// The balance has to come off first, or a name ending in digits would be read
|
||||
// as the count.
|
||||
test("a card with no readable count is unknown, not zero", () => {
|
||||
assert.equal(cardTxnCount({ childText: undefined, cardText: undefined }), undefined);
|
||||
assert.equal(cardTxnCount({ childText: undefined, cardText: "no digits here" }), undefined);
|
||||
assert.equal(cardTxnCount({ childText: "", cardText: "AA" + "a".repeat(198) }), undefined);
|
||||
});
|
||||
|
||||
// Measured on a real web run: the account named "-1" holding 2 transactions
|
||||
// merges to "-1-12 transactions-$119.00", and the maximal digit run reads 12.
|
||||
test("merged text runs a digit-ending name into the count", () => {
|
||||
assert.equal(
|
||||
cardTxnCount({ childText: undefined, cardText: "-1-12 transactions-$119.00" }),
|
||||
"12",
|
||||
);
|
||||
});
|
||||
|
||||
const before = { Checking: "3", Savings: "1" };
|
||||
|
||||
test("healthy window: three submits, three transactions", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: { Checking: "5", Savings: "2" },
|
||||
submitsInWindow: 3,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("rejected submits commit nothing, which is under the bound", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: { Checking: "3", Savings: "1" },
|
||||
submitsInWindow: 4,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The bug, stated directly: one tap, two rows.
|
||||
test("double submit: one action commits two transactions", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: { Checking: "5", Savings: "1" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The point of counting actions against transactions rather than gating on a
|
||||
// one-submit window: a wide window is still a sound comparison.
|
||||
test("wide window: five submits committing six transactions still fires", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: { Checking: "8", Savings: "4" },
|
||||
submitsInWindow: 5,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("boundary: committed equal to the submit count is not a violation", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: { Checking: "4", Savings: "1" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("boundary: one transaction past the submit count is", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: { Checking: "4", Savings: "2" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// Only accounts in both readings count. A card that scrolled out of the
|
||||
// viewport, or one whose count was unreadable, drops out of the sum, so the
|
||||
// result is a lower bound on what committed. Losing a card can only cost a
|
||||
// detection; it must never manufacture one.
|
||||
test("an account missing from the later reading is not counted", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "3", Savings: "1" },
|
||||
countsAfter: { Checking: "3" },
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("an account appearing only in the later reading is not counted", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "3" },
|
||||
countsAfter: { Checking: "3", Travel: "9" },
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("a card that scrolled away and back is not double counted", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "3" },
|
||||
countsAfter: { Checking: "4", Savings: "40" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("an unknown reading on either side is not evidence", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: null,
|
||||
countsAfter: { Checking: "99" },
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "0" },
|
||||
countsAfter: null,
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The real trace this came from: at the violating step of seeds 3 and 5 the
|
||||
// window held exactly one submit action and the account's count moved by two.
|
||||
test("the measured web witness: submits 1, count delta 2", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { INInvestments: "12", Aa: "0" },
|
||||
countsAfter: { INInvestments: "14", Aa: "0" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The length rule, which is what keeps the merged-text prefix honest. The
|
||||
// account named "-1" reads 19 at nine transactions and 110 at ten: same account,
|
||||
// a delta of 91 out of a true delta of 1. Different run lengths are dropped.
|
||||
test("a count crossing a digit boundary is dropped, not convicted on", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { "-1-": "19" },
|
||||
countsAfter: { "-1-": "110" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("same run length keeps the prefixed delta exact", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { "-1-": "110" },
|
||||
countsAfter: { "-1-": "112" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { "-1-": "110" },
|
||||
countsAfter: { "-1-": "111" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("an unreadably long run is not evidence", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "1".repeat(21) },
|
||||
countsAfter: { Checking: "9".repeat(21) },
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The other side of that rule, and the reason it is scoped to merged text: a
|
||||
// count read off its own node has no account name in front of it, so its digits
|
||||
// ARE the count and a decade crossing is just a number getting longer. Both
|
||||
// windows below are real android seed-9 readings that the unscoped length rule
|
||||
// threw away, in runs that then finished clean.
|
||||
test("a dedicated node's count crossing a decade is usable evidence", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: 7 },
|
||||
countsAfter: { Checking: 12 },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Savings: 4 },
|
||||
countsAfter: { Savings: 10 },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// Recovering the window is only worth anything if it still acquits the healthy
|
||||
// case, so the same crossing under a submit that earned it must not fire.
|
||||
test("a dedicated node's healthy decade crossing does not convict", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: 9 },
|
||||
countsAfter: { Checking: 10 },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: 9 },
|
||||
countsAfter: { Checking: 11 },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// The same numbers off merged text, where the digits may not be the count at
|
||||
// all: still dropped.
|
||||
test("the merged-text equivalent of that crossing is still dropped", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "9" },
|
||||
countsAfter: { Checking: "10" },
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "7" },
|
||||
countsAfter: { Checking: "12" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The boundary itself. A pair whose two readings came from different sources is
|
||||
// vouched for by neither rule: the string may carry a name prefix the number
|
||||
// does not, so subtracting them is not a transaction count.
|
||||
test("a pair straddling the two sources is not comparable", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: 7 },
|
||||
countsAfter: { Checking: "12" },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: { Checking: "7" },
|
||||
countsAfter: { Checking: 12 },
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// End to end from the two accessibility shapes, which is where the distinction
|
||||
// is actually made: the same account, the same true counts, read once off a
|
||||
// dedicated node and once off merged card text.
|
||||
const dedicated = (name: string, count: number) => ({
|
||||
name,
|
||||
balance: 0,
|
||||
count: cardTxnCount({ childText: `${count} transactions`, cardText: undefined }),
|
||||
});
|
||||
|
||||
const merged = (initials: string, name: string, count: number) => ({
|
||||
name,
|
||||
balance: 0,
|
||||
count: cardTxnCount({
|
||||
childText: undefined,
|
||||
cardText: `${initials}${name}${count} transactions$0.00`,
|
||||
}),
|
||||
});
|
||||
|
||||
test("a dedicated-node card list convicts across a decade", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: homeTxnCountsOf([dedicated("Checking", 9)]),
|
||||
countsAfter: homeTxnCountsOf([dedicated("Checking", 11)]),
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("the merged-text card list drops the same pair", () => {
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: homeTxnCountsOf([merged("CH", "Checking", 9)]),
|
||||
countsAfter: homeTxnCountsOf([merged("CH", "Checking", 11)]),
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// Home lists whatever fits the viewport, and Folio lets two accounts share a
|
||||
// name, so one name can arrive on two cards. Keying counts by name collapsed
|
||||
// them onto the last card, and the two readings a window compares then came off
|
||||
// DIFFERENT cards: the probe below is a healthy app, one submit, and a scroll.
|
||||
test("two cards sharing a name do not become one count", () => {
|
||||
const before = homeTxnCountsOf([{ name: "Travel", balance: 0, count: 0 }]);
|
||||
const after = homeTxnCountsOf([
|
||||
{ name: "Travel", balance: 0, count: 0 },
|
||||
{ name: "Travel", balance: 500, count: 8 },
|
||||
]);
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: after,
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
assert.deepEqual(after, null);
|
||||
});
|
||||
|
||||
test("a name on two cards is dropped from both readings", () => {
|
||||
const before = homeTxnCountsOf([
|
||||
{ name: "Travel", balance: 0, count: "3" },
|
||||
{ name: "Travel", balance: 0, count: "9" },
|
||||
{ name: "Checking", balance: 0, count: "2" },
|
||||
]);
|
||||
const after = homeTxnCountsOf([
|
||||
{ name: "Travel", balance: 0, count: "3" },
|
||||
{ name: "Travel", balance: 0, count: "11" },
|
||||
{ name: "Checking", balance: 0, count: "2" },
|
||||
]);
|
||||
assert.deepEqual(before, { Checking: "2" });
|
||||
assert.deepEqual(after, { Checking: "2" });
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: after,
|
||||
submitsInWindow: 0,
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
// The other card need not be readable to spoil the identity: an unreadable
|
||||
// count still means the name on the map may not be the card that was read.
|
||||
test("a duplicate name is dropped even when the twin has no count", () => {
|
||||
assert.deepEqual(
|
||||
homeTxnCountsOf([
|
||||
{ name: "Travel", balance: 0, count: 4 },
|
||||
{ name: "Travel", balance: 0, count: undefined },
|
||||
{ name: "Savings", balance: 0, count: 1 },
|
||||
]),
|
||||
{ Savings: 1 },
|
||||
);
|
||||
});
|
||||
|
||||
// Dropping the ambiguous name must not disable the property for the rest.
|
||||
test("a unique name is still counted beside a dropped duplicate", () => {
|
||||
const before = homeTxnCountsOf([
|
||||
{ name: "Travel", balance: 0, count: 3 },
|
||||
{ name: "Travel", balance: 0, count: 1 },
|
||||
{ name: "Checking", balance: 0, count: 4 },
|
||||
]);
|
||||
const after = homeTxnCountsOf([
|
||||
{ name: "Travel", balance: 0, count: 3 },
|
||||
{ name: "Travel", balance: 0, count: 1 },
|
||||
{ name: "Checking", balance: 0, count: 6 },
|
||||
]);
|
||||
assert.deepEqual(before, { Checking: 4 });
|
||||
assert.equal(
|
||||
committedTransactionsExceedSubmits({
|
||||
countsBefore: before,
|
||||
countsAfter: after,
|
||||
submitsInWindow: 1,
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("distinct names are all counted", () => {
|
||||
assert.deepEqual(
|
||||
homeTxnCountsOf([
|
||||
{ name: "Checking", balance: 0, count: "3" },
|
||||
{ name: "Savings", balance: 0, count: "1" },
|
||||
]),
|
||||
{ Checking: "3", Savings: "1" },
|
||||
);
|
||||
});
|
||||
@@ -43,14 +43,39 @@ test("zero returns 0", () => {
|
||||
assert.equal(parseTypedAmount("0"), 0);
|
||||
});
|
||||
|
||||
test("leading plus sign tolerated as positive", () => {
|
||||
assert.equal(parseTypedAmount("+50"), 5000);
|
||||
// The app's parseCents matches ^\d+(\.\d{1,2})?$ against the trimmed input, so
|
||||
// a sign is rejected and no transaction is created. Reading "-50" as 5000 cents
|
||||
// would make the balance property demand a move the app never made.
|
||||
test("leading plus sign rejected, like the app", () => {
|
||||
assert.equal(parseTypedAmount("+50"), 0);
|
||||
});
|
||||
|
||||
test("leading minus sign tolerated as positive", () => {
|
||||
assert.equal(parseTypedAmount("-50"), 5000);
|
||||
test("leading minus sign rejected, like the app", () => {
|
||||
assert.equal(parseTypedAmount("-50"), 0);
|
||||
});
|
||||
|
||||
test("comma-separated thousands accepted", () => {
|
||||
assert.equal(parseTypedAmount("1,234.56"), 123456);
|
||||
});
|
||||
|
||||
// The input corpus types this 21-digit run into every field. parseCents calls
|
||||
// toLongOrNull on the whole part, which is null past Long.MAX, so the app
|
||||
// refuses the submit; float64 would have read it as 1e23 and asked the property
|
||||
// to find a balance move of 1e23 cents that never happened.
|
||||
test("21-digit corpus amount returns 0: the app rejects it", () => {
|
||||
assert.equal(parseTypedAmount("999999999999999999999"), 0);
|
||||
});
|
||||
|
||||
test("amount too large for exact cents returns 0", () => {
|
||||
assert.equal(parseTypedAmount("100000000000000"), 0);
|
||||
});
|
||||
|
||||
// 9007199254740991 cents is Number.MAX_SAFE_INTEGER: the last amount whose
|
||||
// cents survive the multiply intact.
|
||||
test("largest exactly representable amount is kept", () => {
|
||||
assert.equal(parseTypedAmount("90071992547409.91"), 9007199254740991);
|
||||
});
|
||||
|
||||
test("one cent past the safe range returns 0", () => {
|
||||
assert.equal(parseTypedAmount("90071992547409.92"), 0);
|
||||
});
|
||||
@@ -14,6 +14,15 @@ export interface FakeElementSpec {
|
||||
y: number;
|
||||
width: number;
|
||||
height: number;
|
||||
// id/testid/label/alt/title are how the host names a target: it builds the
|
||||
// selector an action carries from them, so a fake needs them to exercise that
|
||||
// naming. alt and title are the fallbacks the hierarchy dump folds into
|
||||
// content-desc, which the host has to fall back to in the same order.
|
||||
id?: string;
|
||||
testid?: string;
|
||||
label?: string;
|
||||
alt?: string;
|
||||
title?: string;
|
||||
// clickable/editable place the element in the selector sets the host queries;
|
||||
// the fake answers those queries directly rather than matching CSS.
|
||||
clickable?: boolean;
|
||||
@@ -28,6 +37,9 @@ export interface FakeElement extends FakeElementSpec {
|
||||
tagName: string;
|
||||
type: string;
|
||||
isContentEditable: boolean;
|
||||
id: string;
|
||||
dataset: Record<string, string | undefined>;
|
||||
getAttribute(name: string): string | null;
|
||||
scrollHeight: number;
|
||||
clientHeight: number;
|
||||
scrollWidth: number;
|
||||
@@ -49,6 +61,10 @@ export function fakeElement(spec: FakeElementSpec): FakeElement {
|
||||
tagName: spec.tag.toUpperCase(),
|
||||
type: spec.tag === "input" ? "text" : "",
|
||||
isContentEditable: editable && spec.tag !== "input" && spec.tag !== "textarea",
|
||||
id: spec.id ?? "",
|
||||
dataset: { testid: spec.testid },
|
||||
getAttribute: (name: string) =>
|
||||
({ "aria-label": spec.label, alt: spec.alt, title: spec.title })[name] ?? null,
|
||||
scrollHeight: spec.overflows ? spec.height * 2 : spec.height,
|
||||
clientHeight: spec.height,
|
||||
scrollWidth: spec.width,
|
||||
|
||||
@@ -128,6 +128,53 @@ test("queryTargets reports a disabled control rather than dropping it", () => {
|
||||
});
|
||||
});
|
||||
|
||||
// A target with no selector is a target no property can name. The action the
|
||||
// picker builds from it carries coordinates only, so `lastAction.on` is empty
|
||||
// and any property matching on WHICH control was acted upon cannot fire.
|
||||
test("queryTargets names a uniquely identified target", () => {
|
||||
const submit = fakeElement({
|
||||
tag: "button", x: 0, y: 0, width: 40, height: 20, clickable: true, id: "TxnSubmit",
|
||||
});
|
||||
const byTestid = fakeElement({
|
||||
tag: "button", x: 0, y: 40, width: 40, height: 20, clickable: true, testid: "cancel",
|
||||
});
|
||||
const byLabel = fakeElement({
|
||||
tag: "button", x: 0, y: 80, width: 40, height: 20, clickable: true, label: "Close",
|
||||
});
|
||||
// alt and title are the fallbacks the hierarchy dump folds into content-desc,
|
||||
// so the host has to fall back to them in the same order or a name it calls
|
||||
// unique resolves to a different element on the Go side.
|
||||
const byAlt = fakeElement({ tag: "img", x: 0, y: 120, width: 40, height: 20, alt: "Logo" });
|
||||
const byTitle = fakeElement({ tag: "div", x: 0, y: 160, width: 40, height: 20, title: "Help" });
|
||||
const anonymous = fakeElement({ tag: "div", x: 0, y: 200, width: 10, height: 10 });
|
||||
withFakeDocument([submit, byTestid, byLabel, byAlt, byTitle, anonymous], () => {
|
||||
const targets = host.queryTargets();
|
||||
assert.equal(targets[0]!.selector, "id:TxnSubmit");
|
||||
assert.equal(targets[1]!.selector, "data-testid:cancel");
|
||||
assert.equal(targets[2]!.selector, "desc:Close");
|
||||
assert.equal(targets[3]!.selector, "desc:Logo");
|
||||
assert.equal(targets[4]!.selector, "desc:Help");
|
||||
assert.equal(targets[5]!.selector, undefined);
|
||||
});
|
||||
});
|
||||
|
||||
// A repeated id (folio's Home screen renders one AccountCard testTag per
|
||||
// account) names no single element, so the runner would re-resolve the action
|
||||
// onto whichever sibling it found first. Better unnamed than mis-aimed.
|
||||
test("queryTargets leaves duplicated identities unnamed", () => {
|
||||
const first = fakeElement({
|
||||
tag: "div", x: 0, y: 0, width: 40, height: 20, clickable: true, id: "AccountCard",
|
||||
});
|
||||
const second = fakeElement({
|
||||
tag: "div", x: 0, y: 40, width: 40, height: 20, clickable: true, id: "AccountCard",
|
||||
});
|
||||
withFakeDocument([first, second], () => {
|
||||
const targets = host.queryTargets();
|
||||
assert.equal(targets[0]!.selector, undefined);
|
||||
assert.equal(targets[1]!.selector, undefined);
|
||||
});
|
||||
});
|
||||
|
||||
test("queryTargets caches within a tick until reset", () => {
|
||||
const button = fakeElement({ tag: "button", x: 0, y: 0, width: 10, height: 10, clickable: true });
|
||||
withFakeDocument([button], () => {
|
||||
@@ -159,6 +206,13 @@ function withState(run: () => void) {
|
||||
}
|
||||
}
|
||||
|
||||
// Every reading leaves the runtime inside a {value} envelope, so an extractor
|
||||
// whose getter returned undefined keeps its index instead of being dropped by
|
||||
// JSON.stringify.
|
||||
function readingOf(values: Record<number, { value?: unknown }>, index: number): unknown {
|
||||
return values[index]!.value;
|
||||
}
|
||||
|
||||
test("named() sets the extractor's display name", () => {
|
||||
const handle = __testing__.runtime.extract(() => "home").named("route");
|
||||
const entry = __testing__.extractors.find((e) => e.handle === handle);
|
||||
@@ -204,6 +258,63 @@ test("an uncaught cross-extractor read aborts evaluateExtractors", () => {
|
||||
);
|
||||
});
|
||||
|
||||
// JSON.stringify drops an undefined-valued key, so a reading written straight
|
||||
// into the table took the extractor's whole INDEX with it when the getter
|
||||
// returned undefined - folio's on(route, tag) off its own screen, which is most
|
||||
// of its extractors on most steps. The host then kept goja's dump-derived value
|
||||
// for those and the page's for the rest, and a property comparing previous to
|
||||
// current across that split convicts an app that did nothing wrong.
|
||||
test("an extractor that returned undefined keeps its index through JSON", () => {
|
||||
__testing__.extractors.length = 0;
|
||||
__testing__.runtime.extract(() => undefined);
|
||||
__testing__.runtime.extract(() => null);
|
||||
__testing__.runtime.extract(() => 5);
|
||||
let table: Record<number, { value?: unknown }> = {};
|
||||
withState(() => {
|
||||
table = __testing__.evaluateExtractors();
|
||||
});
|
||||
|
||||
const overTheWire = JSON.parse(JSON.stringify(table)) as Record<string, { value?: unknown }>;
|
||||
assert.deepEqual(Object.keys(overTheWire), ["0", "1", "2"]);
|
||||
// undefined and null have to stay distinguishable across the wire: the goja
|
||||
// host records undefined for a getter that returned undefined, so reporting
|
||||
// null instead would make `x.current === undefined` answer one thing on
|
||||
// native and another on web.
|
||||
assert.equal("value" in overTheWire["0"]!, false);
|
||||
assert.equal(overTheWire["1"]!.value, null);
|
||||
assert.equal(overTheWire["2"]!.value, 5);
|
||||
});
|
||||
|
||||
// state.lastAction is the one piece of state the page cannot observe for
|
||||
// itself: only the runner knows which action it actually applied. While the web
|
||||
// runtime hardcoded null there, a spec property gated on the last action (e.g.
|
||||
// folio's submitMovesBalanceByTypedAmount, which only looks at taps on
|
||||
// TxnSubmit) was vacuously true on web forever, and the run went green having
|
||||
// checked nothing.
|
||||
function lastActionSeenByASpec(pushed: unknown): unknown {
|
||||
const setLastAction = (globalThis as Record<string, unknown>)
|
||||
.__sanderlingSetLastAction__ as (value: unknown) => void;
|
||||
__testing__.extractors.length = 0;
|
||||
__testing__.runtime.extract((state) => (state as { lastAction: unknown }).lastAction);
|
||||
let out: Record<number, { value?: unknown }> = {};
|
||||
withState(() => {
|
||||
setLastAction(pushed);
|
||||
out = __testing__.evaluateExtractors();
|
||||
});
|
||||
return readingOf(out, 0);
|
||||
}
|
||||
|
||||
test("state.lastAction carries the action the host pushed", () => {
|
||||
const action = { kind: "Tap", on: "id:TxnSubmit" };
|
||||
assert.deepEqual(lastActionSeenByASpec(action), action);
|
||||
});
|
||||
|
||||
test("state.lastAction is null when the host pushed nothing", () => {
|
||||
// The first step of a run, and any step whose action was never applied: the
|
||||
// goja host reports null there, so the web host must too.
|
||||
assert.equal(lastActionSeenByASpec(null), null);
|
||||
});
|
||||
|
||||
// sanitize runs over every extractor's return value before it leaves the
|
||||
// runtime. A user extractor that returns a page object reachable from
|
||||
// document/window can be self-referential, carry functions, or nest deeply;
|
||||
@@ -212,11 +323,11 @@ test("an uncaught cross-extractor read aborts evaluateExtractors", () => {
|
||||
function sanitizeViaExtract(value: unknown): unknown {
|
||||
__testing__.extractors.length = 0;
|
||||
__testing__.runtime.extract(() => value);
|
||||
let out: Record<number, unknown> = {};
|
||||
let out: Record<number, { value?: unknown }> = {};
|
||||
withState(() => {
|
||||
out = __testing__.evaluateExtractors();
|
||||
});
|
||||
return out[0];
|
||||
return readingOf(out, 0);
|
||||
}
|
||||
|
||||
test("sanitize breaks a self-referential cycle instead of overflowing", () => {
|
||||
@@ -521,3 +632,105 @@ test("a non-editable element carries no hintText", () => {
|
||||
assert.equal(attrsOf(element).hintText, undefined);
|
||||
assert.equal(handleOf(element).editable, false);
|
||||
});
|
||||
|
||||
// An ax element is labelled with the selector it was found by, in the same
|
||||
// canonical grammar selectorStringFromJS emits in internal/verifier/marshal.go.
|
||||
// The label is what a spec's own Tap({ on: state.ax.find(...) }) carries to the
|
||||
// runner: with no label the action is coordinates only, `lastAction.on` is
|
||||
// empty, and a property matching on WHICH control was tapped cannot fire.
|
||||
const { selectorTag } = __testing__;
|
||||
|
||||
test("selectorTag renders the selector shapes the goja host renders", () => {
|
||||
assert.equal(selectorTag("testTag:TxnSubmit"), "testTag:TxnSubmit");
|
||||
assert.equal(selectorTag({ testTag: "TxnSubmit" }), "testTag:TxnSubmit");
|
||||
assert.equal(
|
||||
selectorTag([{ testTag: "AddTransactionScreen" }, { testTag: "TxnSubmit" }]),
|
||||
"testTag:AddTransactionScreen > testTag:TxnSubmit",
|
||||
);
|
||||
assert.equal(selectorTag({ testTag: "Row", "aria-label": "first" }), "testTag:Row aria-label:first");
|
||||
assert.equal(selectorTag(undefined), "");
|
||||
});
|
||||
|
||||
// A selector path scopes the second segment to each match of the first. It
|
||||
// returned nothing at all on web while returning matches on native, so folio's
|
||||
// accounts/totalBalance extractors (findAll([{HomeScreen}, {AccountCard}]))
|
||||
// were empty on every web step and the properties over them checked nothing.
|
||||
test("ax.findAll resolves a selector path segment by segment", () => {
|
||||
const rect = { left: 0, top: 0, right: 10, bottom: 10, width: 10, height: 10 };
|
||||
const node = (id: string, answers: Record<string, unknown[]> = {}) => ({
|
||||
id,
|
||||
tagName: "DIV",
|
||||
className: "",
|
||||
textContent: id,
|
||||
dataset: {},
|
||||
getAttribute: () => null,
|
||||
getBoundingClientRect: () => rect,
|
||||
querySelectorAll: (selector: string) => answers[selector] ?? [],
|
||||
});
|
||||
const cardCss = `:is([data-testid="AccountCard"], [id="AccountCard"])`;
|
||||
const screenCss = `:is([data-testid="HomeScreen"], [id="HomeScreen"])`;
|
||||
const cards = [node("first"), node("second")];
|
||||
const home = node("HomeScreen", { [cardCss]: cards });
|
||||
|
||||
const g = globalThis as Record<string, unknown>;
|
||||
const originalDocument = g.document;
|
||||
const originalWindow = g.window;
|
||||
g.document = { querySelectorAll: (selector: string) => (selector === screenCss ? [home] : []) };
|
||||
g.window = {};
|
||||
try {
|
||||
__testing__.extractors.length = 0;
|
||||
__testing__.runtime.extract((state) => {
|
||||
const ax = (state as { ax: { findAll(s: unknown): Record<string, unknown>[] } }).ax;
|
||||
return ax
|
||||
.findAll([{ testTag: "HomeScreen" }, { testTag: "AccountCard" }])
|
||||
.map((card) => card.text);
|
||||
});
|
||||
const values = __testing__.evaluateExtractors();
|
||||
// Scoped to the head match: the cards come from the HomeScreen node, not
|
||||
// from a document-wide sweep for AccountCard.
|
||||
assert.deepEqual(readingOf(values, 0), ["first", "second"]);
|
||||
} finally {
|
||||
g.document = originalDocument;
|
||||
g.window = originalWindow;
|
||||
}
|
||||
});
|
||||
|
||||
test("ax.find and ax.findAll label the element with its selector", () => {
|
||||
const rect = { left: 0, top: 0, right: 10, bottom: 10, width: 10, height: 10 };
|
||||
const submit = {
|
||||
id: "TxnSubmit",
|
||||
tagName: "DIV",
|
||||
className: "",
|
||||
textContent: "Submit",
|
||||
dataset: {},
|
||||
getAttribute: () => null,
|
||||
getBoundingClientRect: () => rect,
|
||||
};
|
||||
const matches = `:is([data-testid="TxnSubmit"], [id="TxnSubmit"])`;
|
||||
const g = globalThis as Record<string, unknown>;
|
||||
const originalDocument = g.document;
|
||||
const originalWindow = g.window;
|
||||
g.document = { querySelectorAll: (selector: string) => (selector === matches ? [submit] : []) };
|
||||
g.window = {};
|
||||
try {
|
||||
__testing__.extractors.length = 0;
|
||||
__testing__.runtime.extract((state) => {
|
||||
const ax = (state as { ax: { find(s: unknown): Record<string, unknown> | undefined } }).ax;
|
||||
return ax.find({ testTag: "TxnSubmit" });
|
||||
});
|
||||
__testing__.runtime.extract((state) => {
|
||||
const ax = (state as { ax: { findAll(s: unknown): Record<string, unknown>[] } }).ax;
|
||||
return ax.findAll({ testTag: "TxnSubmit" });
|
||||
});
|
||||
const values = __testing__.evaluateExtractors();
|
||||
const found = readingOf(values, 0) as Record<string, unknown>;
|
||||
assert.equal(found.__sanderlingSelector, "testTag:TxnSubmit");
|
||||
// findAll passes each element through map(); passing the callback by
|
||||
// reference would hand the array INDEX to the runtime as the selector.
|
||||
const all = readingOf(values, 1) as Record<string, unknown>[];
|
||||
assert.equal(all[0]!.__sanderlingSelector, "testTag:TxnSubmit");
|
||||
} finally {
|
||||
g.document = originalDocument;
|
||||
g.window = originalWindow;
|
||||
}
|
||||
});
|
||||
Whitespace-only changes.
@@ -0,0 +1,180 @@
|
||||
// Sanderling fuzzing sanderling's own replay UI.
|
||||
//
|
||||
// Six of the seven properties here are cross-panel agreements: two panels that
|
||||
// derive the same fact by different paths have to say the same thing. That
|
||||
// holds for any trace, so this spec never needs recalibrating when the fixture
|
||||
// run changes. The seventh is the stock noUncaughtExceptions.
|
||||
//
|
||||
// The hooks it drives (data-testid, data-step, ...) were added to the UI for
|
||||
// this spec. Needing them is the lesson: a UI with no stable handles is a UI
|
||||
// nothing can assert on, and that is as true for a person writing a test as it
|
||||
// is for a fuzzer.
|
||||
|
||||
import { type AccessibilityElement, type Key, PressKey, Tap, actions, always, extract, from, next, weighted } from "@sanderling/spec";
|
||||
import { defaultActions, noUncaughtExceptions } from "@sanderling/spec/defaults";
|
||||
|
||||
function numberOf(value: string | undefined): number | null {
|
||||
if (value === undefined || value === "") return null;
|
||||
const parsed = Number(value);
|
||||
return Number.isFinite(parsed) ? parsed : null;
|
||||
}
|
||||
|
||||
function dataOf(element: AccessibilityElement | undefined, key: string): string | undefined {
|
||||
const attrs = (element as unknown as { attrs?: Record<string, string> } | undefined)?.attrs;
|
||||
return attrs ? attrs[key] : undefined;
|
||||
}
|
||||
|
||||
// The toolbar's own claim about which step is on screen, and how many there are.
|
||||
const toolbar = extract("toolbar", (s) => {
|
||||
const indicator = s.ax.find({ "data-testid": "step-indicator" });
|
||||
if (!indicator) return null;
|
||||
return {
|
||||
step: numberOf(dataOf(indicator, "step")),
|
||||
stepCount: numberOf(dataOf(indicator, "stepCount")),
|
||||
};
|
||||
});
|
||||
|
||||
// The step list's claim: which rows exist, which one is selected, which ones
|
||||
// carry a violation marker.
|
||||
const stepRows = extract("stepRows", (s) =>
|
||||
s.ax.findAll({ "data-testid": "step-row" }).map((row) => ({
|
||||
step: numberOf(dataOf(row, "step")),
|
||||
active: dataOf(row, "active") === "true",
|
||||
violating: dataOf(row, "violations") === "true",
|
||||
})),
|
||||
);
|
||||
|
||||
// What the "state before" panel is showing, read off the image URL it renders.
|
||||
// Scoped to that panel by name rather than by DOM position: an earlier draft
|
||||
// took the first screenshot on the page, and the fuzzer put the before panel on
|
||||
// another tab, which left the after panel's image first and fired the property
|
||||
// against a UI that was behaving correctly.
|
||||
const beforeScreenshotStep = extract("beforeScreenshotStep", (s) => {
|
||||
const image = s.ax.find([{ "data-testid": "state-before" }, { "data-testid": "screenshot" }]);
|
||||
return image ? numberOf(dataOf(image, "step")) : null;
|
||||
});
|
||||
|
||||
// Which tab is selected in each tab strip, as one comparable string.
|
||||
const activeTabs = extract("activeTabs", (s) =>
|
||||
s.ax
|
||||
.findAll({ "data-testid": "tab" })
|
||||
.filter((tab) => dataOf(tab, "active") === "true")
|
||||
.map((tab) => dataOf(tab, "tabId") ?? "")
|
||||
.join(","),
|
||||
);
|
||||
|
||||
// Two counts of one fact: the tab badge counts the step's violation records,
|
||||
// the panel counts the property rows it renders as violated.
|
||||
const violationBadges = extract("violationBadges", (s) =>
|
||||
s.ax.findAll({ "data-testid": "violations-badge" }).map((badge) => numberOf(badge.text)),
|
||||
);
|
||||
const violationPanelCounts = extract("violationPanelCounts", (s) =>
|
||||
s.ax
|
||||
.findAll({ "data-testid": "violations-panel" })
|
||||
.map((panel) => numberOf(dataOf(panel, "violationCount"))),
|
||||
);
|
||||
|
||||
// The step in the URL is user input: it survives a reload, a typo, and every
|
||||
// tap that navigates. It must always land inside the run.
|
||||
const selectedStepIsInRange = always(() => {
|
||||
const current = toolbar.current;
|
||||
if (!current || current.step === null || current.stepCount === null) return true;
|
||||
return current.step >= 1 && current.step <= current.stepCount;
|
||||
});
|
||||
|
||||
// Exactly one row is highlighted whenever the list is on screen. Zero means the
|
||||
// toolbar is showing a step the list has no row for, which is what an
|
||||
// off-by-one or a failed clamp looks like from the list's side.
|
||||
const exactlyOneStepIsSelected = always(() => {
|
||||
const rows = stepRows.current;
|
||||
if (rows.length === 0) return true;
|
||||
return rows.filter((row) => row.active).length === 1;
|
||||
});
|
||||
|
||||
// The step count the toolbar prints and the number of rows the list renders are
|
||||
// two readings of the same run.
|
||||
const stepCountMatchesTheList = always(() => {
|
||||
const current = toolbar.current;
|
||||
const rows = stepRows.current;
|
||||
if (!current || current.stepCount === null || rows.length === 0) return true;
|
||||
return current.stepCount === rows.length;
|
||||
});
|
||||
|
||||
// The screenshot panel builds its URL from the loaded run's step; the toolbar
|
||||
// prints the step from the URL. They must name the same step.
|
||||
const screenshotShowsTheSelectedStep = always(() => {
|
||||
const current = toolbar.current;
|
||||
const shown = beforeScreenshotStep.current;
|
||||
if (!current || current.step === null || shown === null) return true;
|
||||
return shown === current.step;
|
||||
});
|
||||
|
||||
// Switching a tab is a view change, not a navigation: it must never move the
|
||||
// run to another step.
|
||||
const switchingTabsKeepsTheStep = always(
|
||||
next(() => {
|
||||
const previousTabs = activeTabs.previous;
|
||||
const previousToolbar = toolbar.previous;
|
||||
const currentToolbar = toolbar.current;
|
||||
if (previousTabs === undefined || previousTabs === activeTabs.current) return true;
|
||||
if (!previousToolbar || !currentToolbar) return true;
|
||||
return previousToolbar.step === currentToolbar.step;
|
||||
}),
|
||||
);
|
||||
|
||||
// A badge that counts violations the panel cannot show a row for is a violation
|
||||
// the operator can see a number for and never read.
|
||||
const badgeCountMatchesThePanel = always(() => {
|
||||
const badges = violationBadges.current;
|
||||
const panels = violationPanelCounts.current;
|
||||
if (badges.length === 0 || panels.length === 0) return true;
|
||||
const badge = badges[0];
|
||||
if (badge === null) return true;
|
||||
return panels.every((count) => count === null || count === badge);
|
||||
});
|
||||
|
||||
export const properties = {
|
||||
noUncaughtExceptions,
|
||||
selectedStepIsInRange,
|
||||
exactlyOneStepIsSelected,
|
||||
stepCountMatchesTheList,
|
||||
screenshotShowsTheSelectedStep,
|
||||
switchingTabsKeepsTheStep,
|
||||
badgeCountMatchesThePanel,
|
||||
};
|
||||
|
||||
// Step rows get their own weight for the same reason tabs do below: the default
|
||||
// enumeration reaches them now that role="option" is in the tappable set, but
|
||||
// one row out of the page's clickable elements is a thin chance, and both
|
||||
// step-facing properties are vacuous on a run that never selects one.
|
||||
const rowElements = extract("rowElements", (s) => s.ax.findAll({ "data-testid": "step-row" }));
|
||||
|
||||
const selectAStep = actions(() => {
|
||||
const rows = rowElements.current;
|
||||
return rows.length === 0 ? [] : [Tap({ on: from(rows).generate() })];
|
||||
});
|
||||
|
||||
// Left/right are the UI's own prev/next step bindings, and the one path into
|
||||
// the step navigation that does not go through a tap.
|
||||
const arrowKeys = from<Key>(["left", "right"]);
|
||||
const navigateByKeyboard = actions(() => [PressKey({ key: arrowKeys.generate() })]);
|
||||
|
||||
// Tabs get their own weight rather than being left to the undirected tap mix:
|
||||
// with ~15 clickable elements on the page, an undirected run went 40 steps
|
||||
// without switching a single tab, which left both tab-facing properties
|
||||
// vacuously true. Weighting them is what makes those properties mean something.
|
||||
const tabElements = extract("tabElements", (s) => s.ax.findAll({ "data-testid": "tab" }));
|
||||
|
||||
const switchATab = actions(() => {
|
||||
const tabs = tabElements.current;
|
||||
return tabs.length === 0 ? [] : [Tap({ on: from(tabs).generate() })];
|
||||
});
|
||||
|
||||
// defaultActions carries the rest: the jump-to-violation button, the theme
|
||||
// toggle, the link back to the run list, and the scrolling.
|
||||
export const actionsRoot = weighted(
|
||||
[30, selectAStep],
|
||||
[20, navigateByKeyboard],
|
||||
[25, switchATab],
|
||||
[25, defaultActions],
|
||||
);
|
||||
@@ -65,6 +65,7 @@ export default function Tabs({ tabs, defaultTabId, ariaLabel }: TabsProps) {
|
||||
<div className="tabs">
|
||||
<div
|
||||
className="tabs-header"
|
||||
data-testid="tab-strip"
|
||||
role="tablist"
|
||||
aria-label={ariaLabel}
|
||||
onKeyDown={handleKeyDown}
|
||||
@@ -82,6 +83,8 @@ export default function Tabs({ tabs, defaultTabId, ariaLabel }: TabsProps) {
|
||||
}
|
||||
}}
|
||||
type="button"
|
||||
data-testid="tab"
|
||||
data-tab-id={tab.id}
|
||||
role="tab"
|
||||
id={`${panelId}-tab-${tab.id}`}
|
||||
className="tabs-tab"
|
||||
|
||||
@@ -170,6 +170,8 @@ export default function ActionList({
|
||||
}
|
||||
}}
|
||||
className="action-list-item"
|
||||
data-testid="step-row"
|
||||
data-step={step.index}
|
||||
role="option"
|
||||
tabIndex={isActive ? 0 : -1}
|
||||
aria-selected={isActive}
|
||||
|
||||
@@ -9,6 +9,12 @@ export interface ScreenshotProps {
|
||||
deviceHeight?: number;
|
||||
}
|
||||
|
||||
// stepOfSource reads the step out of the screenshot URL the panel is rendering,
|
||||
// so a spec can compare what this panel shows against what the toolbar claims.
|
||||
function stepOfSource(src: string): string | undefined {
|
||||
return /step-0*(\d+)\.png/.exec(src)?.[1];
|
||||
}
|
||||
|
||||
const DEFAULT_WIDTH = 1080;
|
||||
const DEFAULT_HEIGHT = 1920;
|
||||
|
||||
@@ -58,6 +64,8 @@ export default function Screenshot({ src, action, deviceWidth, deviceHeight }: S
|
||||
>
|
||||
<img
|
||||
className="screenshot-image"
|
||||
data-testid="screenshot"
|
||||
data-step={stepOfSource(src)}
|
||||
src={src}
|
||||
alt="device screenshot"
|
||||
onLoad={(event) => {
|
||||
|
||||
@@ -130,13 +130,18 @@ export default function ViolationsPanel({
|
||||
}
|
||||
|
||||
return (
|
||||
<section className="violations-panel">
|
||||
<section
|
||||
className="violations-panel"
|
||||
data-testid="violations-panel"
|
||||
data-violation-count={rows.filter((row) => row.status === "violated").length}
|
||||
>
|
||||
{violationsOnly ? null : (
|
||||
<header className="violations-panel-header">
|
||||
<h2 className="violations-panel-title">properties</h2>
|
||||
<button
|
||||
type="button"
|
||||
className="violations-panel-jump"
|
||||
data-testid="jump-to-first-violation"
|
||||
onClick={onJumpToFirstViolation}
|
||||
disabled={!hasFirstViolation}
|
||||
>
|
||||
@@ -149,7 +154,12 @@ export default function ViolationsPanel({
|
||||
const residual = residuals?.[name];
|
||||
const witness = status === "violated" ? witnesses?.[name] : undefined;
|
||||
return (
|
||||
<li key={name} className="violations-panel-row" data-status={status}>
|
||||
<li
|
||||
key={name}
|
||||
className="violations-panel-row"
|
||||
data-testid="property-row"
|
||||
data-status={status}
|
||||
>
|
||||
<div className="violations-panel-row-head">
|
||||
<span
|
||||
className="violations-panel-badge"
|
||||
|
||||
@@ -172,7 +172,7 @@ export default function RunDetail() {
|
||||
label: "Violations",
|
||||
badge:
|
||||
stepViolations.length > 0 ? (
|
||||
<span className="tabs-badge" data-kind="violation">
|
||||
<span className="tabs-badge" data-testid="violations-badge" data-kind="violation">
|
||||
{stepViolations.length}
|
||||
</span>
|
||||
) : undefined,
|
||||
@@ -239,7 +239,7 @@ export default function RunDetail() {
|
||||
label: "Violations",
|
||||
badge:
|
||||
stepViolations.length > 0 ? (
|
||||
<span className="tabs-badge" data-kind="violation">
|
||||
<span className="tabs-badge" data-testid="violations-badge" data-kind="violation">
|
||||
{stepViolations.length}
|
||||
</span>
|
||||
) : undefined,
|
||||
@@ -267,7 +267,11 @@ export default function RunDetail() {
|
||||
<span>
|
||||
<strong title={run.spec_path}>{basename(run.spec_path)}</strong> seed={run.seed}
|
||||
</span>
|
||||
<span>
|
||||
<span
|
||||
data-testid="step-indicator"
|
||||
data-step={stepIndex}
|
||||
data-step-count={stepCount ?? 0}
|
||||
>
|
||||
step {stepIndex} / {stepCount ?? 0}
|
||||
</span>
|
||||
</div>
|
||||
@@ -319,14 +323,14 @@ export default function RunDetail() {
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
<section className="detail-state-before detail-panel">
|
||||
<section className="detail-state-before detail-panel" data-testid="state-before">
|
||||
<h2>state before</h2>
|
||||
<div className="detail-panel-body">
|
||||
<Tabs tabs={beforeTabs} defaultTabId="screenshot" ariaLabel="state before" />
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section className="detail-state-after detail-panel">
|
||||
<section className="detail-state-after detail-panel" data-testid="state-after">
|
||||
<h2>state after</h2>
|
||||
<div className="detail-panel-body">
|
||||
<Tabs tabs={afterTabs} defaultTabId="screenshot" ariaLabel="state after" />
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
--text-muted: #6b6b6b;
|
||||
--accent-violation: #d32f2f;
|
||||
--accent-positive: #2e7d32;
|
||||
--font-mono: "JetBrains Mono", ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
|
||||
--font-mono: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
|
||||
}
|
||||
|
||||
:root[data-theme="dark"],
|
||||
|
||||
@@ -1,27 +1,3 @@
|
||||
@font-face {
|
||||
font-family: "JetBrains Mono";
|
||||
font-style: normal;
|
||||
font-weight: 400;
|
||||
font-display: swap;
|
||||
src: url("/fonts/JetBrainsMono-Regular.woff2") format("woff2");
|
||||
}
|
||||
|
||||
@font-face {
|
||||
font-family: "JetBrains Mono";
|
||||
font-style: normal;
|
||||
font-weight: 500;
|
||||
font-display: swap;
|
||||
src: url("/fonts/JetBrainsMono-Medium.woff2") format("woff2");
|
||||
}
|
||||
|
||||
@font-face {
|
||||
font-family: "JetBrains Mono";
|
||||
font-style: normal;
|
||||
font-weight: 700;
|
||||
font-display: swap;
|
||||
src: url("/fonts/JetBrainsMono-Bold.woff2") format("woff2");
|
||||
}
|
||||
|
||||
body {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 13px;
|
||||
|
||||
@@ -63,14 +63,82 @@ internal const val STABILITY_POLL_CAP_MILLIS = 2000L
|
||||
// missed.
|
||||
internal const val MIN_STABLE_STREAK_MILLIS = 750L
|
||||
|
||||
// TRANSITION_* bound the wait a snapshot pays when the tree it read shows two
|
||||
// routes at once, which on Compose means a NavHost cross-fade is in flight.
|
||||
// They are shorter than the constants above because this wait has one job,
|
||||
// outlasting the fade, where the poll above must also catch an async effect
|
||||
// that fires late.
|
||||
//
|
||||
// The streak matches the iOS companion's 300ms for the same reason: what this
|
||||
// wait is for is the cross-fade, and the cross-fade is caught by the
|
||||
// route-screen check rather than by streak length, so the streak only has to
|
||||
// be long enough that the reads spanning it are not all inside one frame.
|
||||
// MIN_STABLE_STREAK's 750ms is sized for a poll that must also catch an async
|
||||
// effect firing late, and at ~160ms per read-and-interval it costs three or
|
||||
// four more reads than this does.
|
||||
internal const val TRANSITION_STABLE_STREAK_MILLIS = 300L
|
||||
|
||||
// The interval is the iOS companion's rather than STABILITY_POLL_INTERVAL's
|
||||
// 250ms: the wide interval exists to stop a per-step poll hammering
|
||||
// UiAutomation, and this one runs only on the frames that show two routes,
|
||||
// about a quarter of steps on folio. The read itself paces the loop.
|
||||
internal const val TRANSITION_POLL_INTERVAL_MILLIS = 100L
|
||||
|
||||
// The cap is the iOS companion's 1500ms, and it has to be at least this: the
|
||||
// NavHost cross-fade is a 700ms tween (Compose navigation's default enter and
|
||||
// exit), it starts when the action lands rather than when the snapshot begins,
|
||||
// and the streak above has to fit after it. Measured on the emulator, a 600ms
|
||||
// cap left about a third of fades unfinished.
|
||||
//
|
||||
// A layout that holds two routes at rest pays the full cap on every snapshot
|
||||
// RPC it is read with, and the runner issues more than one: fetchSyncedState
|
||||
// (internal/runner) re-fetches a tree it considers transitional up to 4 times.
|
||||
// It stops early once two consecutive fetches come back byte-identical, so such
|
||||
// a layout costs 2 caps on a still tree and up to 4 on one that jitters
|
||||
// underneath. Per step, not once per step.
|
||||
internal const val TRANSITION_POLL_CAP_MILLIS = 1500L
|
||||
|
||||
// awaitSettledTree reads the hierarchy and, while the tree it gets back holds
|
||||
// more than one route, keeps reading until the cross-fade lands or the cap
|
||||
// expires. It returns the last tree read, so the caller gets the settled one
|
||||
// rather than paying for another read.
|
||||
//
|
||||
// Two nodes carrying the SAME route id are one destination, not a transition;
|
||||
// countRouteScreens counts distinct tags, so a screen that nests a repeat of
|
||||
// its own id does not pay this wait at all.
|
||||
internal fun awaitSettledTree(read: () -> String): String {
|
||||
var json = read()
|
||||
if (countRouteScreens(json) <= 1) return json
|
||||
pollUntilStable(
|
||||
TRANSITION_POLL_CAP_MILLIS,
|
||||
TRANSITION_STABLE_STREAK_MILLIS,
|
||||
TRANSITION_POLL_INTERVAL_MILLIS,
|
||||
) {
|
||||
json = read()
|
||||
stabilitySnapshot(json)
|
||||
}
|
||||
return json
|
||||
}
|
||||
|
||||
// pollUntilStable returns when the snapshot has been non-null and equal to
|
||||
// itself for an uninterrupted stretch of at least MIN_STABLE_STREAK_MILLIS,
|
||||
// capped at timeoutMillis. snapshot must omit transient attributes (e.g.
|
||||
// measure-pass bounds) so layout-only flicker doesn't extend the wait, and
|
||||
// must return null when the snapshot looks transitional (e.g. mid NavHost
|
||||
// cross-fade) so the streak resets and the loop keeps polling instead of
|
||||
// declaring a partial state stable.
|
||||
internal fun pollUntilStable(timeoutMillis: Long, snapshot: () -> String?) {
|
||||
// itself for an uninterrupted stretch of at least streakMillis, capped at
|
||||
// timeoutMillis. snapshot must omit transient attributes (e.g. measure-pass
|
||||
// bounds) so layout-only flicker doesn't extend the wait, and must return null
|
||||
// when the snapshot looks transitional (e.g. mid NavHost cross-fade) so the
|
||||
// streak resets and the loop keeps polling instead of declaring a partial
|
||||
// state stable.
|
||||
//
|
||||
// streakMillis is quiet the poll OBSERVED: the clock starts when the read that
|
||||
// first matched its predecessor returns, so the reads spanning it are not
|
||||
// charged to it. A hierarchy fetch on Android costs more than the poll
|
||||
// interval, and charging it would let a 500ms read clear the default 750ms
|
||||
// streak having watched 250ms of quiet.
|
||||
internal fun pollUntilStable(
|
||||
timeoutMillis: Long,
|
||||
streakMillis: Long = MIN_STABLE_STREAK_MILLIS,
|
||||
intervalMillis: Long = STABILITY_POLL_INTERVAL_MILLIS,
|
||||
snapshot: () -> String?,
|
||||
) {
|
||||
if (timeoutMillis <= 0) return
|
||||
val deadline = System.currentTimeMillis() + timeoutMillis
|
||||
var prior = try {
|
||||
@@ -80,7 +148,7 @@ internal fun pollUntilStable(timeoutMillis: Long, snapshot: () -> String?) {
|
||||
}
|
||||
var streakStart = 0L
|
||||
while (System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(STABILITY_POLL_INTERVAL_MILLIS)
|
||||
Thread.sleep(intervalMillis)
|
||||
val current = try {
|
||||
snapshot()
|
||||
} catch (_: Exception) {
|
||||
@@ -89,7 +157,7 @@ internal fun pollUntilStable(timeoutMillis: Long, snapshot: () -> String?) {
|
||||
val now = System.currentTimeMillis()
|
||||
if (prior != null && current != null && prior == current) {
|
||||
if (streakStart == 0L) streakStart = now
|
||||
if (now - streakStart >= MIN_STABLE_STREAK_MILLIS) return
|
||||
if (now - streakStart >= streakMillis) return
|
||||
} else {
|
||||
streakStart = 0L
|
||||
}
|
||||
@@ -119,36 +187,43 @@ private val ROUTE_TAG_KEYS = setOf(
|
||||
|
||||
private val jsonMapper = com.fasterxml.jackson.module.kotlin.jacksonObjectMapper()
|
||||
|
||||
// countRouteScreens counts DISTINCT route-level destination tags, not the nodes
|
||||
// carrying them. A screen that nests a node repeating its own route id puts two
|
||||
// tagged nodes in the tree while only one destination is on screen; counting
|
||||
// nodes would read that as a cross-fade that never ends, and the caller would
|
||||
// pay its full poll budget on every step of that screen and never settle.
|
||||
internal fun countRouteScreens(treeJson: String): Int {
|
||||
if (treeJson.isBlank()) return 0
|
||||
return try {
|
||||
val root = jsonMapper.readTree(treeJson)
|
||||
countRouteScreens(root)
|
||||
val tags = mutableSetOf<String>()
|
||||
collectRouteScreens(root, tags)
|
||||
tags.size
|
||||
} catch (_: Exception) {
|
||||
0
|
||||
}
|
||||
}
|
||||
|
||||
private fun countRouteScreens(
|
||||
private fun collectRouteScreens(
|
||||
node: com.fasterxml.jackson.databind.JsonNode,
|
||||
): Int {
|
||||
var count = 0
|
||||
into: MutableSet<String>,
|
||||
) {
|
||||
val attributes = node.get("attributes")
|
||||
if (attributes != null && attributes.isObject) {
|
||||
for (key in ROUTE_TAG_KEYS) {
|
||||
val value = attributes.get(key) ?: continue
|
||||
if (value.isNull) continue
|
||||
if (value.asText().endsWith("Screen")) {
|
||||
count++
|
||||
val text = value.asText()
|
||||
if (text.endsWith("Screen")) {
|
||||
into.add(text)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
val children = node.get("children")
|
||||
if (children != null && children.isArray) {
|
||||
for (child in children) count += countRouteScreens(child)
|
||||
for (child in children) collectRouteScreens(child, into)
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// structuralHash hashes a Maestro TreeNode-shaped JSON string by walking it
|
||||
@@ -873,13 +948,29 @@ class MaestroDriverBackend(private val serial: String?) : DriverBackend {
|
||||
override fun recentLogs(sinceUnixMillis: Long, minLevel: String) =
|
||||
readLogcat(serial, sinceUnixMillis, minLevel)
|
||||
|
||||
// snapshot waits out a NavHost cross-fade before it reads, so the runner is
|
||||
// never handed a tree holding two routes at once. It belongs here rather
|
||||
// than in waitForIdle: the runner gives waitForIdle a one-second deadline
|
||||
// and abandons the RPC when it expires, which is not enough room for a
|
||||
// 700ms fade that began before the settle did, and a wait that outlives the
|
||||
// deadline just races the runner's own fetch on the device-side server.
|
||||
// The snapshot RPC carries the step's deadline instead, so the wait can run
|
||||
// to a bound that actually covers the animation.
|
||||
//
|
||||
// The predicate costs nothing on a settled frame: the read it needs is the
|
||||
// read the snapshot was going to do anyway. That is what makes this
|
||||
// affordable, where the structural poll that used to run in waitForIdle was
|
||||
// not: it fetched the hierarchy ~4 more times on every mutating step.
|
||||
override fun snapshot(): SnapshotSample =
|
||||
SnapshotSample(awaitSettledTree { hierarchy() }, screenshot())
|
||||
|
||||
override fun waitForIdle(durationMillis: Long) {
|
||||
// waitForAppToSettle blocks on the View-system animation and maestro's
|
||||
// own structural settle, which is enough on its own. A follow-up
|
||||
// structural-hash poll used to run here, but each hierarchy fetch is
|
||||
// ~500ms on a physical device, so it cost ~2.8s per mutating step for
|
||||
// marginal benefit; the runner already re-fetches while a frame still
|
||||
// looks transitional.
|
||||
// own structural settle. It cannot see a Compose cross-fade: the fade
|
||||
// keeps both routes alive with the tree byte-identical, so a settle
|
||||
// that watches for change returns in the middle of one. That is what
|
||||
// snapshot above waits out; a structural poll here used to try, cost
|
||||
// ~2.8s per mutating step, and was removed.
|
||||
driver.waitForAppToSettle(null, null, durationMillis.toInt())
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
package dev.sanderling.sidecar
|
||||
|
||||
import org.junit.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
// The Android backend's snapshot waits out a NavHost cross-fade before it
|
||||
// reads. Without that wait the runner is handed a tree holding two routes at
|
||||
// once, refuses to act on it, and the step applies nothing; how many steps land
|
||||
// that way varies run to run, so the same seed walks a different trajectory
|
||||
// each time. These cover the wait (awaitSettledTree) and what it costs.
|
||||
class RouteTransitionTest {
|
||||
private fun screen(id: String, child: String = "") =
|
||||
"""{"attributes":{"resource-id":"$id"},"children":[$child]}"""
|
||||
|
||||
private fun tree(vararg children: String) =
|
||||
"""{"attributes":{"resource-id":"root"},"children":[${children.joinToString(",")}]}"""
|
||||
|
||||
private val crossFade = tree(screen("LedgerScreen"), screen("AddTransactionScreen"))
|
||||
private val landed = tree(screen("AddTransactionScreen"))
|
||||
|
||||
@Test fun waitsForTheCrossFadeToLandAndReturnsTheLandedTree() {
|
||||
// The first read caught both routes alive. The wait must keep reading
|
||||
// until only the destination is left, and hand back that tree: the
|
||||
// caller records what it returns, so returning the fade would put the
|
||||
// frame in the trace whether or not the wait happened.
|
||||
var reads = 0
|
||||
val settled = awaitSettledTree {
|
||||
reads++
|
||||
if (reads <= 3) crossFade else landed
|
||||
}
|
||||
assertTrue(reads > 3, "must keep reading until the fade lands, reads=$reads")
|
||||
assertEquals(1, countRouteScreens(settled), "must return a tree with one route")
|
||||
}
|
||||
|
||||
@Test fun settledFrameCostsExactlyOneRead() {
|
||||
// A hierarchy read is the expensive part of a step, and this one is the
|
||||
// read the snapshot was going to do anyway. A frame that is already on
|
||||
// one route must not pay for a second: that cost, on every step, is why
|
||||
// the unconditional structural poll was removed from waitForIdle.
|
||||
var reads = 0
|
||||
val settled = awaitSettledTree {
|
||||
reads++
|
||||
landed
|
||||
}
|
||||
assertEquals(1, reads, "a one-route frame must read once and return")
|
||||
assertEquals(landed, settled)
|
||||
}
|
||||
|
||||
@Test fun repeatedRouteIdIsOneScreenNotATransition() {
|
||||
// A screen nesting a node that repeats its own route id puts two tagged
|
||||
// nodes in the tree while one destination is on screen. Counting nodes
|
||||
// would read that as a fade that never ends: every step of that screen
|
||||
// would burn the whole poll budget and still hand over a frame the
|
||||
// runner refuses to act on.
|
||||
val nested = tree(screen("HomeScreen", screen("HomeScreen")))
|
||||
assertEquals(1, countRouteScreens(nested), "the same id twice is one route")
|
||||
var reads = 0
|
||||
awaitSettledTree {
|
||||
reads++
|
||||
nested
|
||||
}
|
||||
assertEquals(1, reads, "a repeated route id must not be treated as a transition")
|
||||
}
|
||||
|
||||
@Test fun aLayoutThatKeepsTwoRoutesIsBoundedByTheCap() {
|
||||
// Two routes alive at rest is a real layout, not a fade, and no amount
|
||||
// of waiting will resolve it. The wait is bounded by wall clock, so
|
||||
// such a screen costs the cap once per step and nothing more.
|
||||
val start = System.currentTimeMillis()
|
||||
val settled = awaitSettledTree { crossFade }
|
||||
val elapsed = System.currentTimeMillis() - start
|
||||
assertTrue(
|
||||
elapsed >= TRANSITION_POLL_CAP_MILLIS,
|
||||
"must actually wait out a two-route tree, elapsed=${elapsed}ms",
|
||||
)
|
||||
assertTrue(
|
||||
elapsed < TRANSITION_POLL_CAP_MILLIS + 1000L,
|
||||
"must stop at the cap, elapsed=${elapsed}ms",
|
||||
)
|
||||
assertEquals(crossFade, settled, "the caller still gets a tree to record")
|
||||
}
|
||||
|
||||
@Test fun capCoversTheNavHostFadePlusTheStreak() {
|
||||
// Compose navigation's default enter and exit are a 700ms tween, and
|
||||
// the fade starts when the action lands, not when the snapshot begins.
|
||||
// A cap that does not clear the fade and the streak after it hands the
|
||||
// caller a transitional frame, which is the whole defect.
|
||||
val fadeMillis = 700L
|
||||
val start = System.currentTimeMillis()
|
||||
val settled = awaitSettledTree {
|
||||
if (System.currentTimeMillis() - start < fadeMillis) crossFade else landed
|
||||
}
|
||||
val elapsed = System.currentTimeMillis() - start
|
||||
|
||||
assertEquals(landed, settled, "must hand back the landed tree, not the fade")
|
||||
assertTrue(
|
||||
elapsed >= fadeMillis,
|
||||
"cannot have settled before the fade ended, elapsed=${elapsed}ms",
|
||||
)
|
||||
assertTrue(
|
||||
elapsed < TRANSITION_POLL_CAP_MILLIS,
|
||||
"the ${TRANSITION_POLL_CAP_MILLIS}ms cap has to leave room for a ${fadeMillis}ms " +
|
||||
"fade and the ${TRANSITION_STABLE_STREAK_MILLIS}ms streak after it, but the " +
|
||||
"wait ran to the cap instead, elapsed=${elapsed}ms",
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -16,28 +16,68 @@ class StabilityPollTest {
|
||||
assertTrue(elapsed < 3000L, "should not run to cap when stable, elapsed=${elapsed}ms")
|
||||
}
|
||||
|
||||
@Test fun slowSnapshotReadsDoNotEatTheStreak() {
|
||||
// Every other test here uses an instant lambda and so passes whether or
|
||||
// not a read is charged to the streak; this one is the difference.
|
||||
// StubDriverBackend's waitForIdle polls a real `uiautomator dump`,
|
||||
// which costs hundreds of milliseconds, so the slow read is the case it
|
||||
// runs in.
|
||||
val readMillis = 400L
|
||||
val sampleStarts = mutableListOf<Long>()
|
||||
val sampleEnds = mutableListOf<Long>()
|
||||
val start = System.currentTimeMillis()
|
||||
pollUntilStable(5000L) {
|
||||
sampleStarts += System.currentTimeMillis() - start
|
||||
Thread.sleep(readMillis)
|
||||
sampleEnds += System.currentTimeMillis() - start
|
||||
"stable"
|
||||
}
|
||||
|
||||
// What the poll actually watched: the last read began this long after
|
||||
// the first one returned, and every sample in between matched.
|
||||
val observedQuiet = sampleStarts.last() - sampleEnds.first()
|
||||
assertTrue(
|
||||
observedQuiet >= MIN_STABLE_STREAK_MILLIS,
|
||||
"the poll returned having observed only ${observedQuiet}ms of quiet, not " +
|
||||
"${MIN_STABLE_STREAK_MILLIS}ms; starts=$sampleStarts ends=$sampleEnds",
|
||||
)
|
||||
assertTrue(
|
||||
sampleStarts.size >= 3,
|
||||
"a ${readMillis}ms read cannot clear the streak in one pair, starts=$sampleStarts",
|
||||
)
|
||||
}
|
||||
|
||||
@Test fun streakResetsOnAnyChange() {
|
||||
// A late transition that fires after the prior streak has already
|
||||
// begun must reset the clock: the post-transition stable window has
|
||||
// to start over and meet MIN_STABLE_STREAK_MILLIS from scratch.
|
||||
var calls = 0
|
||||
val start = System.currentTimeMillis()
|
||||
// First 4 samples are "calm", then 1 transient change, then "stable"
|
||||
// forever - the calm prefix is meaningless because of the transition.
|
||||
var transientAt = 0L
|
||||
// A short "calm" prefix, one transient change, then "stable" forever.
|
||||
// The prefix is meaningless because of the transition: only the streak
|
||||
// that starts after it can end the wait.
|
||||
pollUntilStable(3000L) {
|
||||
calls++
|
||||
when {
|
||||
calls <= 4 -> "calm"
|
||||
calls == 5 -> "transient"
|
||||
calls <= 2 -> "calm"
|
||||
calls == 3 -> {
|
||||
transientAt = System.currentTimeMillis()
|
||||
"transient"
|
||||
}
|
||||
else -> "stable"
|
||||
}
|
||||
}
|
||||
val elapsed = System.currentTimeMillis() - start
|
||||
val sinceTransition = System.currentTimeMillis() - transientAt
|
||||
assertTrue(
|
||||
elapsed >= MIN_STABLE_STREAK_MILLIS,
|
||||
"post-transition streak must reach ${MIN_STABLE_STREAK_MILLIS}ms, elapsed=${elapsed}ms",
|
||||
calls >= 8,
|
||||
"after the transition the poll needs a fresh matching pair and then a full " +
|
||||
"${MIN_STABLE_STREAK_MILLIS}ms of quiet, which is 8 samples, got $calls",
|
||||
)
|
||||
assertTrue(
|
||||
sinceTransition >= MIN_STABLE_STREAK_MILLIS,
|
||||
"the calm prefix must not count: a full ${MIN_STABLE_STREAK_MILLIS}ms streak has to " +
|
||||
"start over after the transition, returned ${sinceTransition}ms after it",
|
||||
)
|
||||
assertTrue(calls >= 10, "expected enough samples to span calm + transient + stable streak, got $calls")
|
||||
}
|
||||
|
||||
@Test fun transitionalNullsForceLoopToKeepWaiting() {
|
||||
|
||||
@@ -53,6 +53,30 @@ func TestBrowserCounterInvariantHolds(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestBrowserShadowDOMIsReachable drives a page whose entire UI (canvas, the
|
||||
// button over it, and the counter) lives inside a shadow root, the shape
|
||||
// Compose for Web produces. Both the enumeration and the selector lookup have
|
||||
// to cross the boundary for the counter to move at all, so the property firing
|
||||
// is the end-to-end evidence.
|
||||
func TestBrowserShadowDOMIsReachable(t *testing.T) {
|
||||
violations := runFixture(t, "shadow")
|
||||
if !slices.Contains(violations, "counterNeverMoves") {
|
||||
t.Fatalf("nothing inside the shadow root was ever tapped; violations=%v", violations)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBrowserAriaRoleIsTappable drives a page whose only controls are
|
||||
// <li role="option"> rows, the shape the replay UI gives its step list. The
|
||||
// tappable set covered role="button" and nothing else, so a spec had to
|
||||
// hand-write an action to reach a row; with the standard interactive roles
|
||||
// covered, the default tap enumeration finds them and the counter moves.
|
||||
func TestBrowserAriaRoleIsTappable(t *testing.T) {
|
||||
violations := runFixture(t, "aria-roles")
|
||||
if !slices.Contains(violations, "noRowWasEverSelected") {
|
||||
t.Fatalf("no role=\"option\" row was ever tapped; violations=%v", violations)
|
||||
}
|
||||
}
|
||||
|
||||
// runFixture serves the named testdata case over an in-process file server,
|
||||
// drives it through headless Chrome with a fixed seed and a bounded step count,
|
||||
// and returns every property name that was ever reported violated.
|
||||
@@ -180,3 +204,21 @@ func testdataDir(t *testing.T) string {
|
||||
func specSrcDir(t *testing.T) string {
|
||||
return filepath.Join(repoRoot(t), "pkg", "spec", "src")
|
||||
}
|
||||
|
||||
// TestBrowserUndefinedExtractorStaysUndefined drives the four layers of the
|
||||
// undefined reading through one run: the page wraps each reading in a {value}
|
||||
// envelope, the driver unwraps an absent value to an empty payload, the runner
|
||||
// checks the page reported one reading per extractor, and the verifier decodes
|
||||
// the empty payload as undefined. Each layer has its own unit test; only a run
|
||||
// proves they compose. Written straight into the map instead, an undefined
|
||||
// reading lost its whole index to JSON.stringify and that extractor silently
|
||||
// kept goja's dump-derived value while its neighbours held the page's.
|
||||
func TestBrowserUndefinedExtractorStaysUndefined(t *testing.T) {
|
||||
violations := runFixture(t, "undefined-extractor")
|
||||
if slices.Contains(violations, "undefinedStaysUndefined") {
|
||||
t.Error("a reading the page could not take did not reach the spec as undefined")
|
||||
}
|
||||
if !slices.Contains(violations, "counterNeverMoves") {
|
||||
t.Fatalf("nothing was ever tapped, so the property above held vacuously; violations=%v", violations)
|
||||
}
|
||||
}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
<!doctype html>
|
||||
<html>
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>aria roles</title>
|
||||
</head>
|
||||
<body>
|
||||
<!-- Shaped like the replay UI's step list: the rows are the only controls on
|
||||
the page, and they are plain <li> elements carrying role="option". A tap
|
||||
target set that reads element names and role="button" alone sees nothing
|
||||
here at all. -->
|
||||
<ol id="steps" role="listbox" aria-label="Steps">
|
||||
<li id="step-1" role="option">step 1</li>
|
||||
<li id="step-2" role="option">step 2</li>
|
||||
<li id="step-3" role="option">step 3</li>
|
||||
</ol>
|
||||
<div id="selected">0</div>
|
||||
<script>
|
||||
let selections = 0;
|
||||
const display = document.getElementById("selected");
|
||||
for (const row of document.querySelectorAll('[role="option"]')) {
|
||||
row.addEventListener("click", function () {
|
||||
selections += 1;
|
||||
display.textContent = String(selections);
|
||||
});
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
import { always, extract, taps } from "@sanderling/spec";
|
||||
|
||||
const selections = extract((s) => {
|
||||
const el = s.ax.find({ id: "selected" });
|
||||
return el ? parseInt(el.text, 10) || 0 : 0;
|
||||
}).named("selections");
|
||||
|
||||
// Every control on the page is an <li role="option">, the shape the replay UI
|
||||
// gives its own step rows. Nothing else can move this counter, so the spec
|
||||
// carries no action of its own: this property firing IS the evidence that the
|
||||
// default enumeration offered a tap on a role-based control.
|
||||
const noRowWasEverSelected = always(() => selections.current === 0);
|
||||
|
||||
export const properties = { noRowWasEverSelected };
|
||||
|
||||
export const actionsRoot = taps;
|
||||
+43
@@ -0,0 +1,43 @@
|
||||
<!doctype html>
|
||||
<html>
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>shadow</title>
|
||||
</head>
|
||||
<body>
|
||||
<!-- Shaped like Compose for Web: the app mounts a shadow root holding a
|
||||
canvas plus an accessibility overlay. The overlay is pointer-events:
|
||||
none, so a tap aimed at the button lands on the canvas, which routes it
|
||||
by coordinates the way a canvas app does. -->
|
||||
<div id="app"></div>
|
||||
<script>
|
||||
const root = document.getElementById("app").attachShadow({ mode: "open" });
|
||||
root.innerHTML = `
|
||||
<style>
|
||||
#surface { position: absolute; left: 0; top: 0; }
|
||||
#a11y { position: absolute; left: 0; top: 0; pointer-events: none; }
|
||||
#bump { margin: 20px; }
|
||||
</style>
|
||||
<canvas id="surface" width="360" height="240"></canvas>
|
||||
<div id="a11y">
|
||||
<button id="bump">bump</button>
|
||||
<div id="count">0</div>
|
||||
</div>`;
|
||||
const surface = root.getElementById("surface");
|
||||
const bump = root.getElementById("bump");
|
||||
const count = root.getElementById("count");
|
||||
let value = 0;
|
||||
surface.addEventListener("click", function (event) {
|
||||
const rect = bump.getBoundingClientRect();
|
||||
const hit =
|
||||
event.clientX >= rect.left &&
|
||||
event.clientX <= rect.right &&
|
||||
event.clientY >= rect.top &&
|
||||
event.clientY <= rect.bottom;
|
||||
if (!hit) return;
|
||||
value += 1;
|
||||
count.textContent = String(value);
|
||||
});
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
Vendored
+17
@@ -0,0 +1,17 @@
|
||||
import { always, extract, taps } from "@sanderling/spec";
|
||||
|
||||
const bumps = extract((s) => {
|
||||
const el = s.ax.find({ id: "count" });
|
||||
return el ? parseInt(el.text, 10) || 0 : 0;
|
||||
}).named("bumps");
|
||||
|
||||
// The button, the counter that records the taps, and the canvas that services
|
||||
// them all live inside a shadow root. A host that stops at the shadow boundary
|
||||
// enumerates no target to tap and resolves no selector to read, so the counter
|
||||
// can never leave zero: this property firing IS the evidence that both the
|
||||
// enumeration and the selector lookup crossed the boundary.
|
||||
const counterNeverMoves = always(() => bumps.current === 0);
|
||||
|
||||
export const properties = { counterNeverMoves };
|
||||
|
||||
export const actionsRoot = taps;
|
||||
@@ -0,0 +1,18 @@
|
||||
<!doctype html>
|
||||
<html>
|
||||
<body>
|
||||
<!-- One control and one counter. The spec's other two extractors look for
|
||||
ids that are not here, which is the shape a route-scoped extractor has
|
||||
on every step it is off its own screen. -->
|
||||
<div id="app">
|
||||
<button id="go" style="width:200px;height:80px">go</button>
|
||||
<div id="count">0</div>
|
||||
</div>
|
||||
<script>
|
||||
let taps = 0;
|
||||
document.getElementById("go").addEventListener("click", function () {
|
||||
document.getElementById("count").textContent = String(++taps);
|
||||
});
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,26 @@
|
||||
import { always, extract, taps } from "@sanderling/spec";
|
||||
|
||||
// Two extractors that resolve nothing, the shape every route-scoped extractor
|
||||
// has on the steps it is off its own screen (folio has nine of them). JSON has
|
||||
// no undefined, so these are what the {value} envelope exists for.
|
||||
const absent = extract((s) => s.ax.find({ id: "nothing-is-ever-here" })).named("absent");
|
||||
const absentText = extract((s) => s.ax.find({ id: "missing" })?.text).named("absentText");
|
||||
|
||||
const bumps = extract((s) => {
|
||||
const el = s.ax.find({ id: "count" });
|
||||
return el ? parseInt(el.text, 10) || 0 : 0;
|
||||
}).named("bumps");
|
||||
|
||||
// A reading the page could not take must arrive as undefined, not as null and
|
||||
// not as goja's reading of the dump. This property is what a spec sees.
|
||||
const undefinedStaysUndefined = always(
|
||||
() => absent.current === undefined && absentText.current === undefined,
|
||||
);
|
||||
|
||||
// The run has to actually be doing something, or the property above holds for
|
||||
// the wrong reason.
|
||||
const counterNeverMoves = always(() => bumps.current === 0);
|
||||
|
||||
export const properties = { undefinedStaysUndefined, counterNeverMoves };
|
||||
|
||||
export const actionsRoot = taps;
|
||||
Reference in new issue
Block a user