diff --git a/.github/scripts/replay-ui-summary-test.sh b/.github/scripts/replay-ui-summary-test.sh new file mode 100755 index 0000000..b75e988 --- /dev/null +++ b/.github/scripts/replay-ui-summary-test.sh @@ -0,0 +1,125 @@ +#!/usr/bin/env bash +# Drives replay-ui-summary.sh over the traces in testdata/ and checks the +# summary it renders. Run under the flags GitHub Actions uses for a `run:` +# block, because that is where a swallowed failure hides. +# +# testdata/replay-ui-real-run.jsonl is the first 10 steps of the dogfood run in +# actions run 31873049857 on master, with the per-step `hierarchy` dumps and the +# rowElements/tabElements extractors removed so the file stays readable. Nothing +# else was touched. That run was green, and badgeCountMatchesThePanel judged +# nothing in all 80 of its steps. The other two traces are written by hand. +set -euo pipefail + +here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +script="$here/replay-ui-summary.sh" +testdata="$here/testdata" +work="$(mktemp -d)" +trap 'rm -rf "$work"' EXIT + +failed=0 +rendered="" +status=0 + +summarise() { # + rendered="$work/$1.md" + : > "$rendered" + status=0 + GITHUB_STEP_SUMMARY="$rendered" SEED=3 MAX_STEPS=80 \ + bash -eo pipefail "$script" "$2" >/dev/null 2>"$work/$1.err" || status=$? +} + +fail() { + echo "FAIL: $*" >&2 + failed=1 +} + +expect_status() { # + [ "$status" = "$1" ] || fail "$2: exit $status, want $1" +} + +expect_line() { # + grep -qxF -- "$1" "$rendered" || fail "$2: summary has no line '$1'" +} + +expect_absent() { # + if grep -q -- "$1" "$rendered"; then fail "$2: summary should not mention '$1'"; fi +} + +plant() { # -> echoes the output dir + local dir="$work/$1/runs/20260815-075347" + mkdir -p "$dir" + cp "$testdata/$2" "$dir/trace.jsonl" + echo "$work/$1/runs" +} + +# A green run of the real thing. Every count here was measured, not chosen. +summarise real "$(plant real replay-ui-real-run.jsonl)" +expect_status 0 real +expect_line "- 10 steps recorded, 10 verified, 0 with violations" real +expect_line "| selectedStepIsInRange | 10 | 0 |" real +expect_line "| exactlyOneStepIsSelected | 10 | 0 |" real +expect_line "| stepCountMatchesTheList | 10 | 0 |" real +expect_line "| screenshotShowsTheSelectedStep | 10 | 0 |" real +expect_line "| switchingTabsKeepsTheStep | 2 | 8 |" real +expect_line "| badgeCountMatchesThePanel | **0** | 10 |" real +expect_line "- \`badgeCountMatchesThePanel\` judged nothing on this run: no step had a violations badge on screen" real +expect_absent "checked nothing" real + +# Every property judged at least once, including the one the real run never +# reached. Without this the counts above are consistent with a guard that can +# only ever return zero. +summarise every "$(plant every replay-ui-every-property.jsonl)" +expect_status 0 every +expect_line "- 4 steps recorded, 3 verified, 0 with violations" every +expect_line "| noUncaughtExceptions | 3 | 0 |" every +expect_line "| screenshotShowsTheSelectedStep | 3 | 0 |" every +expect_line "| switchingTabsKeepsTheStep | 2 | 1 |" every +expect_line "| badgeCountMatchesThePanel | 1 | 2 |" every + +# The fuzzer sat on the run list: the step page never rendered, so nothing was +# ever compared. This is the run that used to pass. +summarise blind "$(plant blind replay-ui-nothing-rendered.jsonl)" +expect_status 1 blind +expect_line "| selectedStepIsInRange | **0** | 4 |" blind +expect_line "| badgeCountMatchesThePanel | **0** | 4 |" blind +expect_line "- \`exactlyOneStepIsSelected\` judged nothing on this run: no step had a row in the step list" blind +grep -q "this run checked nothing" "$rendered" || fail "blind: no checked-nothing verdict" +grep -q "the step page never rendered" "$work/blind.err" || fail "blind: nothing on stderr" + +# The run directory exists but the trace does not, and the glob matches nothing +# at all. Both are the harness dying before it checked anything. +mkdir -p "$work/empty-run/runs/20260815-075347" +summarise empty-run "$work/empty-run/runs" +expect_status 1 empty-run +grep -q "no trace at" "$rendered" || fail "empty-run: no missing-trace line" + +summarise no-glob "$work/no-glob/runs" +expect_status 1 no-glob +grep -q "nothing under" "$rendered" || fail "no-glob: no missing-directory line" + +# A property this summary does not know about is a summary that silently counts +# six of seven properties, which is the bug one level up. +drifted="$(plant drift replay-ui-every-property.jsonl)" +sed 's/badgeCountMatchesThePanel/badgeAgreesWithThePanel/g' \ + "$testdata/replay-ui-every-property.jsonl" > "$drifted/20260815-075347/trace.jsonl" +summarise drift "$drifted" +expect_status 1 drift +grep -q "these counts cannot be trusted" "$rendered" || fail "drift: no untrusted verdict" + +# Same for a reading the spec no longer declares: the count would quietly go to +# zero and read as a UI that stopped rendering. +sed 's/extract("violationBadges"/extract("violationCounters"/' \ + "$here/../../replay-ui/sanderling/spec.ts" > "$work/renamed-spec.ts" +rendered="$work/renamed.md" +: > "$rendered" +status=0 +GITHUB_STEP_SUMMARY="$rendered" SPEC="$work/renamed-spec.ts" \ + bash -eo pipefail "$script" "$(plant renamed replay-ui-real-run.jsonl)" \ + >/dev/null 2>&1 || status=$? +expect_status 1 renamed +grep -q "no longer declares violationBadges" "$rendered" || fail "renamed: no drift verdict" + +if [ "$failed" = 0 ]; then + echo "replay-ui-summary.sh: ok" +fi +exit "$failed" diff --git a/.github/scripts/replay-ui-summary.sh b/.github/scripts/replay-ui-summary.sh new file mode 100755 index 0000000..af0cf4c --- /dev/null +++ b/.github/scripts/replay-ui-summary.sh @@ -0,0 +1,222 @@ +#!/usr/bin/env bash +# Reads the replay-ui dogfood trace and reports, per property, how many steps +# that property actually judged. Kept out of the workflow YAML so it can be run +# by hand against a local run: +# +# GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood +# +# `sanderling test` exiting 0 says only that no property returned false. Every +# property in replay-ui/sanderling/spec.ts declines to judge when a reading it +# needs is absent, which is right individually and useless in aggregate: a run +# that never rendered the step page returns false nowhere and exits 0, so +# checked-and-clean and checked-nothing arrive at the same green tick. Section 8 +# of docs/development/design-principles.md is the rule this leg was breaking. +set -euo pipefail + +root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +output="${1:-runs/dogfood}" +spec="${SPEC:-$root/replay-ui/sanderling/spec.ts}" +summary="${GITHUB_STEP_SUMMARY:-/dev/null}" + +shopt -s nullglob +run_dirs=("$output"/*/) +shopt -u nullglob + +{ + echo "### replay-ui dogfood" + echo + echo "- seed \`${SEED:-unset}\`, budget ${MAX_STEPS:-unset} steps" +} >> "$summary" + +if [ ${#run_dirs[@]} -eq 0 ]; then + echo "- **nothing under \`$output/\`**: the run never started, so no property was ever evaluated" >> "$summary" + echo "replay-ui: no run directory under $output/, so there is nothing to judge" >&2 + exit 1 +fi + +trace="${run_dirs[${#run_dirs[@]} - 1]}trace.jsonl" +if [ ! -f "$trace" ]; then + echo "- **no trace at \`$trace\`**: the run wrote no steps, so no property was ever evaluated" >> "$summary" + echo "replay-ui: $trace does not exist, so there is nothing to judge" >&2 + exit 1 +fi + +TRACE="$trace" SPEC="$spec" python3 - >> "$summary" <<'PY' +import json +import os +import re +import sys + +trace_path = os.environ["TRACE"] +spec_path = os.environ["SPEC"] + + +def toolbar(values): + reading = values.get("toolbar") + return reading if isinstance(reading, dict) else None + + +def numbered(reading, key): + return reading is not None and reading.get(key) is not None + + +def listing(values, name): + reading = values.get(name) + return reading if isinstance(reading, list) else [] + + +# One entry per property in replay-ui/sanderling/spec.ts, holding the guard that +# property opens with. The trace records extractor values, not verdicts, so +# counting the steps a property really judged means restating its guard here, +# and that restatement is the risk this file carries: a guard that drifts from +# its property would report evidence that does not exist. The two checks at the +# bottom make the two drifts that CAN be caught loud rather than silent. +# +# noUncaughtExceptions carries no guard on purpose: it compares a count, not an +# element, so it judges every step the verifier accepted. +GUARDS = { + "noUncaughtExceptions": [], + "selectedStepIsInRange": [ + ("a toolbar reporting a step and a step count", + lambda c, p: numbered(toolbar(c), "step") and numbered(toolbar(c), "stepCount")), + ], + "exactlyOneStepIsSelected": [ + ("a row in the step list", lambda c, p: len(listing(c, "stepRows")) > 0), + ], + "stepCountMatchesTheList": [ + ("a toolbar reporting a step count", lambda c, p: numbered(toolbar(c), "stepCount")), + ("a row in the step list", lambda c, p: len(listing(c, "stepRows")) > 0), + ], + "screenshotShowsTheSelectedStep": [ + ("a toolbar reporting a step", lambda c, p: numbered(toolbar(c), "step")), + ("a screenshot in the before panel", + lambda c, p: c.get("beforeScreenshotStep") is not None), + ], + "switchingTabsKeepsTheStep": [ + ("a tab selection that changed from the step before", + lambda c, p: p is not None and p.get("activeTabs") != c.get("activeTabs")), + ("a toolbar on both steps", + lambda c, p: p is not None and toolbar(p) is not None and toolbar(c) is not None), + ], + "badgeCountMatchesThePanel": [ + ("a violations badge on screen", + lambda c, p: len(listing(c, "violationBadges")) > 0 + and listing(c, "violationBadges")[0] is not None), + ("a violations panel on screen", + lambda c, p: len(listing(c, "violationPanelCounts")) > 0), + ], +} + +# A zero here is the run failing to render, not the fuzzer getting unlucky: +# every one of these needs only that the step page came up, which it does on the +# first step of any working run. The other three are left to be reported. Two of +# them are trajectory-dependent - switchingTabsKeepsTheStep needs a tab switch +# between consecutive steps, badgeCountMatchesThePanel needs the fuzzer to land +# on a violating step AND open the violations tab there - and a gate that +# convicts on an unlucky seed reports a regression it has not found. +MUST_RENDER = ( + "selectedStepIsInRange", + "exactlyOneStepIsSelected", + "stepCountMatchesTheList", + "screenshotShowsTheSelectedStep", +) + +steps = 0 +verified = 0 +violating = 0 +malformed = 0 +property_names = set() +judged = {name: 0 for name in GUARDS} +met_once = {name: [0] * len(conditions) for name, conditions in GUARDS.items()} + +# The trace records only the extractors that changed at a step, so an +# extractor's value at any step is the last change recorded for it, starting +# from null. Steps the verifier skipped advance nothing and are not evaluations. +values = {} +previous = None +with open(trace_path, encoding="utf-8", errors="replace") as lines: + for line in lines: + if not line.strip(): + continue + try: + step = json.loads(line) + except ValueError: + malformed += 1 + continue + steps += 1 + property_names |= set((step.get("residuals") or {}).keys()) + if step.get("violations"): + violating += 1 + if step.get("skipped_verification") or step.get("transitional"): + continue + for name, change in (step.get("extractor_changes") or {}).items(): + values[name] = change.get("curr") + verified += 1 + for name, conditions in GUARDS.items(): + met = [condition(values, previous) for _, condition in conditions] + if all(met): + judged[name] += 1 + for index, held in enumerate(met): + met_once[name][index] += 1 if held else 0 + previous = dict(values) + +report = [] +report.append("- %d steps recorded, %d verified, %d with violations" + % (steps, verified, violating)) +if malformed: + report.append("- **%d unreadable line(s)** in %s" % (malformed, trace_path)) +report.append("") +report.append("- judged: the property compared real values. declined: a reading it " + "needs was absent, so it returned true without checking anything.") +report.append("") +report.append("| property | judged | declined |") +report.append("| --- | --- | --- |") +for name in GUARDS: + count = judged[name] + report.append("| %s | %s | %d |" + % (name, count if count else "**0**", verified - count)) + +report.append("") +for name, conditions in GUARDS.items(): + if judged[name] or not verified: + continue + absent = [label for index, (label, _) in enumerate(conditions) if not met_once[name][index]] + report.append("- `%s` judged nothing on this run: no step had %s" + % (name, ", nor ".join(absent) if absent else "what its guard needs")) + +blind = None +if not verified: + blind = "%s records %d step(s) and not one of them was verified" % (trace_path, steps) +else: + silent = [name for name in MUST_RENDER if not judged[name]] + if silent: + blind = ("%s declined on every step, so the step page never rendered and the " + "exit code is not evidence about the replay UI" % ", ".join(silent)) + +drift = [] +if property_names and property_names != set(GUARDS): + drift.append("the spec's properties are %s but this summary counts %s" + % (", ".join(sorted(property_names)), ", ".join(sorted(GUARDS)))) +try: + with open(spec_path, encoding="utf-8") as handle: + declared = set(re.findall(r'extract\(\s*"([^"]+)"', handle.read())) +except OSError as error: + drift.append("could not read %s to check its readings still exist: %s" % (spec_path, error)) + declared = None +if declared is not None: + gone = sorted({"toolbar", "stepRows", "beforeScreenshotStep", "activeTabs", + "violationBadges", "violationPanelCounts"} - declared) + if gone: + drift.append("%s no longer declares %s, so the counts above describe readings " + "that do not exist" % (spec_path, ", ".join(gone))) + +if blind: + report.append("- **this run checked nothing**: %s" % blind) +for reason in drift: + report.append("- **these counts cannot be trusted**: %s" % reason) + +print("\n".join(report)) +for reason in ([blind] if blind else []) + drift: + print("replay-ui: %s" % reason, file=sys.stderr) +sys.exit(1 if blind or drift else 0) +PY diff --git a/.github/scripts/testdata/replay-ui-every-property.jsonl b/.github/scripts/testdata/replay-ui-every-property.jsonl new file mode 100644 index 0000000..145e5a4 --- /dev/null +++ b/.github/scripts/testdata/replay-ui-every-property.jsonl @@ -0,0 +1,4 @@ +{"extractor_changes":{"activeTabs":{"curr":"screenshot,screenshot","prev":null},"beforeScreenshotStep":{"curr":1,"prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false}],"prev":null},"toolbar":{"curr":{"step":1,"stepCount":5},"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":1,"timestamp":"2026-08-15T07:53:47Z"} +{"extractor_changes":{"activeTabs":{"curr":"violations,screenshot","prev":"screenshot,screenshot"},"beforeScreenshotStep":{"curr":4,"prev":1},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":true,"step":4,"violating":true},{"active":false,"step":5,"violating":false}],"prev":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false}]},"toolbar":{"curr":{"step":4,"stepCount":5},"prev":{"step":1,"stepCount":5}},"violationBadges":{"curr":[1],"prev":[]},"violationPanelCounts":{"curr":[1],"prev":[]}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":2,"timestamp":"2026-08-15T07:53:47Z"} +{"extractor_changes":{"activeTabs":{"curr":"hierarchy,screenshot","prev":"violations,screenshot"},"violationBadges":{"curr":[],"prev":[1]},"violationPanelCounts":{"curr":[],"prev":[1]}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":3,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"skipped_verification":true,"step":4,"timestamp":"2026-08-15T07:53:47Z"} diff --git a/.github/scripts/testdata/replay-ui-nothing-rendered.jsonl b/.github/scripts/testdata/replay-ui-nothing-rendered.jsonl new file mode 100644 index 0000000..9099693 --- /dev/null +++ b/.github/scripts/testdata/replay-ui-nothing-rendered.jsonl @@ -0,0 +1,4 @@ +{"extractor_changes":{"activeTabs":{"curr":"","prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[],"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":1,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":2,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":3,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":4,"timestamp":"2026-08-15T07:53:47Z"} diff --git a/.github/scripts/testdata/replay-ui-real-run.jsonl b/.github/scripts/testdata/replay-ui-real-run.jsonl new file mode 100644 index 0000000..bf7cf71 --- /dev/null +++ b/.github/scripts/testdata/replay-ui-real-run.jsonl @@ -0,0 +1,10 @@ +{"extractor_changes":{"activeTabs":{"curr":"screenshot,screenshot","prev":null},"beforeScreenshotStep":{"curr":1,"prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":null},"toolbar":{"curr":{"step":1,"stepCount":25},"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"metrics":{"cpu_percent":0,"heap_bytes":2771019,"total_memory_bytes":3893507},"next_action":{"kind":"Tap","selector":"data-testid:tab","tap_point":{"x":774,"y":69},"x":774,"y":69},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":1,"timestamp":"2026-08-15T07:53:47.883914163Z"} +{"extractor_changes":{"activeTabs":{"curr":"screenshot,hierarchy","prev":"screenshot,screenshot"}},"metrics":{"cpu_percent":0,"heap_bytes":3490520,"total_memory_bytes":6061916},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":2,"timestamp":"2026-08-15T07:53:48.356557191Z"} +{"extractor_changes":{"activeTabs":{"curr":"screenshot,properties","prev":"screenshot,hierarchy"},"violationPanelCounts":{"curr":[0],"prev":[]}},"metrics":{"cpu_percent":0,"heap_bytes":3737998,"total_memory_bytes":6595962},"next_action":{"kind":"Tap","resolved_bounds":{"height":109,"width":31,"x":337,"y":308},"selector":"desc:select step 12","tap_point":{"x":352,"y":362},"x":353,"y":363},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":3,"timestamp":"2026-08-15T07:53:48.584625848Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":12,"prev":1},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":12,"stepCount":25},"prev":{"step":1,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":3535304,"total_memory_bytes":7911280},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":4,"timestamp":"2026-08-15T07:53:48.820882198Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":13,"prev":12},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":true,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":13,"stepCount":25},"prev":{"step":12,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":4913839,"total_memory_bytes":8552983},"next_action":{"duration_millis":250,"from_x":98,"from_y":149,"kind":"Swipe","to_x":98,"to_y":749},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/13","step":5,"timestamp":"2026-08-15T07:53:49.038789589Z"} +{"metrics":{"cpu_percent":0,"heap_bytes":4615419,"total_memory_bytes":8970775},"next_action":{"key":"left","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/13","step":6,"timestamp":"2026-08-15T07:53:49.251047338Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":12,"prev":13},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":true,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":12,"stepCount":25},"prev":{"step":13,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":3996016,"total_memory_bytes":8970784},"next_action":{"kind":"Tap","selector":"data-testid:tab","tap_point":{"x":670,"y":69},"x":670,"y":69},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":7,"timestamp":"2026-08-15T07:53:49.451285818Z"} +{"metrics":{"cpu_percent":0,"heap_bytes":4830079,"total_memory_bytes":8974523},"next_action":{"kind":"Tap","selector":"data-testid:step-row","tap_point":{"x":170,"y":256},"x":170,"y":256},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":8,"timestamp":"2026-08-15T07:53:49.658417939Z"} +{"metrics":{"cpu_percent":0,"heap_bytes":3772499,"total_memory_bytes":8974523},"next_action":{"kind":"Tap","selector":"data-testid:step-row","tap_point":{"x":170,"y":363},"x":170,"y":363},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":9,"timestamp":"2026-08-15T07:53:49.865911674Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":6,"prev":12},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":true,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":6,"stepCount":25},"prev":{"step":12,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":4783792,"total_memory_bytes":8974552},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/6","step":10,"timestamp":"2026-08-15T07:53:50.075785262Z"} diff --git a/.github/workflows/folio.yml b/.github/workflows/folio.yml index d8d8c91..fa237b0 100644 --- a/.github/workflows/folio.yml +++ b/.github/workflows/folio.yml @@ -49,11 +49,15 @@ jobs: go-version-file: go.mod cache: true - - name: Set up JDK 17 + - name: Set up the JDKs uses: actions/setup-java@v4 with: distribution: temurin - java-version: "17" + # The metro gradle plugin folio builds with needs a 21 runtime; the + # sidecar toolchain pins 17. Both are installed so gradle can pick. + java-version: | + 17 + 21 - name: Set up Android SDK uses: android-actions/setup-android@v3 @@ -134,11 +138,15 @@ jobs: - name: Install idb-companion, xcodegen and just run: brew install facebook/fb/idb-companion xcodegen just - - name: Set up JDK 17 + - name: Set up the JDKs uses: actions/setup-java@v4 with: distribution: temurin - java-version: "17" + # The metro gradle plugin folio builds with needs a 21 runtime; the + # sidecar toolchain pins 17. Both are installed so gradle can pick. + java-version: | + 17 + 21 # The iOS app builds its Kotlin framework through the folio gradle # project, which configures :app:androidApp and so needs an Android SDK @@ -208,11 +216,15 @@ jobs: go-version-file: go.mod cache: true - - name: Set up JDK 17 + - name: Set up the JDKs uses: actions/setup-java@v4 with: distribution: temurin - java-version: "17" + # The metro gradle plugin folio builds with needs a 21 runtime; the + # sidecar toolchain pins 17. Both are installed so gradle can pick. + java-version: | + 17 + 21 - name: Set up bun uses: oven-sh/setup-bun@v2 diff --git a/.github/workflows/replay-ui.yml b/.github/workflows/replay-ui.yml index f4b27fd..e72d477 100644 --- a/.github/workflows/replay-ui.yml +++ b/.github/workflows/replay-ui.yml @@ -121,20 +121,14 @@ jobs: SEED: ${{ inputs.seed }} MAX_STEPS: ${{ inputs.max-steps }} + # Exit 0 above means no property returned false. It does not mean any + # property was ever evaluated against real content: they all decline to + # judge when the elements they read are absent, so a run that never + # rendered the step page is green and worthless. This step is what tells + # the two apart, and it fails the job when nothing was judged. - name: Summarise if: always() - run: | - { - echo "### replay-ui dogfood" - echo - echo "- seed \`$SEED\`, budget $MAX_STEPS steps" - for dir in runs/dogfood/*/; do - [ -f "$dir/trace.jsonl" ] || continue - steps=$(wc -l < "$dir/trace.jsonl" | tr -d ' ') - violations=$(grep -c '"violations":\[' "$dir/trace.jsonl" || true) - echo "- $steps steps recorded, $violations step(s) with violations" - done - } >> "$GITHUB_STEP_SUMMARY" + run: .github/scripts/replay-ui-summary.sh runs/dogfood env: SEED: ${{ inputs.seed }} MAX_STEPS: ${{ inputs.max-steps }} diff --git a/.gitignore b/.gitignore index 390b12e..fd41a17 100644 --- a/.gitignore +++ b/.gitignore @@ -50,7 +50,7 @@ internal/driver/ioscompanion/companionassets/assets/companion-*.tar.gz # before `go build -tags withcompanion`. Never commit: it's ~5 MB. internal/driver/ioscompanion/runnerassets/assets/runner-*.tar.gz -# spec-api compiled output +# spec package compiled output pkg/spec/dist/ # goreleaser local output diff --git a/.goreleaser.yaml b/.goreleaser.yaml index 1647206..e83ff41 100644 --- a/.goreleaser.yaml +++ b/.goreleaser.yaml @@ -5,10 +5,10 @@ project_name: sanderling before: hooks: # The Go binary embeds the sidecar fat JAR via //go:embed gated by the - # `withsidecar` build tag. Rebuild the JAR and stage it at the embed - # path so the `go build` below picks up fresh bytes. - - make sidecar - - sh -c 'mkdir -p internal/sidecar/assets && cp sidecar/build/libs/sidecar-all.jar internal/sidecar/assets/sidecar-all.jar' + # `withsidecar` build tag. The Makefile target rebuilds the JAR and stages + # it at the embed path, so the `go build` below picks up fresh bytes and + # that path stays spelled out in exactly one place. + - make sidecar-embed builds: - id: sanderling diff --git a/Makefile b/Makefile index f86a2cf..46ad520 100644 --- a/Makefile +++ b/Makefile @@ -29,7 +29,7 @@ WEB_DIST := replay-ui/dist GOLINES := $(shell $(GO) env GOPATH)/bin/golines -.PHONY: bootstrap proto sidecar sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift +.PHONY: bootstrap proto sidecar sidecar-embed sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api test-ci-scripts spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift bootstrap: $(GO) mod download @@ -42,6 +42,10 @@ proto: sidecar: $(SIDECAR_JAR) +# Stages the JAR where //go:embed expects it. Release tooling calls this rather +# than copying the JAR itself, so the embed path is spelled out in one place. +sidecar-embed: $(SIDECAR_EMBED) + sanderling: $(SANDERLING_BIN) $(SANDERLING_BIN): $(SIDECAR_EMBED) $(COMPANION_EMBED) $(RUNNER_EMBED) web-build @@ -117,7 +121,7 @@ fmt-ts: fmt-swift: xcrun swift-format format -i -r companion/Sources -test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test +test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test test-ci-scripts test-go: $(GO) test $(GO_PACKAGES) @@ -141,6 +145,11 @@ test-companion: $(COMPANION_EMBED) $(RUNNER_EMBED) test-kotlin: ANDROID_HOME=$(ANDROID_HOME) $(GRADLE) :sidecar:test +# The CI scripts that read a trace and decide whether a green leg is +# evidence. bash and python3 only, which is all a runner has. +test-ci-scripts: + .github/scripts/replay-ui-summary-test.sh + test-spec-api: cd pkg/spec && npm test --silent @@ -171,7 +180,7 @@ $(PAGE_OUT): build/site/%/index.html: docs/%.md $(DOCS_TEMPLATE) clean: $(GO) clean - rm -rf bin dist pkg/spec-api/dist build/site + rm -rf bin dist pkg/spec/dist build/site $(GRADLE) clean # Local release dry-runs. None of these touch remote registries. diff --git a/docs/development/ci.md b/docs/development/ci.md index db0e291..e82641c 100644 --- a/docs/development/ci.md +++ b/docs/development/ci.md @@ -136,6 +136,24 @@ hold for any trace and need no recalibrating when the fixture changes. The seventh is the stock `noUncaughtExceptions`, which asks nothing of the panels and only fails if the UI throws. Any violation fails the job. +So does a run that judged nothing. Exit 0 says no property returned false, which +is not the same as any property having been evaluated: each one declines to +judge when the elements it reads are absent, so a run where the trace failed to +serve, or where the fuzzer sat on the run list, renders nothing and passes. +`.github/scripts/replay-ui-summary.sh` reads the trace and puts a per-property +count of judged against declined steps in the job summary. Run it by hand with +`GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood`. + +It fails the job when any of the four properties that need nothing beyond the +step page having rendered - `selectedStepIsInRange`, `exactlyOneStepIsSelected`, +`stepCountMatchesTheList`, `screenshotShowsTheSelectedStep` - judged nothing at +all. The other three are reported and not gated, because a zero on them is a +seed getting unlucky rather than a broken leg: `switchingTabsKeepsTheStep` needs +a tab switch between consecutive steps, and `badgeCountMatchesThePanel` needs the +fuzzer to land on a violating step and open the violations tab in that same +step. On the first run measured this way (seed 3, 80 steps) that last one judged +nothing at all, so the fixture reaches it far too rarely to be worth gating on. + ## Reading a failure Both workflows upload their run directories as artifacts, and write the step diff --git a/docs/manual/getting-started.md b/docs/manual/getting-started.md index 7250ae3..5555db7 100644 --- a/docs/manual/getting-started.md +++ b/docs/manual/getting-started.md @@ -20,6 +20,8 @@ The spec package, in your project: npm install --save-dev @sanderling/spec ``` +Both come from the same release tag, and the CLI bundles the package's TypeScript sources when it evaluates your spec, so upgrade them together. Pre-releases are published under npm's `next` tag; `npm install @sanderling/spec` gives you the current stable one. + ## Check your environment ```sh diff --git a/docs/manual/spec-language.md b/docs/manual/spec-language.md index 5b52379..2c1532c 100644 --- a/docs/manual/spec-language.md +++ b/docs/manual/spec-language.md @@ -29,7 +29,7 @@ Every extractor callback receives a `State`: interface State { ax: AccessibilityTree; snapshots: Record; - lastAction: Action | null; + lastAction: (Action & { applied: true | null }) | null; logs: readonly LogEntry[]; exceptions: readonly ExceptionRecord[]; time: number; // ms since run start @@ -40,11 +40,13 @@ interface State { |---|---| | `ax` | Live UI hierarchy for this step | | `snapshots` | Key-value data pushed by the app SDK (empty if SDK not integrated) | -| `lastAction` | The action dispatched in the previous step, or `null` on the first step | +| `lastAction` | The action dispatched in the previous step, or `null` on the first step and on any step that dispatched nothing | | `logs` | Log entries collected since the previous step | | `exceptions` | Uncaught exceptions or `Sanderling.reportError()` calls since the previous step | | `time` | Milliseconds elapsed since the run started | +`lastAction.applied` is `true` when the runner saw the dispatch succeed and `null` when the apply call failed with the action possibly already delivered: an RPC deadline can fire after the tap reached the app, and nothing can find out afterwards. So there are three states, not two. `state.lastAction === null` means no action ran; `applied === null` means one ran whose fate is unknown. A property that attributes an effect to the action ("this submit must move the balance by the typed amount") has to decline unless `applied` is `true`, or a timeout convicts a healthy app. A property that counts what the app COULD have done should include it: an unconfirmed submit belongs in an upper bound on how many submits a window holds. + ## Selectors Selectors are passed to `ax.find()`, `ax.findAll()`, and element-scoped `.find()` / `.findAll()`. diff --git a/examples/folio/sanderling/predicates.ts b/examples/folio/sanderling/predicates.ts index e7ea34b..ab93c60 100644 --- a/examples/folio/sanderling/predicates.ts +++ b/examples/folio/sanderling/predicates.ts @@ -123,10 +123,22 @@ export function readHomeCards(args: { return { value: reading, carrier: reading, fresh: true }; } -function isTapOn( - lastAction: { kind?: string; on?: string | object } | null, - target: string, -): boolean { +// state.lastAction as the two hosts build it (internal/verifier/marshal.go +// lastActionFields), read defensively: every field is what a Go struct decided +// to emit, not something this file can trust a compile-time shape for. +// +// `applied` is true when the runner saw the action's dispatch succeed and null +// when the apply call failed with the gesture possibly already delivered: an +// RPC deadline can fire after the tap landed, and nothing can find out +// afterwards. That is unknown, not "it did not happen", which is what +// `lastAction === null` says. +export interface ObservedAction { + kind?: string; + on?: string | object; + applied?: true | null; +} + +function isTapOn(lastAction: ObservedAction | null, target: string): boolean { if (lastAction == null) return false; if (lastAction.kind !== "Tap" && lastAction.kind !== "DoubleTap") return false; const on = lastAction.on; @@ -138,19 +150,27 @@ function isTapOn( // Repository.createTransaction: AddTransactionViewModel.submit() is the only // caller, AddTransactionEvent.Submit is the only thing that runs it, and the // TxnSubmit button's onClick is the only thing that sends that event. -export function isTxnSubmitTap( - lastAction: { kind?: string; on?: string | object } | null, -): boolean { +export function isTxnSubmitTap(lastAction: ObservedAction | null): boolean { return isTapOn(lastAction, "TxnSubmit"); } // Likewise the only action that creates an account. -export function isAddAccountSubmitTap( - lastAction: { kind?: string; on?: string | object } | null, -): boolean { +export function isAddAccountSubmitTap(lastAction: ObservedAction | null): boolean { return isTapOn(lastAction, "AddAccountSubmit"); } +// Did the runner see this action's dispatch succeed? `applied` is null when the +// apply call failed with the gesture possibly already delivered (an RPC +// deadline can fire after the tap landed), and the runner has no way to find +// out afterwards. Such an action may have caused anything the next reading +// shows, so it counts toward how many submits a window COULD hold, but it never +// licenses attributing an effect to it: a property that demands the effect of +// an action that may never have run convicts the app of the runner's own +// uncertainty. +export function confirmedApplied(lastAction: ObservedAction | null): boolean { + return lastAction != null && lastAction.applied === true; +} + // Counts the submit actions inside the window the balance property compares // over: from the last Home total we read to this step, inclusive of this step's // action. @@ -164,9 +184,15 @@ export function isAddAccountSubmitTap( // // The reset lands on `fresh`, the same event that advances the carrier, so the // count always describes exactly the interval the two compared totals span. +// +// A submit whose dispatch the runner could not confirm counts here, because +// this number is an upper bound on the submits the window holds and the tap may +// well have landed. Leaving it out is what convicted a healthy app: +// committedTransactionsExceedSubmits saw a transaction rise of one against a +// window of zero and called it a double submit. export function countSubmitsInWindow(args: { previousCount: number; - lastAction: { kind?: string; on?: string | object } | null; + lastAction: ObservedAction | null; fresh: boolean; }): { reported: number; next: number } { const { previousCount, lastAction, fresh } = args; @@ -343,7 +369,7 @@ export function homeTxnCountsOf(cards: readonly CardReading[]): Record s.ax.find({ "resource-id": "TxnAmountField" })); +globalThis.properties = { + noAmountField: always(() => amountField.current === undefined), +}; +globalThis.actions = actions(() => []); +` + +const amountFieldTreeJSON = `{ + "attributes": {"resource-id": "root", "bounds": "[0,0,400,800]"}, + "enabled": true, + "children": [ + {"attributes": {"resource-id": "TxnAmountField", "text": "199", "bounds": "[0,100,400,160]"}, + "editable": true, "enabled": true, "children": []} + ] +}` + +// TestRunner_TraceRecordsElementValuedExtractors is the guard on the artifact a +// person opens to decide whether a conviction is real. An element-valued +// extractor used to reach the trace as null on the goja hosts (ios, android): +// its exported value carries the element's find/findAll host functions, which +// json.Marshal refuses, so the encoding failed and both the per-step diff and +// the witness recorded nothing. A witness that reads null for the field the +// property fired on describes a state the property could not have fired in, +// which is worse than a blank. +func TestRunner_TraceRecordsElementValuedExtractors(t *testing.T) { + state := newHarnessWithSpec(t, elementExtractorSpec) + state.mock.HierarchyJSON = amountFieldTreeJSON + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: state.mock, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if !containsProperty(summary.Violations, "noAmountField") { + t.Fatalf("noAmountField did not violate, so the element never reached a predicate: %v", + summary.Violations) + } + + file, err := os.Open(filepath.Join(state.writer.Directory(), "trace.jsonl")) + if err != nil { + t.Fatal(err) + } + defer file.Close() + + type traceLine struct { + Step int `json:"step"` + ExtractorChanges map[string]trace.ExtractorChange `json:"extractor_changes"` + Witnesses map[string]trace.Witness `json:"witnesses"` + } + changes, witnesses := 0, 0 + scanner := bufio.NewScanner(file) + scanner.Buffer(make([]byte, 0, 64*1024), 8*1024*1024) + for scanner.Scan() { + var line traceLine + if err := json.Unmarshal(scanner.Bytes(), &line); err != nil { + t.Fatalf("trace line decode: %v", err) + } + if change, ok := line.ExtractorChanges["amountField"]; ok { + changes++ + assertAmountField(t, fmt.Sprintf("step %d extractor_changes", line.Step), change.Curr) + } + for name, witness := range line.Witnesses { + witnesses++ + assertAmountField(t, fmt.Sprintf("step %d %s witness", line.Step, name), + witness.Extractors["amountField"]) + } + } + if err := scanner.Err(); err != nil { + t.Fatalf("scan trace: %v", err) + } + if changes == 0 { + t.Error("amountField never appears in extractor_changes; the element the run read is not in the trace") + } + if witnesses == 0 { + t.Error("no witness reached the trace; nothing was compared") + } +} + +// assertAmountField reads the recorded element the way a person opening the +// trace would: the field's text is the number the property was judged on. +func assertAmountField(t *testing.T, where string, recorded json.RawMessage) { + t.Helper() + var element struct { + Text string `json:"text"` + } + if err := json.Unmarshal(recorded, &element); err != nil { + t.Fatalf("%s: decode %s: %v", where, recorded, err) + } + if element.Text != "199" { + t.Errorf("%s: recorded element is %s, want its text to read 199", where, recorded) + } +} diff --git a/internal/runner/runner.go b/internal/runner/runner.go index d730556..1ef8d5e 100644 --- a/internal/runner/runner.go +++ b/internal/runner/runner.go @@ -298,11 +298,18 @@ func Run(ctx context.Context, options Options) (Summary, error) { logger.Warn("apply error; marking step transitional", "step", stepIndex, "err", err) transitional = true applySkipped = true - lastAction = nil + // The error says the call failed, not that the gesture never + // reached the app: a deadline that fires after dispatch leaves + // the effect committed. Reporting no action here would let a + // property convict the app for an effect with no cause, so the + // action is reported with its fate unknown instead. + unconfirmed := nextAction + lastAction = &unconfirmed } else { consecutiveApplyFailures = 0 - actionCopy := nextAction - lastAction = &actionCopy + applied := nextAction + applied.Applied = true + lastAction = &applied } } else { lastAction = nil diff --git a/internal/runner/uncertain_last_action_test.go b/internal/runner/uncertain_last_action_test.go new file mode 100644 index 0000000..f687e99 --- /dev/null +++ b/internal/runner/uncertain_last_action_test.go @@ -0,0 +1,150 @@ +package runner + +import ( + "context" + "errors" + "fmt" + "path/filepath" + "sync/atomic" + "testing" + "time" + + "github.com/priyanshujain/sanderling/internal/driver" + mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock" +) + +// An apply error is not proof that nothing landed. An RPC deadline that fires +// after the tap was dispatched leaves the transaction committed, and a runner +// that reports "no action" for it hands +// submitCommitsOneTransactionPerAction a rise of one transaction against a +// window of zero submits: a conviction manufactured out of the runner's own +// uncertainty, on the property carrying most of the detection on android. +// +// The spec below is the real folio predicate pair, imported from the example, +// so what this asserts is the verdict the shipped property reaches. +const uncertainApplySpecTemplate = ` +import { actions, always, extract, next, Tap } from "@sanderling/spec"; +import { + committedTransactionsExceedSubmits, + countSubmitsInWindow, +} from "%s"; + +let submits = 0; +const submitsInWindow = extract("submitsInWindow", state => { + const window = countSubmitsInWindow({ + previousCount: submits, + lastAction: state.lastAction, + fresh: true, + }); + submits = window.next; + return window.reported; +}); + +const counts = extract("counts", state => { + const text = state.ax.find("id:TxnCount")?.text; + return text ? { Travel: parseInt(text, 10) } : null; +}); + +globalThis.properties = { + submitCommitsOneTransactionPerAction: always( + next(() => + !committedTransactionsExceedSubmits({ + countsBefore: counts.previous ?? null, + countsAfter: counts.current, + submitsInWindow: submitsInWindow.current, + }), + ), + ), +}; +globalThis.actions = actions(() => [Tap({ on: "id:TxnSubmit" })]); +` + +const homeWithTxnCount = `{"attributes":{"resource-id":"HomeScreen"},"children":[ + {"attributes":{"resource-id":"TxnCount","text":"%d"},"children":[]}, + {"attributes":{"resource-id":"TxnSubmit","bounds":"[40,80,240,160]"},"children":[],"clickable":true,"enabled":true} +]}` + +// dispatchThenFailDriver is the device condition the runner cannot see through: +// the tap reaches the app and commits, then the call the runner is waiting on +// times out. Every later hierarchy read shows the committed transactions. +type dispatchThenFailDriver struct { + *mockdriver.Driver + commitsPerTap int64 + committed atomic.Int64 +} + +func (d *dispatchThenFailDriver) Tap(context.Context, int, int) error { + return d.dispatchThenFail() +} + +func (d *dispatchThenFailDriver) TapSelector(context.Context, string) error { + return d.dispatchThenFail() +} + +func (d *dispatchThenFailDriver) dispatchThenFail() error { + d.committed.Add(d.commitsPerTap) + return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded") +} + +func (d *dispatchThenFailDriver) Snapshot(context.Context) (string, driver.Image, error) { + return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil +} + +func TestRunner_ApplyErrorAfterDispatchDoesNotConvictTheSubmitCountingProperty(t *testing.T) { + predicates, err := filepath.Abs("../../examples/folio/sanderling/predicates.ts") + if err != nil { + t.Fatal(err) + } + spec := fmt.Sprintf(uncertainApplySpecTemplate, predicates) + + run := func(t *testing.T, commitsPerTap int64) []ViolationRecord { + t.Helper() + state := newHarnessWithSpec(t, spec) + device := &dispatchThenFailDriver{Driver: state.mock, commitsPerTap: commitsPerTap} + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + summary, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: device, + Verifier: state.verifier, + TraceWriter: state.writer, + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + if summary.Steps != 2 { + t.Fatalf("steps = %d, want 2; the run never reached the step that judges the pair", summary.Steps) + } + if got := device.committed.Load(); got != commitsPerTap*2 { + t.Fatalf("the device committed %d transaction(s), want %d; the taps never reached it", + got, commitsPerTap*2) + } + return summary.Violations + } + + t.Run("one transaction per tap is not a double submit", func(t *testing.T) { + if violations := run(t, 1); len(violations) != 0 { + t.Errorf("the counting property convicted a healthy app: %v\n"+ + "one transaction rose against a submit the runner dispatched but "+ + "could not confirm, and the spec was told no action happened", + violations) + } + }) + + // The control. Without it a green above proves nothing: a property that + // never sees a comparable pair is silently vacuous and reports the same + // empty violation list. + t.Run("two transactions per tap still convicts", func(t *testing.T) { + violations := run(t, 2) + if len(violations) == 0 { + t.Fatal("the counting property missed a double submit; the harness never " + + "put the property in a position to fire, so the case above proves nothing") + } + if violations[0].Properties[0] != "submitCommitsOneTransactionPerAction" { + t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", violations[0].Properties) + } + }) +} diff --git a/internal/runner/web_last_action_test.go b/internal/runner/web_last_action_test.go index 22ea458..e3a7ecc 100644 --- a/internal/runner/web_last_action_test.go +++ b/internal/runner/web_last_action_test.go @@ -3,6 +3,7 @@ package runner import ( "context" "encoding/json" + "errors" "testing" "time" @@ -71,7 +72,47 @@ func TestRunner_WebInstallsLastActionInThePage(t *testing.T) { // Every later step carries what the runner actually applied. The shape is // the goja host's (internal/verifier/marshal.go lastActionFields), pinned // against it by TestLastAction_WebJSONMatchesTheGojaObject. - const want = `{"kind":"Tap","on":"id:TxnSubmit"}` + const want = `{"kind":"Tap","applied":true,"on":"id:TxnSubmit"}` + if web.installed[1] != want { + t.Errorf("step 2 installed %s, want %s", web.installed[1], want) + } +} + +// failingTapWebDriver dispatches the tap and then fails the call, the shape an +// RPC deadline takes: the page has the click, the runner has an error. +type failingTapWebDriver struct { + *tappingWebDriver +} + +func (d *failingTapWebDriver) Tap(context.Context, int, int) error { + return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded") +} + +// The web leg of the same three states the goja host reports. "applied":null is +// not "no action": a property gated on the last action still sees the tap and +// decides for itself, which it cannot do if the page is handed a bare null. +func TestRunner_WebInstallsAnUnconfirmedActionWithItsFateUnknown(t *testing.T) { + state := newHarnessWithSpec(t, lastActionSpec) + web := &failingTapWebDriver{tappingWebDriver: &tappingWebDriver{Driver: state.mock}} + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if _, err := Run(ctx, Options{ + Duration: time.Hour, + IdleTimeout: 20 * time.Millisecond, + MaxSteps: 2, + Driver: web, + Verifier: state.verifier, + TraceWriter: state.writer, + }); err != nil { + t.Fatalf("Run: %v", err) + } + + if len(web.installed) < 2 { + t.Fatalf("the page was handed lastAction %d time(s); the web path never installed it", + len(web.installed)) + } + const want = `{"kind":"Tap","applied":null,"on":"id:TxnSubmit"}` if web.installed[1] != want { t.Errorf("step 2 installed %s, want %s", web.installed[1], want) } diff --git a/internal/testrun/testrun.go b/internal/testrun/testrun.go index af57b28..893e5ce 100644 --- a/internal/testrun/testrun.go +++ b/internal/testrun/testrun.go @@ -350,16 +350,20 @@ func resolveRuntimeSibling(specAPIPath, userSpecPath, filename string) string { return "" } -// resolveSpecAPIPath returns the path to pkg/spec/src/index.ts inside -// a sanderling source checkout, searched upward from the spec file and the cwd. -// Returns "" when not found, in which case esbuild resolves @sanderling/spec via -// node_modules the way a downstream user's project would. +// resolveSpecAPIPath returns the path to the spec API's index.ts: a sanderling +// source checkout first, searched upward from the spec file and the cwd, then +// an installed node_modules/@sanderling/spec. Aliasing the installed copy is +// what keeps the spec and the runtime entry on one module graph; resolving the +// bare specifier through package.json "exports" would load dist/ alongside the +// runtime's src/ and give sampler-rng.ts two instances. func resolveSpecAPIPath(specPath string) string { - var candidates []string + var checkout, installed []string if absoluteSpec, err := filepath.Abs(specPath); err == nil { directory := filepath.Dir(absoluteSpec) for { - candidates = append(candidates, filepath.Join(directory, "pkg/spec/src/index.ts")) + checkout = append(checkout, filepath.Join(directory, "pkg/spec/src/index.ts")) + installed = append(installed, + filepath.Join(directory, "node_modules/@sanderling/spec/src/index.ts")) parent := filepath.Dir(directory) if parent == directory { break @@ -368,9 +372,9 @@ func resolveSpecAPIPath(specPath string) string { } } if cwd, err := os.Getwd(); err == nil { - candidates = append(candidates, filepath.Join(cwd, "pkg/spec/src/index.ts")) + checkout = append(checkout, filepath.Join(cwd, "pkg/spec/src/index.ts")) } - for _, candidate := range candidates { + for _, candidate := range append(checkout, installed...) { if _, err := os.Stat(candidate); err == nil { return candidate } diff --git a/internal/testrun/testrun_test.go b/internal/testrun/testrun_test.go index bc3f376..fd6303e 100644 --- a/internal/testrun/testrun_test.go +++ b/internal/testrun/testrun_test.go @@ -2,7 +2,9 @@ package testrun import ( "context" + "encoding/json" "errors" + "io/fs" "os" "path/filepath" "strings" @@ -304,3 +306,153 @@ func TestLaunchAppBoundsWedgedDriver(t *testing.T) { t.Fatal("launchApp never returned: the pre-run launch is unbounded, so a wedged driver hangs the run forever") } } + +// repoFile walks up from the test's working directory and returns the absolute +// path of rel inside the sanderling checkout. +func repoFile(t *testing.T, rel string) string { + t.Helper() + directory, err := os.Getwd() + if err != nil { + t.Fatal(err) + } + for { + candidate := filepath.Join(directory, rel) + if _, err := os.Stat(candidate); err == nil { + return candidate + } + parent := filepath.Dir(directory) + if parent == directory { + t.Fatalf("%s not found above the test directory", rel) + } + directory = parent + } +} + +// publishedFiles returns the "files" entries of pkg/spec/package.json, the +// exact set npm ships in the @sanderling/spec tarball. +func publishedFiles(t *testing.T) []string { + t.Helper() + raw, err := os.ReadFile(repoFile(t, "pkg/spec/package.json")) + if err != nil { + t.Fatal(err) + } + var manifest struct { + Files []string `json:"files"` + } + if err := json.Unmarshal(raw, &manifest); err != nil { + t.Fatal(err) + } + return manifest.Files +} + +// installPublishedPackage reproduces what `npm install @sanderling/spec` +// unpacks into node_modules: only the paths package.json publishes. +func installPublishedPackage(t *testing.T, dest string) { + t.Helper() + specDir := filepath.Dir(repoFile(t, "pkg/spec/package.json")) + for _, entry := range publishedFiles(t) { + source := filepath.Join(specDir, entry) + if _, err := os.Stat(source); err != nil { + continue + } + copyTree(t, source, filepath.Join(dest, entry)) + } +} + +func copyTree(t *testing.T, source, dest string) { + t.Helper() + err := filepath.WalkDir(source, func(path string, entry fs.DirEntry, err error) error { + if err != nil { + return err + } + relative, err := filepath.Rel(source, path) + if err != nil { + return err + } + target := filepath.Join(dest, relative) + if entry.IsDir() { + return os.MkdirAll(target, 0o755) + } + data, err := os.ReadFile(path) + if err != nil { + return err + } + if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil { + return err + } + return os.WriteFile(target, data, 0o644) + }) + if err != nil { + t.Fatal(err) + } +} + +// TestResolveRuntimeSibling_PublishedPackageShipsTheRuntimes pins npm's "files" +// list against the resolver that consumes it. The tarball shipped dist/ alone +// while the node_modules fallback looks for src/goja-runtime.ts, so every +// `npm install @sanderling/spec` user hit "goja-runtime.ts not found". +func TestResolveRuntimeSibling_PublishedPackageShipsTheRuntimes(t *testing.T) { + root := t.TempDir() + installPublishedPackage(t, filepath.Join(root, "node_modules", "@sanderling", "spec")) + specPath := filepath.Join(root, "spec.ts") + if err := os.WriteFile(specPath, []byte(""), 0o644); err != nil { + t.Fatal(err) + } + + for _, filename := range []string{"goja-runtime.ts", "web-runtime.ts"} { + if resolveRuntimeSibling("", specPath, filename) == "" { + t.Errorf("%s unreachable from a published install; package.json publishes %v", + filename, publishedFiles(t)) + } + } +} + +// TestPrepareBundleInputs_InstalledPackageSharesOneModuleGraph pins the +// downstream case: with no sanderling checkout above the spec, the aliases and +// the runtime entry must name the SAME installed copy. An unset alias let +// esbuild resolve @sanderling/spec to dist/ while the runtime came from src/, +// which loads sampler-rng.ts twice; from(), strings(), integers() and emails() +// then read an rng the picker never set and collapse to a fixed default. +func TestPrepareBundleInputs_InstalledPackageSharesOneModuleGraph(t *testing.T) { + root := t.TempDir() + installed := filepath.Join(root, "node_modules", "@sanderling", "spec") + installPublishedPackage(t, installed) + specPath := filepath.Join(root, "sanderling", "spec.ts") + if err := os.MkdirAll(filepath.Dir(specPath), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(specPath, []byte(""), 0o644); err != nil { + t.Fatal(err) + } + + cwd, err := os.Getwd() + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = os.Chdir(cwd) }) + if err := os.Chdir(root); err != nil { + t.Fatal(err) + } + + prep, err := prepareBundleInputs(Options{Spec: specPath}) + if err != nil { + t.Fatal(err) + } + source := filepath.Join(installed, "src") + want := map[string]string{ + "@sanderling/spec": filepath.Join(source, "index.ts"), + "@sanderling/spec/defaults": filepath.Join(source, "defaults/index.ts"), + "@sanderling/spec/defaults/properties": filepath.Join(source, "defaults/properties.ts"), + } + for key, wantValue := range want { + if prep.aliases[key] != wantValue { + t.Errorf("alias %q = %q, want %q", key, prep.aliases[key], wantValue) + } + } + if got := prep.gojaRuntimePath; got != filepath.Join(source, "goja-runtime.ts") { + t.Errorf("gojaRuntimePath = %q, want it beside the aliased index.ts", got) + } + if got := resolveWebRuntimePath(prep.specAPIPath, specPath); got != filepath.Join(source, "web-runtime.ts") { + t.Errorf("webRuntimePath = %q, want it beside the aliased index.ts", got) + } +} diff --git a/internal/verifier/extractor_encoding_test.go b/internal/verifier/extractor_encoding_test.go new file mode 100644 index 0000000..a693dbf --- /dev/null +++ b/internal/verifier/extractor_encoding_test.go @@ -0,0 +1,179 @@ +package verifier + +import ( + "bytes" + "encoding/json" + "testing" +) + +const elementTreeJSON = `{ + "attributes": {"resource-id": "root", "bounds": "[0,0,400,800]"}, + "enabled": true, + "children": [ + {"attributes": {"resource-id": "TxnAmountField", "text": "199", "bounds": "[0,100,400,160]"}, + "editable": true, "enabled": true, "children": []} + ] +}` + +const elementExtractorSpec = ` +const field = __sanderling__.extract(state => state.ax.find({ "resource-id": "TxnAmountField" }), "field"); +globalThis.properties = {}; +` + +// canonicalElement is the trace's record of one ax element, written out in the +// key order encoding/json emits. It is the contract both hosts owe the replay +// UI: an element the reader can read, with no host-function members and nothing +// dropped. Keys the two hosts disagree on (a DOM has no `checked`, a native +// tree has no `dataset`) are each host's own business; the ENCODING is not. +const canonicalElement = `{ + "__sanderlingSelector": "resource-id:TxnAmountField", + "attrs": { + "bounds": "[0,100,400,160]", + "editable": "true", + "enabled": "true", + "resource-id": "TxnAmountField", + "text": "199" + }, + "bounds": {"bottom": 160, "left": 0, "right": 400, "top": 100}, + "checked": false, + "class": "", + "clickable": false, + "desc": "", + "editable": true, + "enabled": true, + "focused": false, + "id": "TxnAmountField", + "selected": false, + "text": "199", + "x": 200, + "y": 130 +}` + +// TestExtractorEncoding_ElementIsIdenticalOnBothHosts holds the two extractor +// paths to one encoding of one element. The goja hosts (ios, android) run the +// getter in-process and encode the value it returned; the web host runs it in +// V8 and injects the page's reading through OverrideExtractorValues. A reader +// opening a trace does not know which host wrote it, so the same element has to +// land as the same bytes either way. +// +// The goja side used to write null here: an ax element carries find/findAll as +// host functions and json.Marshal refuses the whole object over them. +func TestExtractorEncoding_ElementIsIdenticalOnBothHosts(t *testing.T) { + want := compactJSON(t, canonicalElement) + + native := newVerifier(t) + mustLoad(t, native, elementExtractorSpec) + pushTree(t, native, elementTreeJSON) + fromGoja := string(native.extractors[0].curr) + if fromGoja != want { + t.Errorf("goja host encoded the element as\n %s\nwant\n %s", fromGoja, want) + } + + web := newVerifier(t) + mustLoad(t, web, elementExtractorSpec) + if err := web.PushSnapshot(SnapshotInput{}); err != nil { + t.Fatal(err) + } + if _, err := web.OverrideExtractorValues(map[int]json.RawMessage{0: json.RawMessage(want)}); err != nil { + t.Fatal(err) + } + fromWeb := string(web.extractors[0].curr) + if fromWeb != fromGoja { + t.Errorf("the same element reaches the trace as\n %s\non the web host and\n %s\non goja", + fromWeb, fromGoja) + } +} + +// TestExtractorEncoding_MirrorsTheWebSanitizeRule pins the goja host to the +// rule the web host applies before a reading leaves the page (sanitize in +// pkg/spec/src/web-runtime.ts, asserted there by the "sanitize ..." tests in +// pkg/spec/test/web-runtime.test.ts). Two hosts encoding one value two ways is +// the same defect as encoding it not at all: the reader cannot line the traces +// up. +func TestExtractorEncoding_MirrorsTheWebSanitizeRule(t *testing.T) { + for _, test := range []struct { + name string + expression string + want string + }{ + { + name: "function-valued properties are dropped", + expression: `({ keep: 1, fn: () => 7 })`, + want: `{"keep":1}`, + }, + { + name: "a top-level function is not a value", + expression: `(() => 7)`, + want: `null`, + }, + { + name: "a self-referential cycle breaks instead of overflowing", + expression: `(() => { const a = { name: "root" }; a.self = a; return a; })()`, + want: `{"name":"root","self":null}`, + }, + { + name: "arrays and nested plain values are preserved", + expression: `({ items: [1, "two", { ok: true }] })`, + want: `{"items":[1,"two",{"ok":true}]}`, + }, + { + name: "a non-finite number is not a value", + expression: `Number("nope")`, + want: `null`, + }, + } { + t.Run(test.name, func(t *testing.T) { + if got := encodeSpecValue(t, test.expression); got != test.want { + t.Errorf("encoded as %s, want %s", got, test.want) + } + }) + } +} + +// TestExtractorEncoding_BoundsRecursionPastTheDepthLimit mirrors the web host's +// depth cap. state.ax hands out no cyclic element, but a spec returning a value +// it built itself can nest without end, and a walk with no bound takes the run +// down with a stack overflow. +func TestExtractorEncoding_BoundsRecursionPastTheDepthLimit(t *testing.T) { + encoded := encodeSpecValue(t, `(() => { + let deep = { leaf: true }; + for (let i = 0; i < 40; i++) deep = { next: deep }; + return deep; + })()`) + + var node any + if err := json.Unmarshal([]byte(encoded), &node); err != nil { + t.Fatalf("decode %s: %v", encoded, err) + } + for depth := 0; depth < recordableMaxDepth; depth++ { + object, ok := node.(map[string]any) + if !ok { + t.Fatalf("depth %d: recursion stopped early at %v", depth, node) + } + node = object["next"] + } + if node != nil { + t.Errorf("depth %d is %v, want null", recordableMaxDepth, node) + } +} + +// encodeSpecValue returns what the trace records for an extractor whose getter +// returned the given expression. +func encodeSpecValue(t *testing.T, expression string) string { + t.Helper() + verifier := newVerifier(t) + mustLoad(t, verifier, "__sanderling__.extract(state => "+expression+", \"value\");\nglobalThis.properties = {};") + if err := verifier.PushSnapshot(SnapshotInput{}); err != nil { + t.Fatal(err) + } + return string(verifier.extractors[0].curr) +} + +func compactJSON(t *testing.T, source string) string { + t.Helper() + var compact bytes.Buffer + if err := json.Compact(&compact, []byte(source)); err != nil { + t.Fatal(err) + } + return compact.String() +} diff --git a/internal/verifier/marshal.go b/internal/verifier/marshal.go index f9604b5..d738e84 100644 --- a/internal/verifier/marshal.go +++ b/internal/verifier/marshal.go @@ -26,7 +26,7 @@ type stateInput struct { } // stateObject builds the JS-side `state` object matching the State type from -// pkg/spec-api. Fields beyond snapshots/ax are included when the caller +// pkg/spec. Fields beyond snapshots/ax are included when the caller // populated them on stateInput. func stateObject(runtime *goja.Runtime, input stateInput) (*goja.Object, error) { state := runtime.NewObject() @@ -344,7 +344,19 @@ func lastActionFields(action *Action) []actionField { point := func(x, y int) []actionField { return []actionField{{key: "x", value: x}, {key: "y", value: y}} } - fields := []actionField{{key: "kind", value: string(action.Kind)}} + // An action whose apply call failed is not an action that did not happen: + // the dispatch may have landed before the error. That is unknown, and + // unknown is null here for the same reason every other absence in the spec + // surface is, so a property decides for itself instead of being handed a + // "nothing happened" the runner cannot vouch for. + var applied any + if action.Applied { + applied = true + } + fields := []actionField{ + {key: "kind", value: string(action.Kind)}, + {key: "applied", value: applied}, + } if action.On != "" { fields = append(fields, actionField{key: "on", value: action.On}) } @@ -397,7 +409,7 @@ func objectFromFields(runtime *goja.Runtime, fields []actionField) *goja.Object // has no Go-side state object to read: the runner pushes this JSON into the // page before each extractor evaluation. A nil action encodes as JSON null, // the same value the goja host reports on the first step of a run and after a -// step whose action was never applied. +// step whose action was never dispatched. func EncodeLastAction(action *Action) json.RawMessage { if action == nil { return json.RawMessage("null") diff --git a/internal/verifier/marshal_test.go b/internal/verifier/marshal_test.go index 1b91d1f..b76e480 100644 --- a/internal/verifier/marshal_test.go +++ b/internal/verifier/marshal_test.go @@ -168,6 +168,7 @@ func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) { }{ {"nil", nil}, {"Tap", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", X: 12, Y: 34}}, + {"TapApplied", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}}, {"TapWithoutSelector", &Action{Kind: ActionKindTap, X: 12, Y: 34}}, {"DoubleTap", &Action{Kind: ActionKindDoubleTap, On: `desc:say "hi" `}}, {"InputText", &Action{Kind: ActionKindInputText, On: "id:field", Text: "50"}}, @@ -193,3 +194,42 @@ func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) { }) } } + +// A spec has to be able to tell three things apart: no action ran, an action +// ran, and an action was dispatched whose fate the runner cannot vouch for. +// The third used to be reported as the first, which is how a property that +// reasons "an effect landed with no action to cause it" convicts an app over +// an RPC deadline. +func TestLastAction_SeparatesNoActionFromAnActionOfUnknownFate(t *testing.T) { + verifier := newVerifier(t) + mustLoad(t, verifier, ` + globalThis.fate = __sanderling__.extract(state => + state.lastAction === null ? "no action" + : state.lastAction.applied === true ? "applied" + : state.lastAction.applied === null ? "unknown" + : "unreadable"); + `) + + for _, testCase := range []struct { + name string + action *Action + want string + }{ + {"nothing ran", nil, "no action"}, + {"dispatch confirmed", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}, "applied"}, + {"dispatch unconfirmed", &Action{Kind: ActionKindTap, On: "id:TxnSubmit"}, "unknown"}, + } { + t.Run(testCase.name, func(t *testing.T) { + if err := verifier.PushSnapshot(SnapshotInput{ + Snapshots: Snapshots{}, + LastAction: testCase.action, + }); err != nil { + t.Fatal(err) + } + handle := verifier.runtime.GlobalObject().Get("fate").ToObject(verifier.runtime) + if got := handle.Get("current").String(); got != testCase.want { + t.Errorf("the spec read %q, want %q", got, testCase.want) + } + }) + } +} diff --git a/internal/verifier/types.go b/internal/verifier/types.go index 5b35eb7..125703b 100644 --- a/internal/verifier/types.go +++ b/internal/verifier/types.go @@ -33,6 +33,11 @@ type Action struct { // Direction is the scroll direction for ActionKindScroll: one of "up", // "down", "left", "right". Empty for every other kind. Direction string + // Applied is meaningful only on the action a step reports to the spec as + // state.lastAction: true when the runner saw the dispatch succeed, false + // when the apply call failed and nothing can say whether the action + // reached the app. The spec is told which of the two it is. + Applied bool } // LogEntry mirrors a logcat line captured between steps. diff --git a/internal/verifier/worker.go b/internal/verifier/worker.go index e891969..69bd9b8 100644 --- a/internal/verifier/worker.go +++ b/internal/verifier/worker.go @@ -7,6 +7,8 @@ import ( "errors" "fmt" "maps" + "math" + "reflect" "sort" "time" @@ -356,21 +358,76 @@ func (v *Verifier) runExtractor(extractor *extractorState, state goja.Value) (go } // encodeExtractorValue produces a stable JSON encoding of an extractor's -// current value for diff comparison. goja values that don't survive Export -// (e.g. wrapped host functions) yield nil; callers treat nil as "unknown" and -// emit no diff entry. +// current value for diff comparison. Values that still don't survive encoding +// yield nil; callers treat nil as "unknown" and emit no diff entry. func encodeExtractorValue(value goja.Value) []byte { if value == nil || goja.IsUndefined(value) || goja.IsNull(value) { return []byte("null") } - exported := value.Export() - body, err := json.Marshal(exported) + body, err := json.Marshal(recordableValue(value.Export(), 0, map[uintptr]bool{})) if err != nil { return nil } return body } +// recordableMaxDepth mirrors SANITIZE_MAX_DEPTH in pkg/spec/src/web-runtime.ts. +const recordableMaxDepth = 32 + +// recordableValue applies the web host's sanitize rule (web-runtime.ts) to an +// exported goja value: function members are dropped, a cycle or a branch past +// the depth cap becomes null, and a non-finite number becomes null. One rule on +// both hosts is what lets the replay UI render a trace without the reader +// having to know which host produced it. An ax element carries its find and +// findAll host functions, and json.Marshal rejects the whole element over them, +// so without this an element-valued extractor reached the trace as null. +func recordableValue(value any, depth int, seen map[uintptr]bool) any { + switch typed := value.(type) { + case map[string]any: + address := reflect.ValueOf(typed).Pointer() + if depth >= recordableMaxDepth || seen[address] { + return nil + } + seen[address] = true + members := make(map[string]any, len(typed)) + for key, member := range typed { + if reflect.ValueOf(member).Kind() == reflect.Func { + continue + } + members[key] = recordableValue(member, depth+1, seen) + } + return members + case []any: + if depth >= recordableMaxDepth { + return nil + } + // Every zero-length allocation shares one address, so tracking an empty + // array would identify it as every other empty array. It cannot close a + // cycle either way. + if len(typed) > 0 { + address := reflect.ValueOf(typed).Pointer() + if seen[address] { + return nil + } + seen[address] = true + } + members := make([]any, len(typed)) + for index, member := range typed { + members[index] = recordableValue(member, depth+1, seen) + } + return members + case float64: + if math.IsNaN(typed) || math.IsInf(typed, 0) { + return nil + } + return typed + } + if reflect.ValueOf(value).Kind() == reflect.Func { + return nil + } + return value +} + // ChangedExtractors returns the named extractors whose value changed between // the prior PushSnapshot and the current one. The map is keyed by extractor // name; unnamed extractors (extractor_N fallback) are included so the replay diff --git a/pkg/spec/README.md b/pkg/spec/README.md index c651ddf..abfc57b 100644 --- a/pkg/spec/README.md +++ b/pkg/spec/README.md @@ -1,67 +1,13 @@ # @sanderling/spec -TypeScript spec API for [sanderling](https://github.com/priyanshujain/sanderling), a property-based UI fuzzer for mobile and web apps. +TypeScript spec API for [sanderling](https://github.com/priyanshujain/sanderling), a property-based UI fuzzer for Android, iOS and web apps. -Spec authors write properties (what the app must always or eventually do), extractors (structured state from the UI), and action generators (what sanderling is allowed to do). The `sanderling` CLI evaluates the spec in a loop against a running app. - -## Install +A spec exports properties (what the app must always or eventually do), extractors (structured state read off the UI), and action generators (what sanderling is allowed to do). The `sanderling` CLI evaluates the spec against a running app once per step. ```sh npm install --save-dev @sanderling/spec ``` -## Usage +[Getting started](https://priyanshujain.github.io/sanderling/manual/getting-started/) installs the CLI and runs a first spec. The [spec language reference](https://priyanshujain.github.io/sanderling/manual/spec-language/) lists every primitive, and the [case study](https://priyanshujain.github.io/sanderling/manual/case-study/) walks a complete spec end to end. -```ts -import { extract, always, eventually, actions, weighted, taps, swipes, InputText, Tap } from "@sanderling/spec"; - -const loggedIn = extract((s) => !!s.ax.find("id:home-tab-bar")); -const balance = extract((s) => (s.snapshots.balance as number) ?? 0); -const emailField = extract((s) => s.ax.find("id:email-field")); -const submitButton = extract((s) => s.ax.find("id:sign-in-button")); - -export const properties = { - balanceNeverNegative: always(() => balance.current >= 0), - loginSucceeds: eventually(() => loggedIn.current).within(30, "seconds"), -}; - -const doLogin = actions(() => { - if (loggedIn.current) return []; - const email = emailField.current; - const submit = submitButton.current; - if (!email || !submit) return []; - return [InputText({ into: email, text: "test@example.com" }), Tap({ on: submit })]; -}); - -export const actionsRoot = weighted( - [50, doLogin], - [10, taps], - [2, swipes], -); -``` - -## Setup actions - -Some action generators are not fuzz targets but preconditions: they drive the -app from a fresh state into the surface you actually want to fuzz (login, -onboarding, permission grants, seed data). Export them as `setup` instead of -mixing them into `actionsRoot`. The runner tries `setup` first; if it yields -no action, it falls through to `actionsRoot`. State regressing back across the -precondition (e.g. logout under fuzz) automatically re-engages setup. - -```ts -const login = actions(() => { - if (loggedIn.current) return []; - return [InputText({ into: emailField.current!, text: "demo@app.test" }), Tap({ on: submitButton.current! })]; -}); - -export const setup = login; -export const actionsRoot = weighted([60, browse], [40, edit]); - -(globalThis as { setup?: unknown }).setup = setup; -``` - -Setup is just an `ActionGenerator`; compose with `actions`, `weighted`, or -`whenRoute` exactly like the main pool. - -Works identically across Android, iOS, and web targets. +The CLI bundles this package's TypeScript sources at run time, so keep the CLI and the package on the same release. diff --git a/pkg/spec/package.json b/pkg/spec/package.json index ffbfb84..05db2dd 100644 --- a/pkg/spec/package.json +++ b/pkg/spec/package.json @@ -21,6 +21,7 @@ }, "files": [ "dist", + "src", "README.md" ], "repository": { diff --git a/pkg/spec/src/index.ts b/pkg/spec/src/index.ts index 4ac8c31..b80170f 100644 --- a/pkg/spec/src/index.ts +++ b/pkg/spec/src/index.ts @@ -4,6 +4,7 @@ export type { Action, ActionGenerator, AttrSelector, + Direction, DoubleTapAction, EventuallyFormula, ExceptionRecord, @@ -12,11 +13,14 @@ export type { InputTextAction, Key, KnownAttrSelectors, + LastAction, LogEntry, + LongPressAction, Point, PressKeyAction, RawAttrs, Sampler, + ScrollAction, SelectorPath, Snapshots, State, diff --git a/pkg/spec/src/types.ts b/pkg/spec/src/types.ts index f748d83..c28fd05 100644 --- a/pkg/spec/src/types.ts +++ b/pkg/spec/src/types.ts @@ -102,10 +102,20 @@ export interface ExceptionRecord { unixMillis?: number; } +/** + * The previous step's action as the runner reports it. `applied` is true when + * the runner saw the dispatch succeed and null when the apply call failed with + * the gesture possibly already delivered: an RPC deadline can fire after the + * tap landed. Null is unknown, not "it did not happen" (`state.lastAction` is + * itself null for that), so a property attributing an effect to this action + * has to decline unless `applied` is true. + */ +export type LastAction = Action & { applied: true | null }; + export interface State { snapshots: Snapshots; ax: AccessibilityTree; - lastAction: Action | null; + lastAction: LastAction | null; time: number; logs: readonly LogEntry[]; exceptions: readonly ExceptionRecord[]; diff --git a/pkg/spec/test/api.test.ts b/pkg/spec/test/api.test.ts index eb38207..68a3945 100644 --- a/pkg/spec/test/api.test.ts +++ b/pkg/spec/test/api.test.ts @@ -29,6 +29,12 @@ import { weighted, whenRoute, } from "../src/index.ts"; +import type { + Action, + Direction, + LongPressAction, + ScrollAction, +} from "../src/index.ts"; import { setSamplerRng } from "../src/actions.ts"; import { Pcg } from "../src/pcg.ts"; import type { GeneratorNode } from "../src/action-tree.ts"; @@ -397,3 +403,18 @@ test("whenRoute body is skipped for a null route", () => { const node = whenRoute(route, ["home"], () => [Tap({ on: "id:x" })]); assert.deepEqual((node as { generate: () => unknown }).generate(), []); }); + +// The package entry is the only module a spec author can import from, so every +// member of the exported Action union, and the Direction needed to build a +// Scroll, has to be reachable there rather than only from src/types.ts. +test("index exports every action type a spec author annotates with", () => { + const direction: Direction = "down"; + const scroll: ScrollAction = Scroll({ direction, in: "id:list" }); + const longPress: LongPressAction = LongPress({ on: "id:row" }); + const built: Action[] = [scroll, longPress]; + + assert.deepEqual( + built.map(action => action.kind), + ["Scroll", "LongPress"], + ); +}); diff --git a/pkg/spec/test/folio-new-account.test.ts b/pkg/spec/test/folio-new-account.test.ts index def23bb..d5ee3ab 100644 --- a/pkg/spec/test/folio-new-account.test.ts +++ b/pkg/spec/test/folio-new-account.test.ts @@ -3,8 +3,12 @@ import { test } from "node:test"; import { createdAccountHasNonZeroBalance } from "../../../examples/folio/sanderling/predicates.ts"; -const created = { kind: "Tap", on: "testTag:AddAccountScreen > testTag:AddAccountSubmit" }; -const idle = { kind: "Tap", on: "testTag:HomeScreen > testTag:AccountCard" }; +const created = { + kind: "Tap", + on: "testTag:AddAccountScreen > testTag:AddAccountSubmit", + applied: true as const, +}; +const idle = { kind: "Tap", on: "testTag:HomeScreen > testTag:AccountCard", applied: true as const }; const account = (name: string, balance: number | null) => ({ name, balance }); @@ -27,7 +31,7 @@ test("a double-tapped create is judged the same way", () => { assert.equal( createdAccountHasNonZeroBalance({ route: "home", - lastAction: { kind: "DoubleTap", on: "id:AddAccountSubmit" }, + lastAction: { kind: "DoubleTap", on: "id:AddAccountSubmit", applied: true }, typedName: "Travel", before: [account("Checking", 0)], after: [account("Checking", 0), account("Travel", 5000)], @@ -233,3 +237,20 @@ test("a card that was already there is not a card that was just created", () => false, ); }); + +// The apply call failed with the gesture possibly already delivered, so nobody +// knows whether that account was created. The card carrying the typed name may +// be an older one that scrolled into view, and attributing it to a creation +// that may never have happened is a conviction built on a guess. +test("a create the runner could not confirm attributes nothing", () => { + assert.equal( + createdAccountHasNonZeroBalance({ + route: "home", + lastAction: { ...created, applied: null }, + typedName: "Travel", + before: [account("Checking", 0)], + after: [account("Checking", 0), account("Travel", 5000)], + }), + false, + ); +}); diff --git a/pkg/spec/test/folio-submit-balance-predicate.test.ts b/pkg/spec/test/folio-submit-balance-predicate.test.ts index 57a94ce..672fb5b 100644 --- a/pkg/spec/test/folio-submit-balance-predicate.test.ts +++ b/pkg/spec/test/folio-submit-balance-predicate.test.ts @@ -12,7 +12,7 @@ test("single submit: delta matches typed amount", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -26,7 +26,7 @@ test("double submit: delta is twice the typed amount, fires", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -40,7 +40,7 @@ test("DoubleTap kind also caught when delta exceeds typed amount", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 0, @@ -54,7 +54,7 @@ test("wrong action kind: vacuous true even with mismatch", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "InputText", on: submitOn }, + lastAction: { kind: "InputText", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -68,7 +68,7 @@ test("wrong target: vacuous true even with mismatch", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: "testTag:LoginScreen > testTag:LoginSubmit" }, + lastAction: { kind: "Tap", on: "testTag:LoginScreen > testTag:LoginSubmit", applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 1000, @@ -96,7 +96,7 @@ test("zero typedAmount: vacuous true", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 0, prevTotalBalance: 1000, @@ -110,7 +110,7 @@ test("selector as object: coerced safely and TxnSubmit detected", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: { testTag: "TxnSubmit" } }, + lastAction: { kind: "Tap", on: { testTag: "TxnSubmit" }, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 0, @@ -124,7 +124,7 @@ test("selector as object without TxnSubmit: vacuous true", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: { testTag: "LoginSubmit" } }, + lastAction: { kind: "Tap", on: { testTag: "LoginSubmit" }, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: 0, @@ -138,7 +138,7 @@ test("raw whole-dollar input: single submit clears", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("50"), prevTotalBalance: 5000, @@ -152,7 +152,7 @@ test("raw whole-dollar input: double submit fires", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("50"), prevTotalBalance: 5000, @@ -166,7 +166,7 @@ test("decimal input from empty prior balance clears", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("5.50"), prevTotalBalance: 0, @@ -180,7 +180,7 @@ test("DoubleTap kind with raw whole-dollar input fires", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("100"), prevTotalBalance: 0, @@ -194,7 +194,7 @@ test("route gate: ledger landing with stale carrier is skipped", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "ledger", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -208,7 +208,7 @@ test("route gate: add-transaction landing with double-submit delta is skipped", assert.equal( submitChangesBalanceByTypedAmount({ route: "add-transaction", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -222,7 +222,7 @@ test("route gate: null route is skipped", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: null, - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -236,7 +236,7 @@ test("route gate: home landing with matching delta passes", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -250,7 +250,7 @@ test("route gate: home landing with double-insert delta fires", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 5000, prevTotalBalance: 0, @@ -275,7 +275,7 @@ test("above 2^53 a healthy single submit is not reported", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1600, prevTotalBalance: HUGE_BALANCE, @@ -289,7 +289,7 @@ test("above 2^53 a double-submit delta is not reported either", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1600, prevTotalBalance: HUGE_BALANCE, @@ -303,7 +303,7 @@ test("an unreadable previous balance above 2^53 is not evidence", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1600, prevTotalBalance: HUGE_BALANCE, @@ -320,7 +320,7 @@ test("typed amount above 2^53 is not evidence", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 1e23, prevTotalBalance: 0, @@ -336,7 +336,7 @@ test("boundary: a double submit landing exactly on MAX_SAFE_INTEGER still fires" assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 4503599627370495, prevTotalBalance: 0, @@ -350,7 +350,7 @@ test("boundary: a single submit landing exactly on MAX_SAFE_INTEGER passes", () assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 9007199254740991, prevTotalBalance: 0, @@ -364,7 +364,7 @@ test("boundary: one cent past MAX_SAFE_INTEGER stops being evidence", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 4503599627370496, prevTotalBalance: 0, @@ -381,7 +381,7 @@ test("a large but exact difference between safe balances still fires", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 500, prevTotalBalance: -9007199254740991, @@ -397,7 +397,7 @@ test("21-digit typed amount with an unmoved balance is not a violation", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: parseTypedAmount("999999999999999999999"), prevTotalBalance: 220900, @@ -417,7 +417,7 @@ test("freshness: two submits in the window is vacuous, not a conviction", () => assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 2, typedAmount: 19600, prevTotalBalance: 0, @@ -431,7 +431,7 @@ test("freshness: two submits cannot convict even on a clean 2x delta", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 2, typedAmount: 500, prevTotalBalance: 1000, @@ -448,7 +448,7 @@ test("freshness boundary: exactly one submit is the window that convicts", () => assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "DoubleTap", on: submitOn }, + lastAction: { kind: "DoubleTap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 19600, prevTotalBalance: 0, @@ -462,7 +462,7 @@ test("freshness boundary: one submit with a healthy 1x delta still passes", () = assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 1, typedAmount: 19600, prevTotalBalance: 0, @@ -476,7 +476,7 @@ test("freshness boundary: three submits is vacuous", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 3, typedAmount: 500, prevTotalBalance: 0, @@ -493,7 +493,7 @@ test("freshness boundary: a window with no submit in it is vacuous", () => { assert.equal( submitChangesBalanceByTypedAmount({ route: "home", - lastAction: { kind: "Tap", on: submitOn }, + lastAction: { kind: "Tap", on: submitOn, applied: true }, submitsInWindow: 0, typedAmount: 500, prevTotalBalance: 1000, @@ -502,3 +502,21 @@ test("freshness boundary: a window with no submit in it is vacuous", () => { true, ); }); + +// applied: null is the runner saying it dispatched the tap and never learned +// whether it landed. A submit that committed nothing leaves the balance where +// it was, so demanding the typed amount of movement for it convicts an app that +// did exactly what it should have. +test("a submit the runner could not confirm demands no balance move", () => { + assert.equal( + submitChangesBalanceByTypedAmount({ + route: "home", + lastAction: { kind: "Tap", on: submitOn, applied: null }, + submitsInWindow: 1, + typedAmount: 500, + prevTotalBalance: 1000, + currTotalBalance: 1000, + }), + true, + ); +}); diff --git a/pkg/spec/test/folio-submit-window.test.ts b/pkg/spec/test/folio-submit-window.test.ts index da422ed..6e19281 100644 --- a/pkg/spec/test/folio-submit-window.test.ts +++ b/pkg/spec/test/folio-submit-window.test.ts @@ -2,6 +2,7 @@ import assert from "node:assert/strict"; import { test } from "node:test"; import { + committedTransactionsExceedSubmits, countSubmitsInWindow, isTxnSubmitTap, readHomeTotalBalance, @@ -149,3 +150,25 @@ test("an unreadable Home does not close the window", () => { assert.equal(trace[2]?.total, null); assert.equal(trace[3]?.submits, 2); }); + +// The window is an upper bound on the submits it holds, so a submit whose +// dispatch the runner could not confirm belongs in it: the tap may well have +// landed, and a bound that leaves it out is one the transaction it committed +// exceeds. That is the false conviction, a rise of one against a window of +// zero, on the property carrying most of the detection on android. +test("a submit the runner could not confirm still counts toward the window", () => { + const window = countSubmitsInWindow({ + previousCount: 0, + lastAction: { kind: "Tap", on: submitOn, applied: null }, + fresh: true, + }); + assert.equal(window.reported, 1); + assert.equal( + committedTransactionsExceedSubmits({ + countsBefore: { Travel: 3 }, + countsAfter: { Travel: 4 }, + submitsInWindow: window.reported, + }), + false, + ); +}); diff --git a/pkg/spec/test/folio-transition-frame.test.ts b/pkg/spec/test/folio-transition-frame.test.ts index f785642..64bcbf3 100644 --- a/pkg/spec/test/folio-transition-frame.test.ts +++ b/pkg/spec/test/folio-transition-frame.test.ts @@ -94,7 +94,7 @@ test("the measured android transition chain no longer convicts at delta 0", () = const step = ( tags: string[], totalText: string | undefined, - lastAction: { kind: string; on: string } | null, + lastAction: { kind: string; on: string; applied: true } | null, ) => { const route = routeOfFrame(SCREENS, frame(...tags)); const reading = readHomeTotalBalance({ route, totalText, previousCarrier: carrier }); @@ -104,8 +104,12 @@ test("the measured android transition chain no longer convicts at delta 0", () = return { route, total: reading.value, submits: window.reported }; }; - const back = { kind: "DoubleTap", on: "id:BackButton" }; - const phantomSubmit = { kind: "Tap", on: "testTag:AddTransactionScreen > testTag:TxnSubmit" }; + const back = { kind: "DoubleTap", on: "id:BackButton", applied: true as const }; + const phantomSubmit = { + kind: "Tap", + on: "testTag:AddTransactionScreen > testTag:TxnSubmit", + applied: true as const, + }; const transition = step(["AddTransactionScreen", "HomeScreen"], "$86,911.00", back); assert.equal(transition.route, null);