diff --git a/.github/scripts/replay-ui-summary-test.sh b/.github/scripts/replay-ui-summary-test.sh new file mode 100755 index 0000000..b75e988 --- /dev/null +++ b/.github/scripts/replay-ui-summary-test.sh @@ -0,0 +1,125 @@ +#!/usr/bin/env bash +# Drives replay-ui-summary.sh over the traces in testdata/ and checks the +# summary it renders. Run under the flags GitHub Actions uses for a `run:` +# block, because that is where a swallowed failure hides. +# +# testdata/replay-ui-real-run.jsonl is the first 10 steps of the dogfood run in +# actions run 31873049857 on master, with the per-step `hierarchy` dumps and the +# rowElements/tabElements extractors removed so the file stays readable. Nothing +# else was touched. That run was green, and badgeCountMatchesThePanel judged +# nothing in all 80 of its steps. The other two traces are written by hand. +set -euo pipefail + +here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +script="$here/replay-ui-summary.sh" +testdata="$here/testdata" +work="$(mktemp -d)" +trap 'rm -rf "$work"' EXIT + +failed=0 +rendered="" +status=0 + +summarise() { # + rendered="$work/$1.md" + : > "$rendered" + status=0 + GITHUB_STEP_SUMMARY="$rendered" SEED=3 MAX_STEPS=80 \ + bash -eo pipefail "$script" "$2" >/dev/null 2>"$work/$1.err" || status=$? +} + +fail() { + echo "FAIL: $*" >&2 + failed=1 +} + +expect_status() { # + [ "$status" = "$1" ] || fail "$2: exit $status, want $1" +} + +expect_line() { # + grep -qxF -- "$1" "$rendered" || fail "$2: summary has no line '$1'" +} + +expect_absent() { # + if grep -q -- "$1" "$rendered"; then fail "$2: summary should not mention '$1'"; fi +} + +plant() { # -> echoes the output dir + local dir="$work/$1/runs/20260815-075347" + mkdir -p "$dir" + cp "$testdata/$2" "$dir/trace.jsonl" + echo "$work/$1/runs" +} + +# A green run of the real thing. Every count here was measured, not chosen. +summarise real "$(plant real replay-ui-real-run.jsonl)" +expect_status 0 real +expect_line "- 10 steps recorded, 10 verified, 0 with violations" real +expect_line "| selectedStepIsInRange | 10 | 0 |" real +expect_line "| exactlyOneStepIsSelected | 10 | 0 |" real +expect_line "| stepCountMatchesTheList | 10 | 0 |" real +expect_line "| screenshotShowsTheSelectedStep | 10 | 0 |" real +expect_line "| switchingTabsKeepsTheStep | 2 | 8 |" real +expect_line "| badgeCountMatchesThePanel | **0** | 10 |" real +expect_line "- \`badgeCountMatchesThePanel\` judged nothing on this run: no step had a violations badge on screen" real +expect_absent "checked nothing" real + +# Every property judged at least once, including the one the real run never +# reached. Without this the counts above are consistent with a guard that can +# only ever return zero. +summarise every "$(plant every replay-ui-every-property.jsonl)" +expect_status 0 every +expect_line "- 4 steps recorded, 3 verified, 0 with violations" every +expect_line "| noUncaughtExceptions | 3 | 0 |" every +expect_line "| screenshotShowsTheSelectedStep | 3 | 0 |" every +expect_line "| switchingTabsKeepsTheStep | 2 | 1 |" every +expect_line "| badgeCountMatchesThePanel | 1 | 2 |" every + +# The fuzzer sat on the run list: the step page never rendered, so nothing was +# ever compared. This is the run that used to pass. +summarise blind "$(plant blind replay-ui-nothing-rendered.jsonl)" +expect_status 1 blind +expect_line "| selectedStepIsInRange | **0** | 4 |" blind +expect_line "| badgeCountMatchesThePanel | **0** | 4 |" blind +expect_line "- \`exactlyOneStepIsSelected\` judged nothing on this run: no step had a row in the step list" blind +grep -q "this run checked nothing" "$rendered" || fail "blind: no checked-nothing verdict" +grep -q "the step page never rendered" "$work/blind.err" || fail "blind: nothing on stderr" + +# The run directory exists but the trace does not, and the glob matches nothing +# at all. Both are the harness dying before it checked anything. +mkdir -p "$work/empty-run/runs/20260815-075347" +summarise empty-run "$work/empty-run/runs" +expect_status 1 empty-run +grep -q "no trace at" "$rendered" || fail "empty-run: no missing-trace line" + +summarise no-glob "$work/no-glob/runs" +expect_status 1 no-glob +grep -q "nothing under" "$rendered" || fail "no-glob: no missing-directory line" + +# A property this summary does not know about is a summary that silently counts +# six of seven properties, which is the bug one level up. +drifted="$(plant drift replay-ui-every-property.jsonl)" +sed 's/badgeCountMatchesThePanel/badgeAgreesWithThePanel/g' \ + "$testdata/replay-ui-every-property.jsonl" > "$drifted/20260815-075347/trace.jsonl" +summarise drift "$drifted" +expect_status 1 drift +grep -q "these counts cannot be trusted" "$rendered" || fail "drift: no untrusted verdict" + +# Same for a reading the spec no longer declares: the count would quietly go to +# zero and read as a UI that stopped rendering. +sed 's/extract("violationBadges"/extract("violationCounters"/' \ + "$here/../../replay-ui/sanderling/spec.ts" > "$work/renamed-spec.ts" +rendered="$work/renamed.md" +: > "$rendered" +status=0 +GITHUB_STEP_SUMMARY="$rendered" SPEC="$work/renamed-spec.ts" \ + bash -eo pipefail "$script" "$(plant renamed replay-ui-real-run.jsonl)" \ + >/dev/null 2>&1 || status=$? +expect_status 1 renamed +grep -q "no longer declares violationBadges" "$rendered" || fail "renamed: no drift verdict" + +if [ "$failed" = 0 ]; then + echo "replay-ui-summary.sh: ok" +fi +exit "$failed" diff --git a/.github/scripts/replay-ui-summary.sh b/.github/scripts/replay-ui-summary.sh new file mode 100755 index 0000000..af0cf4c --- /dev/null +++ b/.github/scripts/replay-ui-summary.sh @@ -0,0 +1,222 @@ +#!/usr/bin/env bash +# Reads the replay-ui dogfood trace and reports, per property, how many steps +# that property actually judged. Kept out of the workflow YAML so it can be run +# by hand against a local run: +# +# GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood +# +# `sanderling test` exiting 0 says only that no property returned false. Every +# property in replay-ui/sanderling/spec.ts declines to judge when a reading it +# needs is absent, which is right individually and useless in aggregate: a run +# that never rendered the step page returns false nowhere and exits 0, so +# checked-and-clean and checked-nothing arrive at the same green tick. Section 8 +# of docs/development/design-principles.md is the rule this leg was breaking. +set -euo pipefail + +root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +output="${1:-runs/dogfood}" +spec="${SPEC:-$root/replay-ui/sanderling/spec.ts}" +summary="${GITHUB_STEP_SUMMARY:-/dev/null}" + +shopt -s nullglob +run_dirs=("$output"/*/) +shopt -u nullglob + +{ + echo "### replay-ui dogfood" + echo + echo "- seed \`${SEED:-unset}\`, budget ${MAX_STEPS:-unset} steps" +} >> "$summary" + +if [ ${#run_dirs[@]} -eq 0 ]; then + echo "- **nothing under \`$output/\`**: the run never started, so no property was ever evaluated" >> "$summary" + echo "replay-ui: no run directory under $output/, so there is nothing to judge" >&2 + exit 1 +fi + +trace="${run_dirs[${#run_dirs[@]} - 1]}trace.jsonl" +if [ ! -f "$trace" ]; then + echo "- **no trace at \`$trace\`**: the run wrote no steps, so no property was ever evaluated" >> "$summary" + echo "replay-ui: $trace does not exist, so there is nothing to judge" >&2 + exit 1 +fi + +TRACE="$trace" SPEC="$spec" python3 - >> "$summary" <<'PY' +import json +import os +import re +import sys + +trace_path = os.environ["TRACE"] +spec_path = os.environ["SPEC"] + + +def toolbar(values): + reading = values.get("toolbar") + return reading if isinstance(reading, dict) else None + + +def numbered(reading, key): + return reading is not None and reading.get(key) is not None + + +def listing(values, name): + reading = values.get(name) + return reading if isinstance(reading, list) else [] + + +# One entry per property in replay-ui/sanderling/spec.ts, holding the guard that +# property opens with. The trace records extractor values, not verdicts, so +# counting the steps a property really judged means restating its guard here, +# and that restatement is the risk this file carries: a guard that drifts from +# its property would report evidence that does not exist. The two checks at the +# bottom make the two drifts that CAN be caught loud rather than silent. +# +# noUncaughtExceptions carries no guard on purpose: it compares a count, not an +# element, so it judges every step the verifier accepted. +GUARDS = { + "noUncaughtExceptions": [], + "selectedStepIsInRange": [ + ("a toolbar reporting a step and a step count", + lambda c, p: numbered(toolbar(c), "step") and numbered(toolbar(c), "stepCount")), + ], + "exactlyOneStepIsSelected": [ + ("a row in the step list", lambda c, p: len(listing(c, "stepRows")) > 0), + ], + "stepCountMatchesTheList": [ + ("a toolbar reporting a step count", lambda c, p: numbered(toolbar(c), "stepCount")), + ("a row in the step list", lambda c, p: len(listing(c, "stepRows")) > 0), + ], + "screenshotShowsTheSelectedStep": [ + ("a toolbar reporting a step", lambda c, p: numbered(toolbar(c), "step")), + ("a screenshot in the before panel", + lambda c, p: c.get("beforeScreenshotStep") is not None), + ], + "switchingTabsKeepsTheStep": [ + ("a tab selection that changed from the step before", + lambda c, p: p is not None and p.get("activeTabs") != c.get("activeTabs")), + ("a toolbar on both steps", + lambda c, p: p is not None and toolbar(p) is not None and toolbar(c) is not None), + ], + "badgeCountMatchesThePanel": [ + ("a violations badge on screen", + lambda c, p: len(listing(c, "violationBadges")) > 0 + and listing(c, "violationBadges")[0] is not None), + ("a violations panel on screen", + lambda c, p: len(listing(c, "violationPanelCounts")) > 0), + ], +} + +# A zero here is the run failing to render, not the fuzzer getting unlucky: +# every one of these needs only that the step page came up, which it does on the +# first step of any working run. The other three are left to be reported. Two of +# them are trajectory-dependent - switchingTabsKeepsTheStep needs a tab switch +# between consecutive steps, badgeCountMatchesThePanel needs the fuzzer to land +# on a violating step AND open the violations tab there - and a gate that +# convicts on an unlucky seed reports a regression it has not found. +MUST_RENDER = ( + "selectedStepIsInRange", + "exactlyOneStepIsSelected", + "stepCountMatchesTheList", + "screenshotShowsTheSelectedStep", +) + +steps = 0 +verified = 0 +violating = 0 +malformed = 0 +property_names = set() +judged = {name: 0 for name in GUARDS} +met_once = {name: [0] * len(conditions) for name, conditions in GUARDS.items()} + +# The trace records only the extractors that changed at a step, so an +# extractor's value at any step is the last change recorded for it, starting +# from null. Steps the verifier skipped advance nothing and are not evaluations. +values = {} +previous = None +with open(trace_path, encoding="utf-8", errors="replace") as lines: + for line in lines: + if not line.strip(): + continue + try: + step = json.loads(line) + except ValueError: + malformed += 1 + continue + steps += 1 + property_names |= set((step.get("residuals") or {}).keys()) + if step.get("violations"): + violating += 1 + if step.get("skipped_verification") or step.get("transitional"): + continue + for name, change in (step.get("extractor_changes") or {}).items(): + values[name] = change.get("curr") + verified += 1 + for name, conditions in GUARDS.items(): + met = [condition(values, previous) for _, condition in conditions] + if all(met): + judged[name] += 1 + for index, held in enumerate(met): + met_once[name][index] += 1 if held else 0 + previous = dict(values) + +report = [] +report.append("- %d steps recorded, %d verified, %d with violations" + % (steps, verified, violating)) +if malformed: + report.append("- **%d unreadable line(s)** in %s" % (malformed, trace_path)) +report.append("") +report.append("- judged: the property compared real values. declined: a reading it " + "needs was absent, so it returned true without checking anything.") +report.append("") +report.append("| property | judged | declined |") +report.append("| --- | --- | --- |") +for name in GUARDS: + count = judged[name] + report.append("| %s | %s | %d |" + % (name, count if count else "**0**", verified - count)) + +report.append("") +for name, conditions in GUARDS.items(): + if judged[name] or not verified: + continue + absent = [label for index, (label, _) in enumerate(conditions) if not met_once[name][index]] + report.append("- `%s` judged nothing on this run: no step had %s" + % (name, ", nor ".join(absent) if absent else "what its guard needs")) + +blind = None +if not verified: + blind = "%s records %d step(s) and not one of them was verified" % (trace_path, steps) +else: + silent = [name for name in MUST_RENDER if not judged[name]] + if silent: + blind = ("%s declined on every step, so the step page never rendered and the " + "exit code is not evidence about the replay UI" % ", ".join(silent)) + +drift = [] +if property_names and property_names != set(GUARDS): + drift.append("the spec's properties are %s but this summary counts %s" + % (", ".join(sorted(property_names)), ", ".join(sorted(GUARDS)))) +try: + with open(spec_path, encoding="utf-8") as handle: + declared = set(re.findall(r'extract\(\s*"([^"]+)"', handle.read())) +except OSError as error: + drift.append("could not read %s to check its readings still exist: %s" % (spec_path, error)) + declared = None +if declared is not None: + gone = sorted({"toolbar", "stepRows", "beforeScreenshotStep", "activeTabs", + "violationBadges", "violationPanelCounts"} - declared) + if gone: + drift.append("%s no longer declares %s, so the counts above describe readings " + "that do not exist" % (spec_path, ", ".join(gone))) + +if blind: + report.append("- **this run checked nothing**: %s" % blind) +for reason in drift: + report.append("- **these counts cannot be trusted**: %s" % reason) + +print("\n".join(report)) +for reason in ([blind] if blind else []) + drift: + print("replay-ui: %s" % reason, file=sys.stderr) +sys.exit(1 if blind or drift else 0) +PY diff --git a/.github/scripts/testdata/replay-ui-every-property.jsonl b/.github/scripts/testdata/replay-ui-every-property.jsonl new file mode 100644 index 0000000..145e5a4 --- /dev/null +++ b/.github/scripts/testdata/replay-ui-every-property.jsonl @@ -0,0 +1,4 @@ +{"extractor_changes":{"activeTabs":{"curr":"screenshot,screenshot","prev":null},"beforeScreenshotStep":{"curr":1,"prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false}],"prev":null},"toolbar":{"curr":{"step":1,"stepCount":5},"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":1,"timestamp":"2026-08-15T07:53:47Z"} +{"extractor_changes":{"activeTabs":{"curr":"violations,screenshot","prev":"screenshot,screenshot"},"beforeScreenshotStep":{"curr":4,"prev":1},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":true,"step":4,"violating":true},{"active":false,"step":5,"violating":false}],"prev":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false}]},"toolbar":{"curr":{"step":4,"stepCount":5},"prev":{"step":1,"stepCount":5}},"violationBadges":{"curr":[1],"prev":[]},"violationPanelCounts":{"curr":[1],"prev":[]}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":2,"timestamp":"2026-08-15T07:53:47Z"} +{"extractor_changes":{"activeTabs":{"curr":"hierarchy,screenshot","prev":"violations,screenshot"},"violationBadges":{"curr":[],"prev":[1]},"violationPanelCounts":{"curr":[],"prev":[1]}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":3,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"skipped_verification":true,"step":4,"timestamp":"2026-08-15T07:53:47Z"} diff --git a/.github/scripts/testdata/replay-ui-nothing-rendered.jsonl b/.github/scripts/testdata/replay-ui-nothing-rendered.jsonl new file mode 100644 index 0000000..9099693 --- /dev/null +++ b/.github/scripts/testdata/replay-ui-nothing-rendered.jsonl @@ -0,0 +1,4 @@ +{"extractor_changes":{"activeTabs":{"curr":"","prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[],"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":1,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":2,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":3,"timestamp":"2026-08-15T07:53:47Z"} +{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":4,"timestamp":"2026-08-15T07:53:47Z"} diff --git a/.github/scripts/testdata/replay-ui-real-run.jsonl b/.github/scripts/testdata/replay-ui-real-run.jsonl new file mode 100644 index 0000000..bf7cf71 --- /dev/null +++ b/.github/scripts/testdata/replay-ui-real-run.jsonl @@ -0,0 +1,10 @@ +{"extractor_changes":{"activeTabs":{"curr":"screenshot,screenshot","prev":null},"beforeScreenshotStep":{"curr":1,"prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":null},"toolbar":{"curr":{"step":1,"stepCount":25},"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"metrics":{"cpu_percent":0,"heap_bytes":2771019,"total_memory_bytes":3893507},"next_action":{"kind":"Tap","selector":"data-testid:tab","tap_point":{"x":774,"y":69},"x":774,"y":69},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":1,"timestamp":"2026-08-15T07:53:47.883914163Z"} +{"extractor_changes":{"activeTabs":{"curr":"screenshot,hierarchy","prev":"screenshot,screenshot"}},"metrics":{"cpu_percent":0,"heap_bytes":3490520,"total_memory_bytes":6061916},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":2,"timestamp":"2026-08-15T07:53:48.356557191Z"} +{"extractor_changes":{"activeTabs":{"curr":"screenshot,properties","prev":"screenshot,hierarchy"},"violationPanelCounts":{"curr":[0],"prev":[]}},"metrics":{"cpu_percent":0,"heap_bytes":3737998,"total_memory_bytes":6595962},"next_action":{"kind":"Tap","resolved_bounds":{"height":109,"width":31,"x":337,"y":308},"selector":"desc:select step 12","tap_point":{"x":352,"y":362},"x":353,"y":363},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":3,"timestamp":"2026-08-15T07:53:48.584625848Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":12,"prev":1},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":12,"stepCount":25},"prev":{"step":1,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":3535304,"total_memory_bytes":7911280},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":4,"timestamp":"2026-08-15T07:53:48.820882198Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":13,"prev":12},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":true,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":13,"stepCount":25},"prev":{"step":12,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":4913839,"total_memory_bytes":8552983},"next_action":{"duration_millis":250,"from_x":98,"from_y":149,"kind":"Swipe","to_x":98,"to_y":749},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/13","step":5,"timestamp":"2026-08-15T07:53:49.038789589Z"} +{"metrics":{"cpu_percent":0,"heap_bytes":4615419,"total_memory_bytes":8970775},"next_action":{"key":"left","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/13","step":6,"timestamp":"2026-08-15T07:53:49.251047338Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":12,"prev":13},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":true,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":12,"stepCount":25},"prev":{"step":13,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":3996016,"total_memory_bytes":8970784},"next_action":{"kind":"Tap","selector":"data-testid:tab","tap_point":{"x":670,"y":69},"x":670,"y":69},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":7,"timestamp":"2026-08-15T07:53:49.451285818Z"} +{"metrics":{"cpu_percent":0,"heap_bytes":4830079,"total_memory_bytes":8974523},"next_action":{"kind":"Tap","selector":"data-testid:step-row","tap_point":{"x":170,"y":256},"x":170,"y":256},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":8,"timestamp":"2026-08-15T07:53:49.658417939Z"} +{"metrics":{"cpu_percent":0,"heap_bytes":3772499,"total_memory_bytes":8974523},"next_action":{"kind":"Tap","selector":"data-testid:step-row","tap_point":{"x":170,"y":363},"x":170,"y":363},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":9,"timestamp":"2026-08-15T07:53:49.865911674Z"} +{"extractor_changes":{"beforeScreenshotStep":{"curr":6,"prev":12},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":true,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":6,"stepCount":25},"prev":{"step":12,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":4783792,"total_memory_bytes":8974552},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/6","step":10,"timestamp":"2026-08-15T07:53:50.075785262Z"} diff --git a/.github/workflows/replay-ui.yml b/.github/workflows/replay-ui.yml index f4b27fd..e72d477 100644 --- a/.github/workflows/replay-ui.yml +++ b/.github/workflows/replay-ui.yml @@ -121,20 +121,14 @@ jobs: SEED: ${{ inputs.seed }} MAX_STEPS: ${{ inputs.max-steps }} + # Exit 0 above means no property returned false. It does not mean any + # property was ever evaluated against real content: they all decline to + # judge when the elements they read are absent, so a run that never + # rendered the step page is green and worthless. This step is what tells + # the two apart, and it fails the job when nothing was judged. - name: Summarise if: always() - run: | - { - echo "### replay-ui dogfood" - echo - echo "- seed \`$SEED\`, budget $MAX_STEPS steps" - for dir in runs/dogfood/*/; do - [ -f "$dir/trace.jsonl" ] || continue - steps=$(wc -l < "$dir/trace.jsonl" | tr -d ' ') - violations=$(grep -c '"violations":\[' "$dir/trace.jsonl" || true) - echo "- $steps steps recorded, $violations step(s) with violations" - done - } >> "$GITHUB_STEP_SUMMARY" + run: .github/scripts/replay-ui-summary.sh runs/dogfood env: SEED: ${{ inputs.seed }} MAX_STEPS: ${{ inputs.max-steps }} diff --git a/Makefile b/Makefile index f86a2cf..50532e5 100644 --- a/Makefile +++ b/Makefile @@ -29,7 +29,7 @@ WEB_DIST := replay-ui/dist GOLINES := $(shell $(GO) env GOPATH)/bin/golines -.PHONY: bootstrap proto sidecar sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift +.PHONY: bootstrap proto sidecar sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api test-ci-scripts spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift bootstrap: $(GO) mod download @@ -117,7 +117,7 @@ fmt-ts: fmt-swift: xcrun swift-format format -i -r companion/Sources -test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test +test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test test-ci-scripts test-go: $(GO) test $(GO_PACKAGES) @@ -141,6 +141,11 @@ test-companion: $(COMPANION_EMBED) $(RUNNER_EMBED) test-kotlin: ANDROID_HOME=$(ANDROID_HOME) $(GRADLE) :sidecar:test +# The CI scripts that read a trace and decide whether a green leg is +# evidence. bash and python3 only, which is all a runner has. +test-ci-scripts: + .github/scripts/replay-ui-summary-test.sh + test-spec-api: cd pkg/spec && npm test --silent diff --git a/docs/development/ci.md b/docs/development/ci.md index db0e291..e82641c 100644 --- a/docs/development/ci.md +++ b/docs/development/ci.md @@ -136,6 +136,24 @@ hold for any trace and need no recalibrating when the fixture changes. The seventh is the stock `noUncaughtExceptions`, which asks nothing of the panels and only fails if the UI throws. Any violation fails the job. +So does a run that judged nothing. Exit 0 says no property returned false, which +is not the same as any property having been evaluated: each one declines to +judge when the elements it reads are absent, so a run where the trace failed to +serve, or where the fuzzer sat on the run list, renders nothing and passes. +`.github/scripts/replay-ui-summary.sh` reads the trace and puts a per-property +count of judged against declined steps in the job summary. Run it by hand with +`GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood`. + +It fails the job when any of the four properties that need nothing beyond the +step page having rendered - `selectedStepIsInRange`, `exactlyOneStepIsSelected`, +`stepCountMatchesTheList`, `screenshotShowsTheSelectedStep` - judged nothing at +all. The other three are reported and not gated, because a zero on them is a +seed getting unlucky rather than a broken leg: `switchingTabsKeepsTheStep` needs +a tab switch between consecutive steps, and `badgeCountMatchesThePanel` needs the +fuzzer to land on a violating step and open the violations tab in that same +step. On the first run measured this way (seed 3, 80 steps) that last one judged +nothing at all, so the fixture reaches it far too rarely to be worth gating on. + ## Reading a failure Both workflows upload their run directories as artifacts, and write the step