Merge branch 'replay-ui-vacuity-counts' into pr-73-followups

This commit is contained in:
pj committed 2026-08-15 14:04:38 +05:30
commit bbedd462dd
8 files changed
+396 -14

No files matched your search

+125
View File
@@ -0,0 +1,125 @@
#!/usr/bin/env bash
# Drives replay-ui-summary.sh over the traces in testdata/ and checks the
# summary it renders. Run under the flags GitHub Actions uses for a `run:`
# block, because that is where a swallowed failure hides.
#
# testdata/replay-ui-real-run.jsonl is the first 10 steps of the dogfood run in
# actions run 31873049857 on master, with the per-step `hierarchy` dumps and the
# rowElements/tabElements extractors removed so the file stays readable. Nothing
# else was touched. That run was green, and badgeCountMatchesThePanel judged
# nothing in all 80 of its steps. The other two traces are written by hand.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
script="$here/replay-ui-summary.sh"
testdata="$here/testdata"
work="$(mktemp -d)"
trap 'rm -rf "$work"' EXIT
failed=0
rendered=""
status=0
summarise() { # <case name> <output dir>
rendered="$work/$1.md"
: > "$rendered"
status=0
GITHUB_STEP_SUMMARY="$rendered" SEED=3 MAX_STEPS=80 \
bash -eo pipefail "$script" "$2" >/dev/null 2>"$work/$1.err" || status=$?
}
fail() {
echo "FAIL: $*" >&2
failed=1
}
expect_status() { # <want> <case>
[ "$status" = "$1" ] || fail "$2: exit $status, want $1"
}
expect_line() { # <line> <case>
grep -qxF -- "$1" "$rendered" || fail "$2: summary has no line '$1'"
}
expect_absent() { # <pattern> <case>
if grep -q -- "$1" "$rendered"; then fail "$2: summary should not mention '$1'"; fi
}
plant() { # <case name> <fixture> -> echoes the output dir
local dir="$work/$1/runs/20260815-075347"
mkdir -p "$dir"
cp "$testdata/$2" "$dir/trace.jsonl"
echo "$work/$1/runs"
}
# A green run of the real thing. Every count here was measured, not chosen.
summarise real "$(plant real replay-ui-real-run.jsonl)"
expect_status 0 real
expect_line "- 10 steps recorded, 10 verified, 0 with violations" real
expect_line "| selectedStepIsInRange | 10 | 0 |" real
expect_line "| exactlyOneStepIsSelected | 10 | 0 |" real
expect_line "| stepCountMatchesTheList | 10 | 0 |" real
expect_line "| screenshotShowsTheSelectedStep | 10 | 0 |" real
expect_line "| switchingTabsKeepsTheStep | 2 | 8 |" real
expect_line "| badgeCountMatchesThePanel | **0** | 10 |" real
expect_line "- \`badgeCountMatchesThePanel\` judged nothing on this run: no step had a violations badge on screen" real
expect_absent "checked nothing" real
# Every property judged at least once, including the one the real run never
# reached. Without this the counts above are consistent with a guard that can
# only ever return zero.
summarise every "$(plant every replay-ui-every-property.jsonl)"
expect_status 0 every
expect_line "- 4 steps recorded, 3 verified, 0 with violations" every
expect_line "| noUncaughtExceptions | 3 | 0 |" every
expect_line "| screenshotShowsTheSelectedStep | 3 | 0 |" every
expect_line "| switchingTabsKeepsTheStep | 2 | 1 |" every
expect_line "| badgeCountMatchesThePanel | 1 | 2 |" every
# The fuzzer sat on the run list: the step page never rendered, so nothing was
# ever compared. This is the run that used to pass.
summarise blind "$(plant blind replay-ui-nothing-rendered.jsonl)"
expect_status 1 blind
expect_line "| selectedStepIsInRange | **0** | 4 |" blind
expect_line "| badgeCountMatchesThePanel | **0** | 4 |" blind
expect_line "- \`exactlyOneStepIsSelected\` judged nothing on this run: no step had a row in the step list" blind
grep -q "this run checked nothing" "$rendered" || fail "blind: no checked-nothing verdict"
grep -q "the step page never rendered" "$work/blind.err" || fail "blind: nothing on stderr"
# The run directory exists but the trace does not, and the glob matches nothing
# at all. Both are the harness dying before it checked anything.
mkdir -p "$work/empty-run/runs/20260815-075347"
summarise empty-run "$work/empty-run/runs"
expect_status 1 empty-run
grep -q "no trace at" "$rendered" || fail "empty-run: no missing-trace line"
summarise no-glob "$work/no-glob/runs"
expect_status 1 no-glob
grep -q "nothing under" "$rendered" || fail "no-glob: no missing-directory line"
# A property this summary does not know about is a summary that silently counts
# six of seven properties, which is the bug one level up.
drifted="$(plant drift replay-ui-every-property.jsonl)"
sed 's/badgeCountMatchesThePanel/badgeAgreesWithThePanel/g' \
"$testdata/replay-ui-every-property.jsonl" > "$drifted/20260815-075347/trace.jsonl"
summarise drift "$drifted"
expect_status 1 drift
grep -q "these counts cannot be trusted" "$rendered" || fail "drift: no untrusted verdict"
# Same for a reading the spec no longer declares: the count would quietly go to
# zero and read as a UI that stopped rendering.
sed 's/extract("violationBadges"/extract("violationCounters"/' \
"$here/../../replay-ui/sanderling/spec.ts" > "$work/renamed-spec.ts"
rendered="$work/renamed.md"
: > "$rendered"
status=0
GITHUB_STEP_SUMMARY="$rendered" SPEC="$work/renamed-spec.ts" \
bash -eo pipefail "$script" "$(plant renamed replay-ui-real-run.jsonl)" \
>/dev/null 2>&1 || status=$?
expect_status 1 renamed
grep -q "no longer declares violationBadges" "$rendered" || fail "renamed: no drift verdict"
if [ "$failed" = 0 ]; then
echo "replay-ui-summary.sh: ok"
fi
exit "$failed"
+222
View File
@@ -0,0 +1,222 @@
#!/usr/bin/env bash
# Reads the replay-ui dogfood trace and reports, per property, how many steps
# that property actually judged. Kept out of the workflow YAML so it can be run
# by hand against a local run:
#
# GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood
#
# `sanderling test` exiting 0 says only that no property returned false. Every
# property in replay-ui/sanderling/spec.ts declines to judge when a reading it
# needs is absent, which is right individually and useless in aggregate: a run
# that never rendered the step page returns false nowhere and exits 0, so
# checked-and-clean and checked-nothing arrive at the same green tick. Section 8
# of docs/development/design-principles.md is the rule this leg was breaking.
set -euo pipefail
root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
output="${1:-runs/dogfood}"
spec="${SPEC:-$root/replay-ui/sanderling/spec.ts}"
summary="${GITHUB_STEP_SUMMARY:-/dev/null}"
shopt -s nullglob
run_dirs=("$output"/*/)
shopt -u nullglob
{
echo "### replay-ui dogfood"
echo
echo "- seed \`${SEED:-unset}\`, budget ${MAX_STEPS:-unset} steps"
} >> "$summary"
if [ ${#run_dirs[@]} -eq 0 ]; then
echo "- **nothing under \`$output/\`**: the run never started, so no property was ever evaluated" >> "$summary"
echo "replay-ui: no run directory under $output/, so there is nothing to judge" >&2
exit 1
fi
trace="${run_dirs[${#run_dirs[@]} - 1]}trace.jsonl"
if [ ! -f "$trace" ]; then
echo "- **no trace at \`$trace\`**: the run wrote no steps, so no property was ever evaluated" >> "$summary"
echo "replay-ui: $trace does not exist, so there is nothing to judge" >&2
exit 1
fi
TRACE="$trace" SPEC="$spec" python3 - >> "$summary" <<'PY'
import json
import os
import re
import sys
trace_path = os.environ["TRACE"]
spec_path = os.environ["SPEC"]
def toolbar(values):
reading = values.get("toolbar")
return reading if isinstance(reading, dict) else None
def numbered(reading, key):
return reading is not None and reading.get(key) is not None
def listing(values, name):
reading = values.get(name)
return reading if isinstance(reading, list) else []
# One entry per property in replay-ui/sanderling/spec.ts, holding the guard that
# property opens with. The trace records extractor values, not verdicts, so
# counting the steps a property really judged means restating its guard here,
# and that restatement is the risk this file carries: a guard that drifts from
# its property would report evidence that does not exist. The two checks at the
# bottom make the two drifts that CAN be caught loud rather than silent.
#
# noUncaughtExceptions carries no guard on purpose: it compares a count, not an
# element, so it judges every step the verifier accepted.
GUARDS = {
"noUncaughtExceptions": [],
"selectedStepIsInRange": [
("a toolbar reporting a step and a step count",
lambda c, p: numbered(toolbar(c), "step") and numbered(toolbar(c), "stepCount")),
],
"exactlyOneStepIsSelected": [
("a row in the step list", lambda c, p: len(listing(c, "stepRows")) > 0),
],
"stepCountMatchesTheList": [
("a toolbar reporting a step count", lambda c, p: numbered(toolbar(c), "stepCount")),
("a row in the step list", lambda c, p: len(listing(c, "stepRows")) > 0),
],
"screenshotShowsTheSelectedStep": [
("a toolbar reporting a step", lambda c, p: numbered(toolbar(c), "step")),
("a screenshot in the before panel",
lambda c, p: c.get("beforeScreenshotStep") is not None),
],
"switchingTabsKeepsTheStep": [
("a tab selection that changed from the step before",
lambda c, p: p is not None and p.get("activeTabs") != c.get("activeTabs")),
("a toolbar on both steps",
lambda c, p: p is not None and toolbar(p) is not None and toolbar(c) is not None),
],
"badgeCountMatchesThePanel": [
("a violations badge on screen",
lambda c, p: len(listing(c, "violationBadges")) > 0
and listing(c, "violationBadges")[0] is not None),
("a violations panel on screen",
lambda c, p: len(listing(c, "violationPanelCounts")) > 0),
],
}
# A zero here is the run failing to render, not the fuzzer getting unlucky:
# every one of these needs only that the step page came up, which it does on the
# first step of any working run. The other three are left to be reported. Two of
# them are trajectory-dependent - switchingTabsKeepsTheStep needs a tab switch
# between consecutive steps, badgeCountMatchesThePanel needs the fuzzer to land
# on a violating step AND open the violations tab there - and a gate that
# convicts on an unlucky seed reports a regression it has not found.
MUST_RENDER = (
"selectedStepIsInRange",
"exactlyOneStepIsSelected",
"stepCountMatchesTheList",
"screenshotShowsTheSelectedStep",
)
steps = 0
verified = 0
violating = 0
malformed = 0
property_names = set()
judged = {name: 0 for name in GUARDS}
met_once = {name: [0] * len(conditions) for name, conditions in GUARDS.items()}
# The trace records only the extractors that changed at a step, so an
# extractor's value at any step is the last change recorded for it, starting
# from null. Steps the verifier skipped advance nothing and are not evaluations.
values = {}
previous = None
with open(trace_path, encoding="utf-8", errors="replace") as lines:
for line in lines:
if not line.strip():
continue
try:
step = json.loads(line)
except ValueError:
malformed += 1
continue
steps += 1
property_names |= set((step.get("residuals") or {}).keys())
if step.get("violations"):
violating += 1
if step.get("skipped_verification") or step.get("transitional"):
continue
for name, change in (step.get("extractor_changes") or {}).items():
values[name] = change.get("curr")
verified += 1
for name, conditions in GUARDS.items():
met = [condition(values, previous) for _, condition in conditions]
if all(met):
judged[name] += 1
for index, held in enumerate(met):
met_once[name][index] += 1 if held else 0
previous = dict(values)
report = []
report.append("- %d steps recorded, %d verified, %d with violations"
% (steps, verified, violating))
if malformed:
report.append("- **%d unreadable line(s)** in %s" % (malformed, trace_path))
report.append("")
report.append("- judged: the property compared real values. declined: a reading it "
"needs was absent, so it returned true without checking anything.")
report.append("")
report.append("| property | judged | declined |")
report.append("| --- | --- | --- |")
for name in GUARDS:
count = judged[name]
report.append("| %s | %s | %d |"
% (name, count if count else "**0**", verified - count))
report.append("")
for name, conditions in GUARDS.items():
if judged[name] or not verified:
continue
absent = [label for index, (label, _) in enumerate(conditions) if not met_once[name][index]]
report.append("- `%s` judged nothing on this run: no step had %s"
% (name, ", nor ".join(absent) if absent else "what its guard needs"))
blind = None
if not verified:
blind = "%s records %d step(s) and not one of them was verified" % (trace_path, steps)
else:
silent = [name for name in MUST_RENDER if not judged[name]]
if silent:
blind = ("%s declined on every step, so the step page never rendered and the "
"exit code is not evidence about the replay UI" % ", ".join(silent))
drift = []
if property_names and property_names != set(GUARDS):
drift.append("the spec's properties are %s but this summary counts %s"
% (", ".join(sorted(property_names)), ", ".join(sorted(GUARDS))))
try:
with open(spec_path, encoding="utf-8") as handle:
declared = set(re.findall(r'extract\(\s*"([^"]+)"', handle.read()))
except OSError as error:
drift.append("could not read %s to check its readings still exist: %s" % (spec_path, error))
declared = None
if declared is not None:
gone = sorted({"toolbar", "stepRows", "beforeScreenshotStep", "activeTabs",
"violationBadges", "violationPanelCounts"} - declared)
if gone:
drift.append("%s no longer declares %s, so the counts above describe readings "
"that do not exist" % (spec_path, ", ".join(gone)))
if blind:
report.append("- **this run checked nothing**: %s" % blind)
for reason in drift:
report.append("- **these counts cannot be trusted**: %s" % reason)
print("\n".join(report))
for reason in ([blind] if blind else []) + drift:
print("replay-ui: %s" % reason, file=sys.stderr)
sys.exit(1 if blind or drift else 0)
PY
@@ -0,0 +1,4 @@
{"extractor_changes":{"activeTabs":{"curr":"screenshot,screenshot","prev":null},"beforeScreenshotStep":{"curr":1,"prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false}],"prev":null},"toolbar":{"curr":{"step":1,"stepCount":5},"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":1,"timestamp":"2026-08-15T07:53:47Z"}
{"extractor_changes":{"activeTabs":{"curr":"violations,screenshot","prev":"screenshot,screenshot"},"beforeScreenshotStep":{"curr":4,"prev":1},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":true,"step":4,"violating":true},{"active":false,"step":5,"violating":false}],"prev":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false}]},"toolbar":{"curr":{"step":4,"stepCount":5},"prev":{"step":1,"stepCount":5}},"violationBadges":{"curr":[1],"prev":[]},"violationPanelCounts":{"curr":[1],"prev":[]}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":2,"timestamp":"2026-08-15T07:53:47Z"}
{"extractor_changes":{"activeTabs":{"curr":"hierarchy,screenshot","prev":"violations,screenshot"},"violationBadges":{"curr":[],"prev":[1]},"violationPanelCounts":{"curr":[],"prev":[1]}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":3,"timestamp":"2026-08-15T07:53:47Z"}
{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"skipped_verification":true,"step":4,"timestamp":"2026-08-15T07:53:47Z"}
@@ -0,0 +1,4 @@
{"extractor_changes":{"activeTabs":{"curr":"","prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[],"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":1,"timestamp":"2026-08-15T07:53:47Z"}
{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":2,"timestamp":"2026-08-15T07:53:47Z"}
{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":3,"timestamp":"2026-08-15T07:53:47Z"}
{"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"op":"true"}},"step":4,"timestamp":"2026-08-15T07:53:47Z"}
@@ -0,0 +1,10 @@
{"extractor_changes":{"activeTabs":{"curr":"screenshot,screenshot","prev":null},"beforeScreenshotStep":{"curr":1,"prev":null},"extractor_0":{"curr":0,"prev":null},"extractor_1":{"curr":0,"prev":null},"stepRows":{"curr":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":null},"toolbar":{"curr":{"step":1,"stepCount":25},"prev":null},"violationBadges":{"curr":[],"prev":null},"violationPanelCounts":{"curr":[],"prev":null}},"metrics":{"cpu_percent":0,"heap_bytes":2771019,"total_memory_bytes":3893507},"next_action":{"kind":"Tap","selector":"data-testid:tab","tap_point":{"x":774,"y":69},"x":774,"y":69},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":1,"timestamp":"2026-08-15T07:53:47.883914163Z"}
{"extractor_changes":{"activeTabs":{"curr":"screenshot,hierarchy","prev":"screenshot,screenshot"}},"metrics":{"cpu_percent":0,"heap_bytes":3490520,"total_memory_bytes":6061916},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":2,"timestamp":"2026-08-15T07:53:48.356557191Z"}
{"extractor_changes":{"activeTabs":{"curr":"screenshot,properties","prev":"screenshot,hierarchy"},"violationPanelCounts":{"curr":[0],"prev":[]}},"metrics":{"cpu_percent":0,"heap_bytes":3737998,"total_memory_bytes":6595962},"next_action":{"kind":"Tap","resolved_bounds":{"height":109,"width":31,"x":337,"y":308},"selector":"desc:select step 12","tap_point":{"x":352,"y":362},"x":353,"y":363},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/1","step":3,"timestamp":"2026-08-15T07:53:48.584625848Z"}
{"extractor_changes":{"beforeScreenshotStep":{"curr":12,"prev":1},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":true,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":12,"stepCount":25},"prev":{"step":1,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":3535304,"total_memory_bytes":7911280},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":4,"timestamp":"2026-08-15T07:53:48.820882198Z"}
{"extractor_changes":{"beforeScreenshotStep":{"curr":13,"prev":12},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":true,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":13,"stepCount":25},"prev":{"step":12,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":4913839,"total_memory_bytes":8552983},"next_action":{"duration_millis":250,"from_x":98,"from_y":149,"kind":"Swipe","to_x":98,"to_y":749},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/13","step":5,"timestamp":"2026-08-15T07:53:49.038789589Z"}
{"metrics":{"cpu_percent":0,"heap_bytes":4615419,"total_memory_bytes":8970775},"next_action":{"key":"left","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/13","step":6,"timestamp":"2026-08-15T07:53:49.251047338Z"}
{"extractor_changes":{"beforeScreenshotStep":{"curr":12,"prev":13},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":true,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":12,"stepCount":25},"prev":{"step":13,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":3996016,"total_memory_bytes":8970784},"next_action":{"kind":"Tap","selector":"data-testid:tab","tap_point":{"x":670,"y":69},"x":670,"y":69},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":7,"timestamp":"2026-08-15T07:53:49.451285818Z"}
{"metrics":{"cpu_percent":0,"heap_bytes":4830079,"total_memory_bytes":8974523},"next_action":{"kind":"Tap","selector":"data-testid:step-row","tap_point":{"x":170,"y":256},"x":170,"y":256},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":8,"timestamp":"2026-08-15T07:53:49.658417939Z"}
{"metrics":{"cpu_percent":0,"heap_bytes":3772499,"total_memory_bytes":8974523},"next_action":{"kind":"Tap","selector":"data-testid:step-row","tap_point":{"x":170,"y":363},"x":170,"y":363},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/12","step":9,"timestamp":"2026-08-15T07:53:49.865911674Z"}
{"extractor_changes":{"beforeScreenshotStep":{"curr":6,"prev":12},"stepRows":{"curr":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":true,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":false,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}],"prev":[{"active":false,"step":1,"violating":false},{"active":false,"step":2,"violating":false},{"active":false,"step":3,"violating":false},{"active":false,"step":4,"violating":true},{"active":false,"step":5,"violating":false},{"active":false,"step":6,"violating":false},{"active":false,"step":7,"violating":false},{"active":false,"step":8,"violating":false},{"active":false,"step":9,"violating":false},{"active":false,"step":10,"violating":false},{"active":false,"step":11,"violating":false},{"active":true,"step":12,"violating":false},{"active":false,"step":13,"violating":false},{"active":false,"step":14,"violating":false},{"active":false,"step":15,"violating":false},{"active":false,"step":16,"violating":false},{"active":false,"step":17,"violating":false},{"active":false,"step":18,"violating":false},{"active":false,"step":19,"violating":false},{"active":false,"step":20,"violating":false},{"active":false,"step":21,"violating":false},{"active":false,"step":22,"violating":false},{"active":false,"step":23,"violating":false},{"active":false,"step":24,"violating":false},{"active":false,"step":25,"violating":false}]},"toolbar":{"curr":{"step":6,"stepCount":25},"prev":{"step":12,"stepCount":25}}},"metrics":{"cpu_percent":0,"heap_bytes":4783792,"total_memory_bytes":8974552},"next_action":{"key":"right","kind":"PressKey"},"residuals":{"badgeCountMatchesThePanel":{"op":"true"},"exactlyOneStepIsSelected":{"op":"true"},"noUncaughtExceptions":{"op":"true"},"screenshotShowsTheSelectedStep":{"op":"true"},"selectedStepIsInRange":{"op":"true"},"stepCountMatchesTheList":{"op":"true"},"switchingTabsKeepsTheStep":{"name":"p6","op":"predicate"}},"screen":"/runs/20260815-075332/steps/6","step":10,"timestamp":"2026-08-15T07:53:50.075785262Z"}
+6 -12
View File
@@ -121,20 +121,14 @@ jobs:
SEED: ${{ inputs.seed }} SEED: ${{ inputs.seed }}
MAX_STEPS: ${{ inputs.max-steps }} MAX_STEPS: ${{ inputs.max-steps }}
# Exit 0 above means no property returned false. It does not mean any
# property was ever evaluated against real content: they all decline to
# judge when the elements they read are absent, so a run that never
# rendered the step page is green and worthless. This step is what tells
# the two apart, and it fails the job when nothing was judged.
- name: Summarise - name: Summarise
if: always() if: always()
run: | run: .github/scripts/replay-ui-summary.sh runs/dogfood
{
echo "### replay-ui dogfood"
echo
echo "- seed \`$SEED\`, budget $MAX_STEPS steps"
for dir in runs/dogfood/*/; do
[ -f "$dir/trace.jsonl" ] || continue
steps=$(wc -l < "$dir/trace.jsonl" | tr -d ' ')
violations=$(grep -c '"violations":\[' "$dir/trace.jsonl" || true)
echo "- $steps steps recorded, $violations step(s) with violations"
done
} >> "$GITHUB_STEP_SUMMARY"
env: env:
SEED: ${{ inputs.seed }} SEED: ${{ inputs.seed }}
MAX_STEPS: ${{ inputs.max-steps }} MAX_STEPS: ${{ inputs.max-steps }}
+7 -2
View File
@@ -29,7 +29,7 @@ WEB_DIST := replay-ui/dist
GOLINES := $(shell $(GO) env GOPATH)/bin/golines GOLINES := $(shell $(GO) env GOPATH)/bin/golines
.PHONY: bootstrap proto sidecar sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift .PHONY: bootstrap proto sidecar sanderling sanderling-web sanderling-android sanderling-ios install test test-go test-browser test-companion test-kotlin test-spec-api test-ci-scripts spec-typecheck web-test web-typecheck web-build web-dev replay-dev docs clean release-cli release-npm-dry fmt fmt-go fmt-kotlin fmt-ts fmt-swift
bootstrap: bootstrap:
$(GO) mod download $(GO) mod download
@@ -117,7 +117,7 @@ fmt-ts:
fmt-swift: fmt-swift:
xcrun swift-format format -i -r companion/Sources xcrun swift-format format -i -r companion/Sources
test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test test: test-go test-kotlin spec-typecheck test-spec-api web-typecheck web-test test-ci-scripts
test-go: test-go:
$(GO) test $(GO_PACKAGES) $(GO) test $(GO_PACKAGES)
@@ -141,6 +141,11 @@ test-companion: $(COMPANION_EMBED) $(RUNNER_EMBED)
test-kotlin: test-kotlin:
ANDROID_HOME=$(ANDROID_HOME) $(GRADLE) :sidecar:test ANDROID_HOME=$(ANDROID_HOME) $(GRADLE) :sidecar:test
# The CI scripts that read a trace and decide whether a green leg is
# evidence. bash and python3 only, which is all a runner has.
test-ci-scripts:
.github/scripts/replay-ui-summary-test.sh
test-spec-api: test-spec-api:
cd pkg/spec && npm test --silent cd pkg/spec && npm test --silent
+18
View File
@@ -136,6 +136,24 @@ hold for any trace and need no recalibrating when the fixture changes. The
seventh is the stock `noUncaughtExceptions`, which asks nothing of the panels seventh is the stock `noUncaughtExceptions`, which asks nothing of the panels
and only fails if the UI throws. Any violation fails the job. and only fails if the UI throws. Any violation fails the job.
So does a run that judged nothing. Exit 0 says no property returned false, which
is not the same as any property having been evaluated: each one declines to
judge when the elements it reads are absent, so a run where the trace failed to
serve, or where the fuzzer sat on the run list, renders nothing and passes.
`.github/scripts/replay-ui-summary.sh` reads the trace and puts a per-property
count of judged against declined steps in the job summary. Run it by hand with
`GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood`.
It fails the job when any of the four properties that need nothing beyond the
step page having rendered - `selectedStepIsInRange`, `exactlyOneStepIsSelected`,
`stepCountMatchesTheList`, `screenshotShowsTheSelectedStep` - judged nothing at
all. The other three are reported and not gated, because a zero on them is a
seed getting unlucky rather than a broken leg: `switchingTabsKeepsTheStep` needs
a tab switch between consecutive steps, and `badgeCountMatchesThePanel` needs the
fuzzer to land on a violating step and open the violations tab in that same
step. On the first run measured this way (seed 3, 80 steps) that last one judged
nothing at all, so the fixture reaches it far too rarely to be worth gating on.
## Reading a failure ## Reading a failure
Both workflows upload their run directories as artifacts, and write the step Both workflows upload their run directories as artifacts, and write the step