Merge branch 'one-examples-workflow' into correctness-and-spec-skills

This commit is contained in:
pj committed 2026-08-16 01:28:11 +05:30
commit a7c5193615
14 files changed
+786 -468

No files matched your search

+81
View File
@@ -0,0 +1,81 @@
name: folio app
description: Install the toolchain folio needs on one platform, and build the app there.
inputs:
platform:
description: android, ios or web
required: true
runs:
using: composite
steps:
- name: Set up the JDKs
uses: actions/setup-java@v5
with:
distribution: temurin
# The metro gradle plugin folio builds with needs a 21 runtime; the
# sidecar toolchain pins 17. Both are installed so gradle can pick.
java-version: |
17
21
# The iOS app builds its Kotlin framework through the folio gradle project,
# which configures :app:androidApp, so this is needed off Android too.
# `make sanderling-android` wants it as well, for the sidecar JAR.
- name: Set up Android SDK
if: inputs.platform != 'web'
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
- name: Cache Gradle
if: inputs.platform != 'ios'
uses: actions/cache@v6
with:
path: |
~/.gradle/caches
~/.gradle/wrapper
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
restore-keys: |
folio-gradle-${{ runner.os }}-
# idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so
# it has to be named by its full tap path. xcodegen and just are core.
- name: Install idb-companion, xcodegen and just
if: inputs.platform == 'ios'
shell: bash
run: brew install facebook/fb/idb-companion xcodegen just
# Both asset tarballs are built by the prepare scripts, and the runner
# bundle is an xcodebuild of companion/Sources. Keyed on the scripts and
# the versions the Makefile embeds, so a later run reuses them. This has to
# land before `make sanderling-ios`, which is what consumes them.
- name: Cache the companion and runner bundles
if: inputs.platform == 'ios'
uses: actions/cache@v6
with:
path: |
internal/driver/ioscompanion/companionassets/assets
internal/driver/ioscompanion/runnerassets/assets
key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }}
# Without this the emulator falls back to software rendering and every
# step costs several seconds.
- name: Enable KVM
if: inputs.platform == 'android'
shell: bash
run: |
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \
| sudo tee /etc/udev/rules.d/99-kvm4all.rules
sudo udevadm control --reload-rules
sudo udevadm trigger --name-match=kvm
- name: Build the folio APK
if: inputs.platform == 'android'
shell: bash
working-directory: examples/folio
run: ./gradlew :app:androidApp:assembleDebug
- name: Build the folio wasmJs app
if: inputs.platform == 'web'
shell: bash
working-directory: examples/folio
run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution
@@ -0,0 +1,31 @@
name: folio on a simulator
description: Boot an iOS simulator, install folio on it, and leave the app stopped.
inputs:
device:
description: simulator device name
default: iPhone 16 Pro
runs:
using: composite
steps:
- name: Boot a simulator
shell: bash
run: |
xcrun simctl boot "$IOS_DEVICE" || true
xcrun simctl bootstatus "$IOS_DEVICE" -b
env:
IOS_DEVICE: ${{ inputs.device }}
- name: Build and install folio
shell: bash
working-directory: examples/folio
run: just ios
env:
IOS_DEVICE: ${{ inputs.device }}
# `just ios` leaves the app running, and the run's first act is to clear
# its state. Stopping it here means the run always opens the same way.
- name: Stop the app before the run
shell: bash
run: xcrun simctl terminate booted app.folio || true
@@ -0,0 +1,30 @@
name: headless chrome
description: Install Chrome and prove it starts headless before a driver depends on it.
runs:
using: composite
steps:
# stable is setup-chrome v2's own default, spelled out so a new release of
# the action cannot move the browser these jobs drive. The alternative it
# offers is Chrome for Testing latest, which tracks ahead of the channel
# users run.
- uses: browser-actions/setup-chrome@2e1d749697dd1612b833dba4a722266286fbefcd # v2.1.2
with:
chrome-version: stable
# Ubuntu 24.04 (current ubuntu-latest) restricts unprivileged user
# namespaces via AppArmor, which stops headless Chrome from starting even
# with --no-sandbox: the process launches but never opens its DevTools
# socket. Re-enable them so the driver's Chrome can come up.
- name: Allow Chrome under unprivileged user namespaces
shell: bash
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
# Fail here with Chrome's own stderr if the browser can't launch, instead
# of letting the driver report an opaque DevTools timeout downstream.
- name: Verify headless Chrome starts
shell: bash
run: |
chrome --version
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
--dump-dom 'data:text/html,<title>ok</title>'
@@ -0,0 +1,55 @@
name: replay ui fixture
description: Record a trace with sanderling, then serve it with sanderling replay.
outputs:
url:
description: the step page of the served run, for a spec to drive
value: ${{ steps.serve.outputs.url }}
runs:
using: composite
steps:
# A trace with a violation and uncaught exceptions in it, so the UI has
# something to render in every panel the spec looks at. No
# --exit-on-violation here: the run is the fixture, and stopping it at the
# first violation would leave a four-step trace to fuzz.
- name: Record a fixture trace
shell: bash
run: |
python3 -m http.server 8792 --bind 127.0.0.1 \
--directory test/browser/testdata/throwing &
ready=""
for _ in $(seq 1 30); do
curl -sf http://127.0.0.1:8792/ >/dev/null && { ready=1; break; }
sleep 1
done
if [ -z "$ready" ]; then
echo "the fixture http server never answered on 127.0.0.1:8792" >&2
exit 1
fi
./bin/sanderling test \
--platform web \
--spec test/browser/testdata/throwing/spec.ts \
--bundle-id http://127.0.0.1:8792/ \
--duration 5m --max-steps 25 --seed 7 \
--output runs/fixture
- name: Serve the trace with sanderling replay
id: serve
shell: bash
run: |
# Flags before the positional argument: Go's flag package stops
# parsing at the first non-flag word.
./bin/sanderling replay --port 8793 --no-open runs/fixture &
ready=""
for _ in $(seq 1 30); do
curl -sf http://127.0.0.1:8793/api/runs >/dev/null && { ready=1; break; }
sleep 1
done
if [ -z "$ready" ]; then
echo "sanderling replay never served /api/runs on 127.0.0.1:8793" >&2
exit 1
fi
run_id="$(ls runs/fixture | head -1)"
echo "url=http://127.0.0.1:8793/runs/$run_id/steps/1" >> "$GITHUB_OUTPUT"
curl -sf "http://127.0.0.1:8793/runs/$run_id/steps/1" >/dev/null
+284
View File
@@ -0,0 +1,284 @@
#!/usr/bin/env bash
# Drives folio-run.sh through a stubbed `sanderling` binary and checks the
# verdict it reaches from each shape of trace. What is under test is the
# classification, not the fuzzer: the stub writes the trace the run would have
# written and exits the code the run would have exited.
#
# folio-run.sh is invoked as `bash -eo pipefail -c <script>`, which is what a
# `run:` block does: -e is set on the shell that calls the script, and does not
# cross the shebang into it. Running the script itself under -e would kill it at
# the first non-zero `sanderling test`, which is the exit code it exists to read.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
script="$here/folio-run.sh"
spec="$here/../../examples/folio/sanderling/spec.ts"
work="$(mktemp -d)"
trap 'rm -rf "$work"' EXIT
failed=0
case_dir=""
status=0
spec_override=""
run() { # <case> <platform> <stub exit code> [none] ; trace lines on stdin
local name="$1" platform="$2" code="$3" trace_mode="${4:-file}"
case_dir="$work/$name"
mkdir -p "$case_dir"
cat > "$case_dir/trace.jsonl"
printf '#!/usr/bin/env bash\n' > "$case_dir/sanderling"
cat >> "$case_dir/sanderling" <<'STUB'
printf '%s\n' "$@" > "$STUB_ARGV"
out="" ; prev=""
for arg in "$@"; do
[ "$prev" = "--output" ] && out="$arg"
prev="$arg"
done
if [ "$STUB_TRACE_MODE" = file ]; then
mkdir -p "$out/20260815-101500"
cp "$STUB_TRACE" "$out/20260815-101500/trace.jsonl"
fi
exit "$STUB_CODE"
STUB
chmod +x "$case_dir/sanderling"
status=0
cd "$case_dir"
STUB_ARGV="$case_dir/argv" STUB_TRACE="$case_dir/trace.jsonl" \
STUB_TRACE_MODE="$trace_mode" STUB_CODE="$code" \
SANDERLING="$case_dir/sanderling" SPEC="${spec_override:-$spec}" \
GITHUB_STEP_SUMMARY="$case_dir/summary.md" \
SEED=7 MAX_STEPS=240 DURATION=20m \
bash -eo pipefail -c "'$script' '$platform'" \
> "$case_dir/out" 2> "$case_dir/err" || status=$?
cd "$here"
spec_override=""
}
fail() { echo "FAIL: $*" >&2; failed=1; }
expect_status() { # <want> <case>
[ "$status" = "$1" ] || fail "$2: exit $status, want $1"
}
# grep reads both files directly: piping `cat` into `grep -q` returns 141 under
# pipefail, because grep leaves on the first match and cat takes the SIGPIPE.
expect_says() { # <text> <case>
grep -qF -- "$1" "$case_dir/out" "$case_dir/err" || fail "$2: nothing said '$1'"
}
expect_silent() { # <text> <case>
if grep -qF -- "$1" "$case_dir/out" "$case_dir/err"; then
fail "$2: should not have said '$1'"
fi
}
expect_summary() { # <text> <case>
grep -qF -- "$1" "$case_dir/summary.md" || fail "$2: summary has no '$1'"
}
expect_argv() { # <text> <case>
grep -qxF -- "$1" "$case_dir/argv" || fail "$2: '$1' never reached the binary"
}
convicting='{"violations":["submitCommitsOneTransactionPerAction"],"witnesses":{"submitCommitsOneTransactionPerAction":{"is_error":false,"reason":"one tap, two rows"}}}'
unrelated='{"violations":["newAccountBalanceIsZero"],"witnesses":{"newAccountBalanceIsZero":{"is_error":false,"reason":"opened at 4.00"}}}'
threw='{"violations":["submitMovesBalanceByAtMostTypedAmount"],"witnesses":{"submitMovesBalanceByAtMostTypedAmount":{"is_error":true,"reason":"TypeError: cannot read text of undefined"}}}'
on_txn='{"route":"AddTransactionScreen","violations":[]}'
off_txn='{"route":"HomeScreen","violations":[]}'
# --- the flags the calibrated runs were measured with reach the binary --------
run argv-ios ios 2 <<TRACE
$on_txn
$convicting
TRACE
expect_status 0 argv-ios
expect_argv "--exit-on-violation" argv-ios
expect_argv "--platform" argv-ios
expect_argv "ios" argv-ios
expect_argv "--seed" argv-ios
expect_argv "7" argv-ios
expect_argv "240" argv-ios
expect_argv "20m" argv-ios
expect_argv "iPhone 16 Pro" argv-ios
run argv-android android 2 <<TRACE
$on_txn
$convicting
TRACE
expect_status 0 argv-android
expect_argv "--exit-on-violation" argv-android
expect_argv "app.folio" argv-android
expect_argv "examples/folio/app/androidApp/build/outputs/apk/debug/androidApp-debug.apk" argv-android
# --- ios and web: a conviction is required -----------------------------------
run convicted ios 2 <<TRACE
$on_txn
$convicting
TRACE
expect_status 0 convicted
expect_says "found the submit bug in 2 steps (submitCommitsOneTransactionPerAction)" convicted
expect_summary "- convicted on: submitCommitsOneTransactionPerAction" convicted
# Exit 2 with a violation of a real but ungated property is not this leg's bug.
run other-violation ios 2 <<TRACE
$on_txn
$unrelated
TRACE
expect_status 1 other-violation
expect_says "not the double-submit this leg gates on" other-violation
expect_summary "- also violated: newAccountBalanceIsZero" other-violation
# A predicate that threw reaches exit 2 by the identical path, and is not a
# verdict about folio at all.
run threw-ios ios 2 <<TRACE
$on_txn
$threw
TRACE
expect_status 1 threw-ios
expect_says "a predicate threw, so exit 2 is not a verdict about folio" threw-ios
expect_says "TypeError: cannot read text of undefined" threw-ios
expect_silent "found the submit bug" threw-ios
expect_summary "**a predicate threw**" threw-ios
# The bug is still in folio, so a clean run is the fuzzer no longer reaching it.
run clean-ios ios 0 <<TRACE
$on_txn
$on_txn
TRACE
expect_status 1 clean-ios
expect_says "the double-submit bug was NOT found in 2 steps" clean-ios
run harness-ios ios 3 <<TRACE
$on_txn
TRACE
expect_status 3 harness-ios
expect_says "the harness failed with exit 3" harness-ios
# --- android: a health gate, with a conviction as a bonus --------------------
run android-convicted android 2 <<TRACE
$on_txn
$convicting
TRACE
expect_status 0 android-convicted
expect_says "found the submit bug in 2 steps (a bonus, not required)" android-convicted
run android-other android 2 <<TRACE
$on_txn
$unrelated
TRACE
expect_status 0 android-other
expect_says "judging health only" android-other
run android-healthy android 0 <<TRACE
$on_txn
$on_txn
TRACE
expect_status 0 android-healthy
expect_says "healthy run over 2 steps, reached the transaction screen" android-healthy
# Health means it got to the screen the double-submit lives on. Anything less is
# a leg that proved nothing, however green the exit code.
run android-stalled android 0 <<TRACE
$off_txn
{"route":"LedgerScreen","violations":[]}
TRACE
expect_status 1 android-stalled
expect_says "never reached AddTransactionScreen over 2 steps" android-stalled
expect_says "routes the trace does record: HomeScreen,LedgerScreen" android-stalled
# The thrown-predicate check has to bite on android too: this is the leg whose
# gate is loose enough to swallow it.
run threw-android android 2 <<TRACE
$on_txn
$threw
TRACE
expect_status 1 threw-android
expect_says "a predicate threw" threw-android
expect_silent "judging health only" threw-android
run harness-android android 4 <<TRACE
$on_txn
TRACE
expect_status 4 harness-android
expect_says "the harness failed with exit 4" harness-android
# --- traces that are not there, or are there and say nothing -----------------
# Exit 2 and no run directory at all: the glob matches nothing, and an empty
# trace must not be judged as a folio that behaved.
run no-trace ios 2 none </dev/null
expect_status 1 no-trace
expect_says "wrote no trace" no-trace
expect_silent "Traceback" no-trace
run no-trace-android android 0 none </dev/null
expect_status 1 no-trace-android
expect_says "wrote no trace" no-trace-android
# A zero-byte trace: the file is there, so the missing-trace check passes and
# the classification has to survive reading nothing out of it.
run zero-byte ios 0 file </dev/null
expect_status 1 zero-byte
expect_says "NOT found in 0 steps" zero-byte
expect_silent "Traceback" zero-byte
# A line the recorder truncated is skipped, not fatal, and the violation on the
# readable line is still found.
run malformed ios 2 <<TRACE
$on_txn
{"violations":["submitCommitsOne
$convicting
TRACE
expect_status 0 malformed
expect_says "found the submit bug" malformed
expect_silent "Traceback" malformed
run bad-platform windows 0 none </dev/null
expect_status 64 bad-platform
expect_says "unknown platform: windows" bad-platform
# --- the gate names still exist in the spec they gate ------------------------
# Nothing but this check ties GATED_PROPERTIES to the spec. Rename a gated
# property and the classification matches nothing: ios and web report a
# different bug, android reclassifies a conviction as health and stays green.
sed 's/submitCommitsOneTransactionPerAction/submitCommitsOneTxnPerAction/g' \
"$spec" > "$work/renamed-spec.ts"
spec_override="$work/renamed-spec.ts"
run drift-renamed ios 2 <<TRACE
$on_txn
$convicting
TRACE
expect_status 1 drift-renamed
expect_says "no longer declares submitCommitsOneTransactionPerAction" drift-renamed
if [ -e "$case_dir/argv" ]; then
fail "drift-renamed: the run started before the gate was checked"
fi
spec_override="$work/absent-spec.ts"
run drift-missing ios 2 none </dev/null
expect_status 1 drift-missing
expect_says "cannot read" drift-missing
echo "export const notProperties = { a };" > "$work/shapeless-spec.ts"
spec_override="$work/shapeless-spec.ts"
run drift-shapeless ios 2 none </dev/null
expect_status 1 drift-shapeless
expect_says "declares no \`export const properties" drift-shapeless
# And the same check against the spec as it stands: this is the assertion that
# fails at `make test` when someone renames a property without moving the gate.
run gate-matches-spec ios 2 <<TRACE
$on_txn
$convicting
TRACE
expect_status 0 gate-matches-spec
expect_silent "no longer declares" gate-matches-spec
if [ "$failed" = 0 ]; then
echo "folio-run.sh: ok"
fi
exit "$failed"
+46 -6
View File
@@ -18,9 +18,49 @@ max_steps="${MAX_STEPS:-240}"
duration="${DURATION:-20m}" duration="${DURATION:-20m}"
sanderling="${SANDERLING:-./bin/sanderling}" sanderling="${SANDERLING:-./bin/sanderling}"
output="runs/folio-$platform" output="runs/folio-$platform"
spec="examples/folio/sanderling/spec.ts" spec="${SPEC:-examples/folio/sanderling/spec.ts}"
summary="${GITHUB_STEP_SUMMARY:-/dev/null}" summary="${GITHUB_STEP_SUMMARY:-/dev/null}"
# The two properties that state folio's double-submit. Anything else the spec
# proves false is a different finding, and this leg has nothing to say about it.
GATED_PROPERTIES="submitMovesBalanceByAtMostTypedAmount,submitCommitsOneTransactionPerAction"
# A gate is only as good as these names, and nothing else ties them to the spec.
# Rename a property there and the classification below matches nothing: ios and
# web blame the spec for finding a different bug, and android reclassifies a
# real conviction as "judging health only" and stays green. Checked before the
# run so a rename costs seconds rather than the whole budget.
SPEC="$spec" GATED="$GATED_PROPERTIES" SELF="$0" python3 - <<'PY' || exit 1
import os, re, sys
spec_path = os.environ["SPEC"]
gated = [name for name in os.environ["GATED"].split(",") if name]
try:
with open(spec_path, encoding="utf-8") as handle:
source = handle.read()
except OSError as error:
sys.exit("folio: cannot read %s to check the gated properties still exist: %s"
% (spec_path, error))
block = re.search(r"export\s+const\s+properties\s*=\s*\{(.*?)\}", source, re.S)
if block is None:
sys.exit("folio: %s declares no `export const properties = {...}`, so the gated "
"properties cannot be checked against it" % spec_path)
declared = set()
for entry in re.sub(r"//[^\n]*", "", block.group(1)).split(","):
name = entry.split(":")[0].strip()
if re.fullmatch(r"[A-Za-z_$][A-Za-z0-9_$]*", name):
declared.add(name)
missing = [name for name in gated if name not in declared]
if missing:
sys.exit("folio: %s no longer declares %s, so this leg gates on a property that "
"cannot be violated and every real conviction would read as a different "
"finding. Update GATED_PROPERTIES in %s."
% (spec_path, ", ".join(missing), os.environ["SELF"]))
PY
folio_args=(--bundle-id app.folio) folio_args=(--bundle-id app.folio)
case "$platform" in case "$platform" in
android) android)
@@ -89,14 +129,14 @@ esac
code=$? code=$?
run_dir="$(ls -d "$output"/*/ 2>/dev/null | tail -1)" run_dir="$(ls -d "$output"/*/ 2>/dev/null | tail -1)"
trace="${run_dir:-.}/trace.jsonl" # No run directory means no trace. Defaulting the directory to `.` here reads a
# stray ./trace.jsonl and reports it as this run's evidence, which is how a run
# that wrote nothing at all reached "found the submit bug" and exit 0.
trace=""
[ -n "$run_dir" ] && trace="${run_dir}trace.jsonl"
steps=0 steps=0
[ -f "$trace" ] && steps=$(wc -l < "$trace" | tr -d ' ') [ -f "$trace" ] && steps=$(wc -l < "$trace" | tr -d ' ')
# The two properties that state folio's double-submit. Anything else the spec
# proves false is a different finding, and this leg has nothing to say about it.
GATED_PROPERTIES="submitMovesBalanceByAtMostTypedAmount,submitCommitsOneTransactionPerAction"
# Exit 2 means "the run recorded a violation", and that is NOT the same as "the # Exit 2 means "the run recorded a violation", and that is NOT the same as "the
# run convicted folio". A predicate that THROWS is recorded as a violation too, # run convicted folio". A predicate that THROWS is recorded as a violation too,
# with is_error set and the thrown text as its reason, and it reaches exit 2 by # with is_error set and the thrown text as its reason, and it reaches exit 2 by
+1 -1
View File
@@ -3,7 +3,7 @@
# summary it renders. Run under the flags GitHub Actions uses for a `run:` # summary it renders. Run under the flags GitHub Actions uses for a `run:`
# block, because that is where a swallowed failure hides. # block, because that is where a swallowed failure hides.
# #
# testdata/replay-ui-real-run.jsonl is the first 10 steps of the dogfood run in # testdata/replay-ui-real-run.jsonl is the first 10 steps of the replay-ui run in
# actions run 31873049857 on master, with the per-step `hierarchy` dumps and the # actions run 31873049857 on master, with the per-step `hierarchy` dumps and the
# rowElements/tabElements extractors removed so the file stays readable. Nothing # rowElements/tabElements extractors removed so the file stays readable. Nothing
# else was touched. That run was green, and badgeCountMatchesThePanel judged # else was touched. That run was green, and badgeCountMatchesThePanel judged
+4 -4
View File
@@ -1,9 +1,9 @@
#!/usr/bin/env bash #!/usr/bin/env bash
# Reads the replay-ui dogfood trace and reports, per property, how many steps # Reads the replay-ui fuzzing trace and reports, per property, how many steps
# that property actually judged. Kept out of the workflow YAML so it can be run # that property actually judged. Kept out of the workflow YAML so it can be run
# by hand against a local run: # by hand against a local run:
# #
# GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood # GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/replay-ui
# #
# `sanderling test` exiting 0 says only that no property returned false. Every # `sanderling test` exiting 0 says only that no property returned false. Every
# property in replay-ui/sanderling/spec.ts declines to judge when a reading it # property in replay-ui/sanderling/spec.ts declines to judge when a reading it
@@ -14,7 +14,7 @@
set -euo pipefail set -euo pipefail
root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
output="${1:-runs/dogfood}" output="${1:-runs/replay-ui}"
spec="${SPEC:-$root/replay-ui/sanderling/spec.ts}" spec="${SPEC:-$root/replay-ui/sanderling/spec.ts}"
summary="${GITHUB_STEP_SUMMARY:-/dev/null}" summary="${GITHUB_STEP_SUMMARY:-/dev/null}"
@@ -23,7 +23,7 @@ run_dirs=("$output"/*/)
shopt -u nullglob shopt -u nullglob
{ {
echo "### replay-ui dogfood" echo "### the replay ui"
echo echo
echo "- seed \`${SEED:-unset}\`, budget ${MAX_STEPS:-unset} steps" echo "- seed \`${SEED:-unset}\`, budget ${MAX_STEPS:-unset} steps"
} >> "$summary" } >> "$summary"
+5 -24
View File
@@ -34,7 +34,7 @@ jobs:
java-version: "17" java-version: "17"
- name: Set up Android SDK - name: Set up Android SDK
uses: android-actions/setup-android@v4 uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
- name: Set up Node 22 - name: Set up Node 22
uses: actions/setup-node@v7 uses: actions/setup-node@v7
@@ -44,7 +44,7 @@ jobs:
cache-dependency-path: pkg/spec/package-lock.json cache-dependency-path: pkg/spec/package-lock.json
- name: Set up bun - name: Set up bun
uses: oven-sh/setup-bun@v2 uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with: with:
bun-version: "1.3.13" bun-version: "1.3.13"
@@ -100,7 +100,7 @@ jobs:
# JAVA_HOME after `make test` rather than installing both up front # JAVA_HOME after `make test` rather than installing both up front
# leaves every step above this one on exactly the JDK it ran on before. # leaves every step above this one on exactly the JDK it ran on before.
- name: Set up JDK 21 for folio - name: Set up JDK 21 for folio
uses: actions/setup-java@v4 uses: actions/setup-java@v5
with: with:
distribution: temurin distribution: temurin
java-version: "21" java-version: "21"
@@ -119,27 +119,8 @@ jobs:
go-version-file: go.mod go-version-file: go.mod
cache: true cache: true
# Pin stable: a dev Chromium's remote-debugging socket is flaky under - name: Set up headless Chrome
# the driver, even though the browser otherwise launches headless. uses: ./.github/actions/headless-chrome
- name: Set up Chrome
uses: browser-actions/setup-chrome@v2
with:
chrome-version: stable
# Ubuntu 24.04 (current ubuntu-latest) restricts unprivileged user
# namespaces via AppArmor, which stops headless Chrome from starting even
# with --no-sandbox: the process launches but never opens its DevTools
# socket. Re-enable them so the driver's Chrome can come up.
- name: Allow Chrome under unprivileged user namespaces
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
# Fail here with Chrome's own stderr if the browser can't launch, instead
# of letting the driver report an opaque DevTools timeout downstream.
- name: Verify headless Chrome starts
run: |
chrome --version
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
--dump-dom 'data:text/html,<title>ok</title>'
- name: Drive web fixtures through headless Chrome - name: Drive web fixtures through headless Chrome
run: make test-browser run: make test-browser
+5
View File
@@ -30,6 +30,11 @@ jobs:
- name: Build site - name: Build site
run: make docs run: make docs
# No include-hidden-files: v4 stopped uploading dot-files by default, and
# build/site has none. It is pandoc output plus a copy of docs/_assets,
# which holds three ordinary files. _assets is underscore-prefixed, not
# hidden, and deploy-pages serves the artifact without running Jekyll, so
# it needs no .nojekyll either.
- uses: actions/upload-pages-artifact@v5 - uses: actions/upload-pages-artifact@v5
with: with:
path: build/site path: build/site
+193
View File
@@ -0,0 +1,193 @@
name: examples
# Every example sanderling ships, fuzzed the same way: build sanderling for a
# platform, bring the target up, run a spec against it, classify the trace it
# wrote, upload the run. Only the bring-up differs, and that lives in the
# per-target actions under .github/actions/.
#
# Dispatch-only: these take tens of minutes and they demonstrate the product
# loop, they do not gate a merge.
#
# folio on ios and in the browser expect the bug: folio double-submits a
# transaction on a double tap, so the run is supposed to end with exit 2. Exit 0
# means the fuzzer stopped finding a bug that is still there; exit 1 means the
# harness broke. The two are worth telling apart, which is why
# --exit-on-violation exits 2 and not 1.
#
# folio on android is a health gate. It convicts in four runs out of five, which
# is real evidence but not a gate: the fifth would report a regression it had
# not found. Its budget is set so the conviction it usually gets is a bonus.
#
# the replay ui leg fuzzes sanderling's own replay UI, and any violation fails
# it. Its properties are cross-panel agreements that hold for any trace, so none
# of them needs recalibrating when the fixture changes.
on:
workflow_dispatch:
inputs:
targets:
description: which examples to fuzz
type: choice
options: [all, folio, android, ios, web, replay-ui]
default: all
seed:
description: seed override (0 = each target's calibrated seed)
default: "0"
max-steps:
description: step budget override (0 = each target's calibrated budget)
default: "0"
duration:
description: wall-clock budget override (empty = each target's calibrated budget)
default: ""
permissions:
contents: read
jobs:
plan:
runs-on: ubuntu-latest
permissions: {}
outputs:
examples: ${{ steps.pick.outputs.examples }}
steps:
# The matrix is built here rather than written out under strategy.matrix
# because a job-level `if:` cannot read the matrix context, so a static
# matrix has no way to leave a leg out. jq -c keeps the value on one line,
# which is what makes the $GITHUB_OUTPUT write below safe.
- name: Pick the examples to fuzz
id: pick
run: |
examples='[
{"target":"android","name":"folio on android","app":"folio",
"runs-on":"ubuntu-latest","timeout":90,"sanderling":"android",
"seed":"9","max-steps":"200","duration":"20m","artifact":"folio-android"},
{"target":"ios","name":"folio on ios","app":"folio",
"runs-on":"macos-15","timeout":90,"sanderling":"ios",
"seed":"7","max-steps":"240","duration":"20m","artifact":"folio-ios"},
{"target":"web","name":"folio in the browser","app":"folio",
"runs-on":"ubuntu-latest","timeout":60,"sanderling":"web","chrome":true,
"seed":"3","max-steps":"240","duration":"20m","artifact":"folio-web"},
{"target":"replay-ui","name":"the replay ui","app":"replay-ui",
"runs-on":"ubuntu-latest","timeout":45,"sanderling":"web","chrome":true,
"seed":"3","max-steps":"80","duration":"10m","artifact":"replay-ui-runs"}
]'
picked="$(jq -c --arg want "$TARGETS" \
'map(select($want == "all" or .target == $want or .app == $want))' \
<<<"$examples")"
if [ "$picked" = "[]" ]; then
echo "examples: '$TARGETS' selects no example, so this dispatch would run nothing" >&2
exit 1
fi
echo "examples=$picked" >> "$GITHUB_OUTPUT"
env:
TARGETS: ${{ inputs.targets }}
fuzz:
needs: plan
name: fuzz ${{ matrix.name }}
runs-on: ${{ matrix.runs-on }}
timeout-minutes: ${{ matrix.timeout }}
strategy:
# Each leg is its own evidence. One target failing must not cancel the
# others, which is how these ran as separate jobs.
fail-fast: false
matrix:
include: ${{ fromJSON(needs.plan.outputs.examples) }}
env:
SEED: ${{ inputs.seed != '0' && inputs.seed || matrix.seed }}
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || matrix.max-steps }}
DURATION: ${{ inputs.duration != '' && inputs.duration || matrix.duration }}
IOS_DEVICE: iPhone 16 Pro
steps:
- uses: actions/checkout@v7
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: go.mod
cache: true
- name: Set up bun
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: "1.3.13"
- name: Set up headless Chrome
if: matrix.chrome
uses: ./.github/actions/headless-chrome
- name: Build the folio app
if: matrix.app == 'folio'
uses: ./.github/actions/folio-app
with:
platform: ${{ matrix.target }}
# The UI the replay-ui spec drives is the one embedded in this binary, so
# the build has to come after any change to replay-ui/src.
- name: Build sanderling
run: make "sanderling-$SANDERLING"
env:
SANDERLING: ${{ matrix.sanderling }}
- name: Put folio on the simulator
if: matrix.target == 'ios'
uses: ./.github/actions/folio-simulator
- name: Serve a trace to fuzz
id: fixture
if: matrix.target == 'replay-ui'
uses: ./.github/actions/replay-ui-fixture
- name: Fuzz folio on an emulator
if: matrix.target == 'android'
uses: reactivecircus/android-emulator-runner@a421e43855164a8197daf9d8d40fe71c6996bb0d # v2.38.0
with:
api-level: 34
target: google_apis
arch: x86_64
emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim
disable-animations: true
script: .github/scripts/folio-run.sh android
- name: Fuzz folio
if: matrix.app == 'folio' && matrix.target != 'android'
run: .github/scripts/folio-run.sh "$TARGET"
env:
TARGET: ${{ matrix.target }}
# Inputs go through env rather than into the script text: a `${{ }}` is
# substituted before bash ever sees the line, so a seed of `$(id)` would
# run as a command.
- name: Fuzz the replay UI
if: matrix.target == 'replay-ui'
run: |
./bin/sanderling test \
--platform web \
--spec replay-ui/sanderling/spec.ts \
--bundle-id "$RUN_URL" \
--duration "$DURATION" \
--max-steps "$MAX_STEPS" \
--seed "$SEED" \
--exit-on-violation \
--output runs/replay-ui
env:
RUN_URL: ${{ steps.fixture.outputs.url }}
# Exit 0 above means no property returned false. It does not mean any
# property was ever evaluated against real content: they all decline to
# judge when the elements they read are absent, so a run that never
# rendered the step page is green and worthless. This step is what tells
# the two apart, and it fails the job when nothing was judged. folio's
# legs make the same call inside folio-run.sh, where the exit code it is
# judging is in scope.
- name: Classify the replay UI run
if: ${{ always() && matrix.target == 'replay-ui' }}
run: .github/scripts/replay-ui-summary.sh runs/replay-ui
- name: Upload the run
if: always()
uses: actions/upload-artifact@v7
with:
name: ${{ matrix.artifact }}
path: runs/
retention-days: 14
-277
View File
@@ -1,277 +0,0 @@
name: folio
# One spec, three platforms. Dispatch-only: each job boots a device or a
# browser, builds the folio app for that platform, and runs
# examples/folio/sanderling/spec.ts against it.
#
# web and ios are expect-the-bug jobs: folio double-submits a transaction on a
# double tap, so the run is supposed to end with exit 2. Exit 0 means the fuzzer
# stopped finding a bug that is still there; exit 1 means the harness broke. The
# two are worth telling apart, which is why --exit-on-violation exits 2 and not
# 1.
#
# android is a health gate. It convicts in four runs out of five, which is real
# evidence but not a gate: the fifth would report a regression it had not found.
# The budget is set so the conviction it usually gets is reported as a bonus.
on:
workflow_dispatch:
inputs:
platforms:
description: which legs to run
type: choice
options: [all, android, ios, web]
default: all
seed:
description: seed override (0 = each job's calibrated seed)
default: "0"
duration:
description: wall-clock budget per run
default: 20m
max-steps:
description: step budget override (0 = each job's calibrated budget)
default: "0"
permissions:
contents: read
jobs:
android:
timeout-minutes: 90
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'android' }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: go.mod
cache: true
- name: Set up the JDKs
uses: actions/setup-java@v5
with:
distribution: temurin
# The metro gradle plugin folio builds with needs a 21 runtime; the
# sidecar toolchain pins 17. Both are installed so gradle can pick.
java-version: |
17
21
- name: Set up Android SDK
uses: android-actions/setup-android@v4
- name: Set up bun
uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3.13"
- name: Cache Gradle
uses: actions/cache@v6
with:
path: |
~/.gradle/caches
~/.gradle/wrapper
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
restore-keys: |
folio-gradle-${{ runner.os }}-
# Without this the emulator falls back to software rendering and every
# step costs several seconds.
- name: Enable KVM
run: |
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \
| sudo tee /etc/udev/rules.d/99-kvm4all.rules
sudo udevadm control --reload-rules
sudo udevadm trigger --name-match=kvm
- name: Build the folio APK
run: ./gradlew :app:androidApp:assembleDebug
working-directory: examples/folio
- name: Build sanderling
run: make sanderling-android
- name: Fuzz folio on an emulator
uses: reactivecircus/android-emulator-runner@v2
with:
api-level: 34
target: google_apis
arch: x86_64
emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim
disable-animations: true
script: .github/scripts/folio-run.sh android
env:
SEED: ${{ inputs.seed != '0' && inputs.seed || '9' }}
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '200' }}
DURATION: ${{ inputs.duration }}
- name: Upload the run
if: always()
uses: actions/upload-artifact@v7
with:
name: folio-android
path: runs/
retention-days: 14
ios:
timeout-minutes: 90
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'ios' }}
runs-on: macos-15
steps:
- uses: actions/checkout@v7
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: go.mod
cache: true
- name: Set up bun
uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3.13"
# idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so
# it has to be named by its full tap path. xcodegen and just are core.
- name: Install idb-companion, xcodegen and just
run: brew install facebook/fb/idb-companion xcodegen just
- name: Set up the JDKs
uses: actions/setup-java@v5
with:
distribution: temurin
# The metro gradle plugin folio builds with needs a 21 runtime; the
# sidecar toolchain pins 17. Both are installed so gradle can pick.
java-version: |
17
21
# The iOS app builds its Kotlin framework through the folio gradle
# project, which configures :app:androidApp and so needs an Android SDK
# even on this leg.
- name: Set up Android SDK
uses: android-actions/setup-android@v4
# Both asset tarballs are built by the prepare scripts, and the runner
# bundle is an xcodebuild of companion/Sources. Keyed on the scripts and
# the versions the Makefile embeds, so a later run reuses them.
- name: Cache the companion and runner bundles
uses: actions/cache@v6
with:
path: |
internal/driver/ioscompanion/companionassets/assets
internal/driver/ioscompanion/runnerassets/assets
key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }}
- name: Build sanderling
run: make sanderling-ios
- name: Boot a simulator
run: |
xcrun simctl boot "$IOS_DEVICE" || true
xcrun simctl bootstatus "$IOS_DEVICE" -b
env:
IOS_DEVICE: iPhone 16 Pro
- name: Build and install folio
run: just ios
working-directory: examples/folio
env:
IOS_DEVICE: iPhone 16 Pro
# `just ios` leaves the app running, and the run's first act is to clear
# its state. Stopping it here means the run always opens the same way.
- name: Stop the app before the run
run: xcrun simctl terminate booted app.folio || true
- name: Fuzz folio on the simulator
run: .github/scripts/folio-run.sh ios
env:
SEED: ${{ inputs.seed != '0' && inputs.seed || '7' }}
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }}
DURATION: ${{ inputs.duration }}
IOS_DEVICE: iPhone 16 Pro
- name: Upload the run
if: always()
uses: actions/upload-artifact@v7
with:
name: folio-ios
path: runs/
retention-days: 14
web:
timeout-minutes: 60
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'web' }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: go.mod
cache: true
- name: Set up the JDKs
uses: actions/setup-java@v5
with:
distribution: temurin
# The metro gradle plugin folio builds with needs a 21 runtime; the
# sidecar toolchain pins 17. Both are installed so gradle can pick.
java-version: |
17
21
- name: Set up bun
uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3.13"
- name: Cache Gradle
uses: actions/cache@v6
with:
path: |
~/.gradle/caches
~/.gradle/wrapper
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
restore-keys: |
folio-gradle-${{ runner.os }}-
- name: Set up Chrome
uses: browser-actions/setup-chrome@v2
with:
chrome-version: stable
- name: Allow Chrome under unprivileged user namespaces
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
- name: Verify headless Chrome starts
run: |
chrome --version
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
--dump-dom 'data:text/html,<title>ok</title>'
- name: Build the folio wasmJs app
run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution
working-directory: examples/folio
- name: Build sanderling
run: make sanderling-web
- name: Fuzz folio in the browser
run: .github/scripts/folio-run.sh web
env:
SEED: ${{ inputs.seed != '0' && inputs.seed || '3' }}
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }}
DURATION: ${{ inputs.duration }}
- name: Upload the run
if: always()
uses: actions/upload-artifact@v7
with:
name: folio-web
path: runs/
retention-days: 14
+51 -14
View File
@@ -11,26 +11,55 @@ on:
required: true required: true
type: string type: string
permissions:
contents: read
concurrency: concurrency:
group: release-${{ github.ref }} group: release-${{ github.ref }}
cancel-in-progress: false cancel-in-progress: false
jobs: jobs:
resolve-tag:
name: Resolve and validate the tag
runs-on: ubuntu-latest
permissions: {}
outputs:
tag: ${{ steps.tag.outputs.tag }}
version: ${{ steps.tag.outputs.version }}
steps:
# A refname is attacker-controlled text and git permits backtick, `$`,
# `(`, `;`, `&` and `|` in it, so it goes through env: a `${{ }}` is
# substituted before bash ever sees the line. Every later job reads these
# outputs rather than the refname, and nothing reaches a shell before it
# has matched the pattern. The pattern is anchored and admits no newline,
# which is what stops the value below forging a second $GITHUB_OUTPUT key.
- name: Validate the tag
id: tag
run: |
pattern='^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z]+(\.[0-9A-Za-z]+)*)?$'
if [[ ! "$TAG" =~ $pattern ]]; then
echo "release: refusing to publish from '$TAG'" >&2
echo "release: a release tag is vMAJOR.MINOR.PATCH with an optional -prerelease, e.g. v0.1.0 or v0.0.1-rc1" >&2
exit 1
fi
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
echo "version=${TAG#v}" >> "$GITHUB_OUTPUT"
env:
TAG: ${{ inputs.tag || github.ref_name }}
release-npm: release-npm:
name: Publish @sanderling/spec to npm name: Publish @sanderling/spec to npm
needs: resolve-tag
runs-on: ubuntu-latest runs-on: ubuntu-latest
env: permissions:
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} contents: read
steps: steps:
- uses: actions/checkout@v7 - uses: actions/checkout@v7
with: with:
ref: ${{ inputs.tag || github.ref }} ref: ${{ needs.resolve-tag.outputs.tag }}
# `npm ci` below runs dependency lifecycle scripts, and no step in
- name: Resolve version # this job needs the git credential afterwards.
id: ver persist-credentials: false
run: |
raw="${{ inputs.tag || github.ref_name }}"
echo "version=${raw#v}" >> "$GITHUB_OUTPUT"
- name: Set up Node 22 - name: Set up Node 22
uses: actions/setup-node@v7 uses: actions/setup-node@v7
@@ -46,28 +75,36 @@ jobs:
- name: Stamp version - name: Stamp version
working-directory: pkg/spec working-directory: pkg/spec
run: npm version ${{ steps.ver.outputs.version }} --no-git-tag-version --allow-same-version run: npm version "$VERSION" --no-git-tag-version --allow-same-version
env:
VERSION: ${{ needs.resolve-tag.outputs.version }}
- name: Publish - name: Publish
working-directory: pkg/spec working-directory: pkg/spec
# npm tag pre-releases (e.g. 0.1.0-rc1) as "next" so npm install @sanderling/spec # npm tag pre-releases (e.g. 0.1.0-rc1) as "next" so npm install @sanderling/spec
# keeps resolving the latest stable. # keeps resolving the latest stable.
run: | run: |
if [[ "${{ steps.ver.outputs.version }}" == *-* ]]; then if [[ "$VERSION" == *-* ]]; then
npm publish --access public --tag next npm publish --access public --tag next
else else
npm publish --access public npm publish --access public
fi fi
# The publish credential is scoped to the one step that publishes rather
# than to the job, so no other step runs with it in reach.
env:
VERSION: ${{ needs.resolve-tag.outputs.version }}
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
release-cli: release-cli:
name: Publish sanderling CLI to GitHub Releases name: Publish sanderling CLI to GitHub Releases
needs: resolve-tag
runs-on: ubuntu-latest runs-on: ubuntu-latest
permissions: permissions:
contents: write contents: write
steps: steps:
- uses: actions/checkout@v7 - uses: actions/checkout@v7
with: with:
ref: ${{ inputs.tag || github.ref }} ref: ${{ needs.resolve-tag.outputs.tag }}
fetch-depth: 0 fetch-depth: 0
- name: Set up Go - name: Set up Go
@@ -83,7 +120,7 @@ jobs:
java-version: "17" java-version: "17"
- name: Set up Android SDK - name: Set up Android SDK
uses: android-actions/setup-android@v4 uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
- name: Cache Gradle - name: Cache Gradle
uses: actions/cache@v6 uses: actions/cache@v6
@@ -99,7 +136,7 @@ jobs:
run: make sidecar run: make sidecar
- name: Run GoReleaser - name: Run GoReleaser
uses: goreleaser/goreleaser-action@v7 uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3
with: with:
version: "~> v2" version: "~> v2"
args: release --clean args: release --clean
-142
View File
@@ -1,142 +0,0 @@
name: replay-ui
# Sanderling fuzzing sanderling's own replay UI. Dispatch-only: it takes minutes
# and it is a demo of the product loop, not a merge gate.
#
# The shape is: produce a real trace, serve it with `sanderling replay`, then run
# a spec against that UI. Any violation fails the job. Six of the seven
# properties in replay-ui/sanderling/spec.ts are cross-panel agreements that hold
# for any trace; the seventh is the stock noUncaughtExceptions. None of them
# needs recalibrating when the fixture changes.
on:
workflow_dispatch:
inputs:
seed:
description: seed for the dogfood run
default: "3"
max-steps:
description: step budget for the dogfood run
default: "80"
permissions:
contents: read
jobs:
dogfood:
timeout-minutes: 45
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: go.mod
cache: true
- name: Set up bun
uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3.13"
# Pinned stable plus the AppArmor sysctl: the same setup ci.yml's browser
# job needs to get headless Chrome up on ubuntu-latest.
- name: Set up Chrome
uses: browser-actions/setup-chrome@v2
with:
chrome-version: stable
- name: Allow Chrome under unprivileged user namespaces
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
- name: Verify headless Chrome starts
run: |
chrome --version
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
--dump-dom 'data:text/html,<title>ok</title>'
# The UI the spec drives is the one embedded in this binary, so the build
# has to come after any change to replay-ui/src.
- name: Build sanderling
run: make sanderling-web
# A trace with a violation and uncaught exceptions in it, so the UI has
# something to render in every panel the spec looks at. No
# --exit-on-violation here: the run is the fixture, and stopping it at the
# first violation would leave a four-step trace to fuzz.
- name: Record a fixture trace
run: |
python3 -m http.server 8792 --bind 127.0.0.1 \
--directory test/browser/testdata/throwing &
ready=""
for _ in $(seq 1 30); do
curl -sf http://127.0.0.1:8792/ >/dev/null && { ready=1; break; }
sleep 1
done
if [ -z "$ready" ]; then
echo "the fixture http server never answered on 127.0.0.1:8792" >&2
exit 1
fi
./bin/sanderling test \
--platform web \
--spec test/browser/testdata/throwing/spec.ts \
--bundle-id http://127.0.0.1:8792/ \
--duration 5m --max-steps 25 --seed 7 \
--output runs/fixture
- name: Serve the trace with sanderling replay
run: |
# Flags before the positional argument: Go's flag package stops
# parsing at the first non-flag word.
./bin/sanderling replay --port 8793 --no-open runs/fixture &
ready=""
for _ in $(seq 1 30); do
curl -sf http://127.0.0.1:8793/api/runs >/dev/null && { ready=1; break; }
sleep 1
done
if [ -z "$ready" ]; then
echo "sanderling replay never served /api/runs on 127.0.0.1:8793" >&2
exit 1
fi
run_id="$(ls runs/fixture | head -1)"
echo "RUN_URL=http://127.0.0.1:8793/runs/$run_id/steps/1" >> "$GITHUB_ENV"
curl -sf "http://127.0.0.1:8793/runs/$run_id/steps/1" >/dev/null
# Inputs go through env rather than into the script text: a `${{ }}` is
# substituted before bash ever sees the line, so a seed of `$(id)` would
# run as a command.
- name: Fuzz the replay UI
run: |
./bin/sanderling test \
--platform web \
--spec replay-ui/sanderling/spec.ts \
--bundle-id "$RUN_URL" \
--duration 10m \
--max-steps "$MAX_STEPS" \
--seed "$SEED" \
--exit-on-violation \
--output runs/dogfood
env:
SEED: ${{ inputs.seed }}
MAX_STEPS: ${{ inputs.max-steps }}
# Exit 0 above means no property returned false. It does not mean any
# property was ever evaluated against real content: they all decline to
# judge when the elements they read are absent, so a run that never
# rendered the step page is green and worthless. This step is what tells
# the two apart, and it fails the job when nothing was judged.
- name: Summarise
if: always()
run: .github/scripts/replay-ui-summary.sh runs/dogfood
env:
SEED: ${{ inputs.seed }}
MAX_STEPS: ${{ inputs.max-steps }}
- name: Upload runs
if: always()
uses: actions/upload-artifact@v7
with:
name: replay-ui-runs
path: runs/
retention-days: 14