mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
Merge branch 'one-examples-workflow' into correctness-and-spec-skills
This commit is contained in:
commit
a7c5193615
14 files changed
+786
-468
No files matched your search
@@ -0,0 +1,81 @@
|
||||
name: folio app
|
||||
description: Install the toolchain folio needs on one platform, and build the app there.
|
||||
|
||||
inputs:
|
||||
platform:
|
||||
description: android, ios or web
|
||||
required: true
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Set up the JDKs
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
# The metro gradle plugin folio builds with needs a 21 runtime; the
|
||||
# sidecar toolchain pins 17. Both are installed so gradle can pick.
|
||||
java-version: |
|
||||
17
|
||||
21
|
||||
|
||||
# The iOS app builds its Kotlin framework through the folio gradle project,
|
||||
# which configures :app:androidApp, so this is needed off Android too.
|
||||
# `make sanderling-android` wants it as well, for the sidecar JAR.
|
||||
- name: Set up Android SDK
|
||||
if: inputs.platform != 'web'
|
||||
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
|
||||
|
||||
- name: Cache Gradle
|
||||
if: inputs.platform != 'ios'
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
|
||||
restore-keys: |
|
||||
folio-gradle-${{ runner.os }}-
|
||||
|
||||
# idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so
|
||||
# it has to be named by its full tap path. xcodegen and just are core.
|
||||
- name: Install idb-companion, xcodegen and just
|
||||
if: inputs.platform == 'ios'
|
||||
shell: bash
|
||||
run: brew install facebook/fb/idb-companion xcodegen just
|
||||
|
||||
# Both asset tarballs are built by the prepare scripts, and the runner
|
||||
# bundle is an xcodebuild of companion/Sources. Keyed on the scripts and
|
||||
# the versions the Makefile embeds, so a later run reuses them. This has to
|
||||
# land before `make sanderling-ios`, which is what consumes them.
|
||||
- name: Cache the companion and runner bundles
|
||||
if: inputs.platform == 'ios'
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
internal/driver/ioscompanion/companionassets/assets
|
||||
internal/driver/ioscompanion/runnerassets/assets
|
||||
key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }}
|
||||
|
||||
# Without this the emulator falls back to software rendering and every
|
||||
# step costs several seconds.
|
||||
- name: Enable KVM
|
||||
if: inputs.platform == 'android'
|
||||
shell: bash
|
||||
run: |
|
||||
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \
|
||||
| sudo tee /etc/udev/rules.d/99-kvm4all.rules
|
||||
sudo udevadm control --reload-rules
|
||||
sudo udevadm trigger --name-match=kvm
|
||||
|
||||
- name: Build the folio APK
|
||||
if: inputs.platform == 'android'
|
||||
shell: bash
|
||||
working-directory: examples/folio
|
||||
run: ./gradlew :app:androidApp:assembleDebug
|
||||
|
||||
- name: Build the folio wasmJs app
|
||||
if: inputs.platform == 'web'
|
||||
shell: bash
|
||||
working-directory: examples/folio
|
||||
run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution
|
||||
@@ -0,0 +1,31 @@
|
||||
name: folio on a simulator
|
||||
description: Boot an iOS simulator, install folio on it, and leave the app stopped.
|
||||
|
||||
inputs:
|
||||
device:
|
||||
description: simulator device name
|
||||
default: iPhone 16 Pro
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Boot a simulator
|
||||
shell: bash
|
||||
run: |
|
||||
xcrun simctl boot "$IOS_DEVICE" || true
|
||||
xcrun simctl bootstatus "$IOS_DEVICE" -b
|
||||
env:
|
||||
IOS_DEVICE: ${{ inputs.device }}
|
||||
|
||||
- name: Build and install folio
|
||||
shell: bash
|
||||
working-directory: examples/folio
|
||||
run: just ios
|
||||
env:
|
||||
IOS_DEVICE: ${{ inputs.device }}
|
||||
|
||||
# `just ios` leaves the app running, and the run's first act is to clear
|
||||
# its state. Stopping it here means the run always opens the same way.
|
||||
- name: Stop the app before the run
|
||||
shell: bash
|
||||
run: xcrun simctl terminate booted app.folio || true
|
||||
@@ -0,0 +1,30 @@
|
||||
name: headless chrome
|
||||
description: Install Chrome and prove it starts headless before a driver depends on it.
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
# stable is setup-chrome v2's own default, spelled out so a new release of
|
||||
# the action cannot move the browser these jobs drive. The alternative it
|
||||
# offers is Chrome for Testing latest, which tracks ahead of the channel
|
||||
# users run.
|
||||
- uses: browser-actions/setup-chrome@2e1d749697dd1612b833dba4a722266286fbefcd # v2.1.2
|
||||
with:
|
||||
chrome-version: stable
|
||||
|
||||
# Ubuntu 24.04 (current ubuntu-latest) restricts unprivileged user
|
||||
# namespaces via AppArmor, which stops headless Chrome from starting even
|
||||
# with --no-sandbox: the process launches but never opens its DevTools
|
||||
# socket. Re-enable them so the driver's Chrome can come up.
|
||||
- name: Allow Chrome under unprivileged user namespaces
|
||||
shell: bash
|
||||
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
|
||||
# Fail here with Chrome's own stderr if the browser can't launch, instead
|
||||
# of letting the driver report an opaque DevTools timeout downstream.
|
||||
- name: Verify headless Chrome starts
|
||||
shell: bash
|
||||
run: |
|
||||
chrome --version
|
||||
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
|
||||
--dump-dom 'data:text/html,<title>ok</title>'
|
||||
@@ -0,0 +1,55 @@
|
||||
name: replay ui fixture
|
||||
description: Record a trace with sanderling, then serve it with sanderling replay.
|
||||
|
||||
outputs:
|
||||
url:
|
||||
description: the step page of the served run, for a spec to drive
|
||||
value: ${{ steps.serve.outputs.url }}
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
# A trace with a violation and uncaught exceptions in it, so the UI has
|
||||
# something to render in every panel the spec looks at. No
|
||||
# --exit-on-violation here: the run is the fixture, and stopping it at the
|
||||
# first violation would leave a four-step trace to fuzz.
|
||||
- name: Record a fixture trace
|
||||
shell: bash
|
||||
run: |
|
||||
python3 -m http.server 8792 --bind 127.0.0.1 \
|
||||
--directory test/browser/testdata/throwing &
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf http://127.0.0.1:8792/ >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "the fixture http server never answered on 127.0.0.1:8792" >&2
|
||||
exit 1
|
||||
fi
|
||||
./bin/sanderling test \
|
||||
--platform web \
|
||||
--spec test/browser/testdata/throwing/spec.ts \
|
||||
--bundle-id http://127.0.0.1:8792/ \
|
||||
--duration 5m --max-steps 25 --seed 7 \
|
||||
--output runs/fixture
|
||||
|
||||
- name: Serve the trace with sanderling replay
|
||||
id: serve
|
||||
shell: bash
|
||||
run: |
|
||||
# Flags before the positional argument: Go's flag package stops
|
||||
# parsing at the first non-flag word.
|
||||
./bin/sanderling replay --port 8793 --no-open runs/fixture &
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf http://127.0.0.1:8793/api/runs >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "sanderling replay never served /api/runs on 127.0.0.1:8793" >&2
|
||||
exit 1
|
||||
fi
|
||||
run_id="$(ls runs/fixture | head -1)"
|
||||
echo "url=http://127.0.0.1:8793/runs/$run_id/steps/1" >> "$GITHUB_OUTPUT"
|
||||
curl -sf "http://127.0.0.1:8793/runs/$run_id/steps/1" >/dev/null
|
||||
Executable
+284
@@ -0,0 +1,284 @@
|
||||
#!/usr/bin/env bash
|
||||
# Drives folio-run.sh through a stubbed `sanderling` binary and checks the
|
||||
# verdict it reaches from each shape of trace. What is under test is the
|
||||
# classification, not the fuzzer: the stub writes the trace the run would have
|
||||
# written and exits the code the run would have exited.
|
||||
#
|
||||
# folio-run.sh is invoked as `bash -eo pipefail -c <script>`, which is what a
|
||||
# `run:` block does: -e is set on the shell that calls the script, and does not
|
||||
# cross the shebang into it. Running the script itself under -e would kill it at
|
||||
# the first non-zero `sanderling test`, which is the exit code it exists to read.
|
||||
set -euo pipefail
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
script="$here/folio-run.sh"
|
||||
spec="$here/../../examples/folio/sanderling/spec.ts"
|
||||
work="$(mktemp -d)"
|
||||
trap 'rm -rf "$work"' EXIT
|
||||
|
||||
failed=0
|
||||
case_dir=""
|
||||
status=0
|
||||
spec_override=""
|
||||
|
||||
run() { # <case> <platform> <stub exit code> [none] ; trace lines on stdin
|
||||
local name="$1" platform="$2" code="$3" trace_mode="${4:-file}"
|
||||
case_dir="$work/$name"
|
||||
mkdir -p "$case_dir"
|
||||
cat > "$case_dir/trace.jsonl"
|
||||
printf '#!/usr/bin/env bash\n' > "$case_dir/sanderling"
|
||||
cat >> "$case_dir/sanderling" <<'STUB'
|
||||
printf '%s\n' "$@" > "$STUB_ARGV"
|
||||
out="" ; prev=""
|
||||
for arg in "$@"; do
|
||||
[ "$prev" = "--output" ] && out="$arg"
|
||||
prev="$arg"
|
||||
done
|
||||
if [ "$STUB_TRACE_MODE" = file ]; then
|
||||
mkdir -p "$out/20260815-101500"
|
||||
cp "$STUB_TRACE" "$out/20260815-101500/trace.jsonl"
|
||||
fi
|
||||
exit "$STUB_CODE"
|
||||
STUB
|
||||
chmod +x "$case_dir/sanderling"
|
||||
status=0
|
||||
cd "$case_dir"
|
||||
STUB_ARGV="$case_dir/argv" STUB_TRACE="$case_dir/trace.jsonl" \
|
||||
STUB_TRACE_MODE="$trace_mode" STUB_CODE="$code" \
|
||||
SANDERLING="$case_dir/sanderling" SPEC="${spec_override:-$spec}" \
|
||||
GITHUB_STEP_SUMMARY="$case_dir/summary.md" \
|
||||
SEED=7 MAX_STEPS=240 DURATION=20m \
|
||||
bash -eo pipefail -c "'$script' '$platform'" \
|
||||
> "$case_dir/out" 2> "$case_dir/err" || status=$?
|
||||
cd "$here"
|
||||
spec_override=""
|
||||
}
|
||||
|
||||
fail() { echo "FAIL: $*" >&2; failed=1; }
|
||||
|
||||
expect_status() { # <want> <case>
|
||||
[ "$status" = "$1" ] || fail "$2: exit $status, want $1"
|
||||
}
|
||||
|
||||
# grep reads both files directly: piping `cat` into `grep -q` returns 141 under
|
||||
# pipefail, because grep leaves on the first match and cat takes the SIGPIPE.
|
||||
expect_says() { # <text> <case>
|
||||
grep -qF -- "$1" "$case_dir/out" "$case_dir/err" || fail "$2: nothing said '$1'"
|
||||
}
|
||||
|
||||
expect_silent() { # <text> <case>
|
||||
if grep -qF -- "$1" "$case_dir/out" "$case_dir/err"; then
|
||||
fail "$2: should not have said '$1'"
|
||||
fi
|
||||
}
|
||||
|
||||
expect_summary() { # <text> <case>
|
||||
grep -qF -- "$1" "$case_dir/summary.md" || fail "$2: summary has no '$1'"
|
||||
}
|
||||
|
||||
expect_argv() { # <text> <case>
|
||||
grep -qxF -- "$1" "$case_dir/argv" || fail "$2: '$1' never reached the binary"
|
||||
}
|
||||
|
||||
convicting='{"violations":["submitCommitsOneTransactionPerAction"],"witnesses":{"submitCommitsOneTransactionPerAction":{"is_error":false,"reason":"one tap, two rows"}}}'
|
||||
unrelated='{"violations":["newAccountBalanceIsZero"],"witnesses":{"newAccountBalanceIsZero":{"is_error":false,"reason":"opened at 4.00"}}}'
|
||||
threw='{"violations":["submitMovesBalanceByAtMostTypedAmount"],"witnesses":{"submitMovesBalanceByAtMostTypedAmount":{"is_error":true,"reason":"TypeError: cannot read text of undefined"}}}'
|
||||
on_txn='{"route":"AddTransactionScreen","violations":[]}'
|
||||
off_txn='{"route":"HomeScreen","violations":[]}'
|
||||
|
||||
# --- the flags the calibrated runs were measured with reach the binary --------
|
||||
|
||||
run argv-ios ios 2 <<TRACE
|
||||
$on_txn
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 0 argv-ios
|
||||
expect_argv "--exit-on-violation" argv-ios
|
||||
expect_argv "--platform" argv-ios
|
||||
expect_argv "ios" argv-ios
|
||||
expect_argv "--seed" argv-ios
|
||||
expect_argv "7" argv-ios
|
||||
expect_argv "240" argv-ios
|
||||
expect_argv "20m" argv-ios
|
||||
expect_argv "iPhone 16 Pro" argv-ios
|
||||
|
||||
run argv-android android 2 <<TRACE
|
||||
$on_txn
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 0 argv-android
|
||||
expect_argv "--exit-on-violation" argv-android
|
||||
expect_argv "app.folio" argv-android
|
||||
expect_argv "examples/folio/app/androidApp/build/outputs/apk/debug/androidApp-debug.apk" argv-android
|
||||
|
||||
# --- ios and web: a conviction is required -----------------------------------
|
||||
|
||||
run convicted ios 2 <<TRACE
|
||||
$on_txn
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 0 convicted
|
||||
expect_says "found the submit bug in 2 steps (submitCommitsOneTransactionPerAction)" convicted
|
||||
expect_summary "- convicted on: submitCommitsOneTransactionPerAction" convicted
|
||||
|
||||
# Exit 2 with a violation of a real but ungated property is not this leg's bug.
|
||||
run other-violation ios 2 <<TRACE
|
||||
$on_txn
|
||||
$unrelated
|
||||
TRACE
|
||||
expect_status 1 other-violation
|
||||
expect_says "not the double-submit this leg gates on" other-violation
|
||||
expect_summary "- also violated: newAccountBalanceIsZero" other-violation
|
||||
|
||||
# A predicate that threw reaches exit 2 by the identical path, and is not a
|
||||
# verdict about folio at all.
|
||||
run threw-ios ios 2 <<TRACE
|
||||
$on_txn
|
||||
$threw
|
||||
TRACE
|
||||
expect_status 1 threw-ios
|
||||
expect_says "a predicate threw, so exit 2 is not a verdict about folio" threw-ios
|
||||
expect_says "TypeError: cannot read text of undefined" threw-ios
|
||||
expect_silent "found the submit bug" threw-ios
|
||||
expect_summary "**a predicate threw**" threw-ios
|
||||
|
||||
# The bug is still in folio, so a clean run is the fuzzer no longer reaching it.
|
||||
run clean-ios ios 0 <<TRACE
|
||||
$on_txn
|
||||
$on_txn
|
||||
TRACE
|
||||
expect_status 1 clean-ios
|
||||
expect_says "the double-submit bug was NOT found in 2 steps" clean-ios
|
||||
|
||||
run harness-ios ios 3 <<TRACE
|
||||
$on_txn
|
||||
TRACE
|
||||
expect_status 3 harness-ios
|
||||
expect_says "the harness failed with exit 3" harness-ios
|
||||
|
||||
# --- android: a health gate, with a conviction as a bonus --------------------
|
||||
|
||||
run android-convicted android 2 <<TRACE
|
||||
$on_txn
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 0 android-convicted
|
||||
expect_says "found the submit bug in 2 steps (a bonus, not required)" android-convicted
|
||||
|
||||
run android-other android 2 <<TRACE
|
||||
$on_txn
|
||||
$unrelated
|
||||
TRACE
|
||||
expect_status 0 android-other
|
||||
expect_says "judging health only" android-other
|
||||
|
||||
run android-healthy android 0 <<TRACE
|
||||
$on_txn
|
||||
$on_txn
|
||||
TRACE
|
||||
expect_status 0 android-healthy
|
||||
expect_says "healthy run over 2 steps, reached the transaction screen" android-healthy
|
||||
|
||||
# Health means it got to the screen the double-submit lives on. Anything less is
|
||||
# a leg that proved nothing, however green the exit code.
|
||||
run android-stalled android 0 <<TRACE
|
||||
$off_txn
|
||||
{"route":"LedgerScreen","violations":[]}
|
||||
TRACE
|
||||
expect_status 1 android-stalled
|
||||
expect_says "never reached AddTransactionScreen over 2 steps" android-stalled
|
||||
expect_says "routes the trace does record: HomeScreen,LedgerScreen" android-stalled
|
||||
|
||||
# The thrown-predicate check has to bite on android too: this is the leg whose
|
||||
# gate is loose enough to swallow it.
|
||||
run threw-android android 2 <<TRACE
|
||||
$on_txn
|
||||
$threw
|
||||
TRACE
|
||||
expect_status 1 threw-android
|
||||
expect_says "a predicate threw" threw-android
|
||||
expect_silent "judging health only" threw-android
|
||||
|
||||
run harness-android android 4 <<TRACE
|
||||
$on_txn
|
||||
TRACE
|
||||
expect_status 4 harness-android
|
||||
expect_says "the harness failed with exit 4" harness-android
|
||||
|
||||
# --- traces that are not there, or are there and say nothing -----------------
|
||||
|
||||
# Exit 2 and no run directory at all: the glob matches nothing, and an empty
|
||||
# trace must not be judged as a folio that behaved.
|
||||
run no-trace ios 2 none </dev/null
|
||||
expect_status 1 no-trace
|
||||
expect_says "wrote no trace" no-trace
|
||||
expect_silent "Traceback" no-trace
|
||||
|
||||
run no-trace-android android 0 none </dev/null
|
||||
expect_status 1 no-trace-android
|
||||
expect_says "wrote no trace" no-trace-android
|
||||
|
||||
# A zero-byte trace: the file is there, so the missing-trace check passes and
|
||||
# the classification has to survive reading nothing out of it.
|
||||
run zero-byte ios 0 file </dev/null
|
||||
expect_status 1 zero-byte
|
||||
expect_says "NOT found in 0 steps" zero-byte
|
||||
expect_silent "Traceback" zero-byte
|
||||
|
||||
# A line the recorder truncated is skipped, not fatal, and the violation on the
|
||||
# readable line is still found.
|
||||
run malformed ios 2 <<TRACE
|
||||
$on_txn
|
||||
{"violations":["submitCommitsOne
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 0 malformed
|
||||
expect_says "found the submit bug" malformed
|
||||
expect_silent "Traceback" malformed
|
||||
|
||||
run bad-platform windows 0 none </dev/null
|
||||
expect_status 64 bad-platform
|
||||
expect_says "unknown platform: windows" bad-platform
|
||||
|
||||
# --- the gate names still exist in the spec they gate ------------------------
|
||||
|
||||
# Nothing but this check ties GATED_PROPERTIES to the spec. Rename a gated
|
||||
# property and the classification matches nothing: ios and web report a
|
||||
# different bug, android reclassifies a conviction as health and stays green.
|
||||
sed 's/submitCommitsOneTransactionPerAction/submitCommitsOneTxnPerAction/g' \
|
||||
"$spec" > "$work/renamed-spec.ts"
|
||||
spec_override="$work/renamed-spec.ts"
|
||||
run drift-renamed ios 2 <<TRACE
|
||||
$on_txn
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 1 drift-renamed
|
||||
expect_says "no longer declares submitCommitsOneTransactionPerAction" drift-renamed
|
||||
if [ -e "$case_dir/argv" ]; then
|
||||
fail "drift-renamed: the run started before the gate was checked"
|
||||
fi
|
||||
|
||||
spec_override="$work/absent-spec.ts"
|
||||
run drift-missing ios 2 none </dev/null
|
||||
expect_status 1 drift-missing
|
||||
expect_says "cannot read" drift-missing
|
||||
|
||||
echo "export const notProperties = { a };" > "$work/shapeless-spec.ts"
|
||||
spec_override="$work/shapeless-spec.ts"
|
||||
run drift-shapeless ios 2 none </dev/null
|
||||
expect_status 1 drift-shapeless
|
||||
expect_says "declares no \`export const properties" drift-shapeless
|
||||
|
||||
# And the same check against the spec as it stands: this is the assertion that
|
||||
# fails at `make test` when someone renames a property without moving the gate.
|
||||
run gate-matches-spec ios 2 <<TRACE
|
||||
$on_txn
|
||||
$convicting
|
||||
TRACE
|
||||
expect_status 0 gate-matches-spec
|
||||
expect_silent "no longer declares" gate-matches-spec
|
||||
|
||||
if [ "$failed" = 0 ]; then
|
||||
echo "folio-run.sh: ok"
|
||||
fi
|
||||
exit "$failed"
|
||||
@@ -18,9 +18,49 @@ max_steps="${MAX_STEPS:-240}"
|
||||
duration="${DURATION:-20m}"
|
||||
sanderling="${SANDERLING:-./bin/sanderling}"
|
||||
output="runs/folio-$platform"
|
||||
spec="examples/folio/sanderling/spec.ts"
|
||||
spec="${SPEC:-examples/folio/sanderling/spec.ts}"
|
||||
summary="${GITHUB_STEP_SUMMARY:-/dev/null}"
|
||||
|
||||
# The two properties that state folio's double-submit. Anything else the spec
|
||||
# proves false is a different finding, and this leg has nothing to say about it.
|
||||
GATED_PROPERTIES="submitMovesBalanceByAtMostTypedAmount,submitCommitsOneTransactionPerAction"
|
||||
|
||||
# A gate is only as good as these names, and nothing else ties them to the spec.
|
||||
# Rename a property there and the classification below matches nothing: ios and
|
||||
# web blame the spec for finding a different bug, and android reclassifies a
|
||||
# real conviction as "judging health only" and stays green. Checked before the
|
||||
# run so a rename costs seconds rather than the whole budget.
|
||||
SPEC="$spec" GATED="$GATED_PROPERTIES" SELF="$0" python3 - <<'PY' || exit 1
|
||||
import os, re, sys
|
||||
|
||||
spec_path = os.environ["SPEC"]
|
||||
gated = [name for name in os.environ["GATED"].split(",") if name]
|
||||
try:
|
||||
with open(spec_path, encoding="utf-8") as handle:
|
||||
source = handle.read()
|
||||
except OSError as error:
|
||||
sys.exit("folio: cannot read %s to check the gated properties still exist: %s"
|
||||
% (spec_path, error))
|
||||
|
||||
block = re.search(r"export\s+const\s+properties\s*=\s*\{(.*?)\}", source, re.S)
|
||||
if block is None:
|
||||
sys.exit("folio: %s declares no `export const properties = {...}`, so the gated "
|
||||
"properties cannot be checked against it" % spec_path)
|
||||
|
||||
declared = set()
|
||||
for entry in re.sub(r"//[^\n]*", "", block.group(1)).split(","):
|
||||
name = entry.split(":")[0].strip()
|
||||
if re.fullmatch(r"[A-Za-z_$][A-Za-z0-9_$]*", name):
|
||||
declared.add(name)
|
||||
|
||||
missing = [name for name in gated if name not in declared]
|
||||
if missing:
|
||||
sys.exit("folio: %s no longer declares %s, so this leg gates on a property that "
|
||||
"cannot be violated and every real conviction would read as a different "
|
||||
"finding. Update GATED_PROPERTIES in %s."
|
||||
% (spec_path, ", ".join(missing), os.environ["SELF"]))
|
||||
PY
|
||||
|
||||
folio_args=(--bundle-id app.folio)
|
||||
case "$platform" in
|
||||
android)
|
||||
@@ -89,14 +129,14 @@ esac
|
||||
code=$?
|
||||
|
||||
run_dir="$(ls -d "$output"/*/ 2>/dev/null | tail -1)"
|
||||
trace="${run_dir:-.}/trace.jsonl"
|
||||
# No run directory means no trace. Defaulting the directory to `.` here reads a
|
||||
# stray ./trace.jsonl and reports it as this run's evidence, which is how a run
|
||||
# that wrote nothing at all reached "found the submit bug" and exit 0.
|
||||
trace=""
|
||||
[ -n "$run_dir" ] && trace="${run_dir}trace.jsonl"
|
||||
steps=0
|
||||
[ -f "$trace" ] && steps=$(wc -l < "$trace" | tr -d ' ')
|
||||
|
||||
# The two properties that state folio's double-submit. Anything else the spec
|
||||
# proves false is a different finding, and this leg has nothing to say about it.
|
||||
GATED_PROPERTIES="submitMovesBalanceByAtMostTypedAmount,submitCommitsOneTransactionPerAction"
|
||||
|
||||
# Exit 2 means "the run recorded a violation", and that is NOT the same as "the
|
||||
# run convicted folio". A predicate that THROWS is recorded as a violation too,
|
||||
# with is_error set and the thrown text as its reason, and it reaches exit 2 by
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
# summary it renders. Run under the flags GitHub Actions uses for a `run:`
|
||||
# block, because that is where a swallowed failure hides.
|
||||
#
|
||||
# testdata/replay-ui-real-run.jsonl is the first 10 steps of the dogfood run in
|
||||
# testdata/replay-ui-real-run.jsonl is the first 10 steps of the replay-ui run in
|
||||
# actions run 31873049857 on master, with the per-step `hierarchy` dumps and the
|
||||
# rowElements/tabElements extractors removed so the file stays readable. Nothing
|
||||
# else was touched. That run was green, and badgeCountMatchesThePanel judged
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
#!/usr/bin/env bash
|
||||
# Reads the replay-ui dogfood trace and reports, per property, how many steps
|
||||
# Reads the replay-ui fuzzing trace and reports, per property, how many steps
|
||||
# that property actually judged. Kept out of the workflow YAML so it can be run
|
||||
# by hand against a local run:
|
||||
#
|
||||
# GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/dogfood
|
||||
# GITHUB_STEP_SUMMARY=/dev/stdout .github/scripts/replay-ui-summary.sh runs/replay-ui
|
||||
#
|
||||
# `sanderling test` exiting 0 says only that no property returned false. Every
|
||||
# property in replay-ui/sanderling/spec.ts declines to judge when a reading it
|
||||
@@ -14,7 +14,7 @@
|
||||
set -euo pipefail
|
||||
|
||||
root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
output="${1:-runs/dogfood}"
|
||||
output="${1:-runs/replay-ui}"
|
||||
spec="${SPEC:-$root/replay-ui/sanderling/spec.ts}"
|
||||
summary="${GITHUB_STEP_SUMMARY:-/dev/null}"
|
||||
|
||||
@@ -23,7 +23,7 @@ run_dirs=("$output"/*/)
|
||||
shopt -u nullglob
|
||||
|
||||
{
|
||||
echo "### replay-ui dogfood"
|
||||
echo "### the replay ui"
|
||||
echo
|
||||
echo "- seed \`${SEED:-unset}\`, budget ${MAX_STEPS:-unset} steps"
|
||||
} >> "$summary"
|
||||
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
java-version: "17"
|
||||
|
||||
- name: Set up Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
|
||||
|
||||
- name: Set up Node 22
|
||||
uses: actions/setup-node@v7
|
||||
@@ -44,7 +44,7 @@ jobs:
|
||||
cache-dependency-path: pkg/spec/package-lock.json
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
@@ -100,7 +100,7 @@ jobs:
|
||||
# JAVA_HOME after `make test` rather than installing both up front
|
||||
# leaves every step above this one on exactly the JDK it ran on before.
|
||||
- name: Set up JDK 21 for folio
|
||||
uses: actions/setup-java@v4
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "21"
|
||||
@@ -119,27 +119,8 @@ jobs:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
# Pin stable: a dev Chromium's remote-debugging socket is flaky under
|
||||
# the driver, even though the browser otherwise launches headless.
|
||||
- name: Set up Chrome
|
||||
uses: browser-actions/setup-chrome@v2
|
||||
with:
|
||||
chrome-version: stable
|
||||
|
||||
# Ubuntu 24.04 (current ubuntu-latest) restricts unprivileged user
|
||||
# namespaces via AppArmor, which stops headless Chrome from starting even
|
||||
# with --no-sandbox: the process launches but never opens its DevTools
|
||||
# socket. Re-enable them so the driver's Chrome can come up.
|
||||
- name: Allow Chrome under unprivileged user namespaces
|
||||
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
|
||||
# Fail here with Chrome's own stderr if the browser can't launch, instead
|
||||
# of letting the driver report an opaque DevTools timeout downstream.
|
||||
- name: Verify headless Chrome starts
|
||||
run: |
|
||||
chrome --version
|
||||
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
|
||||
--dump-dom 'data:text/html,<title>ok</title>'
|
||||
- name: Set up headless Chrome
|
||||
uses: ./.github/actions/headless-chrome
|
||||
|
||||
- name: Drive web fixtures through headless Chrome
|
||||
run: make test-browser
|
||||
@@ -30,6 +30,11 @@ jobs:
|
||||
- name: Build site
|
||||
run: make docs
|
||||
|
||||
# No include-hidden-files: v4 stopped uploading dot-files by default, and
|
||||
# build/site has none. It is pandoc output plus a copy of docs/_assets,
|
||||
# which holds three ordinary files. _assets is underscore-prefixed, not
|
||||
# hidden, and deploy-pages serves the artifact without running Jekyll, so
|
||||
# it needs no .nojekyll either.
|
||||
- uses: actions/upload-pages-artifact@v5
|
||||
with:
|
||||
path: build/site
|
||||
|
||||
@@ -0,0 +1,193 @@
|
||||
name: examples
|
||||
|
||||
# Every example sanderling ships, fuzzed the same way: build sanderling for a
|
||||
# platform, bring the target up, run a spec against it, classify the trace it
|
||||
# wrote, upload the run. Only the bring-up differs, and that lives in the
|
||||
# per-target actions under .github/actions/.
|
||||
#
|
||||
# Dispatch-only: these take tens of minutes and they demonstrate the product
|
||||
# loop, they do not gate a merge.
|
||||
#
|
||||
# folio on ios and in the browser expect the bug: folio double-submits a
|
||||
# transaction on a double tap, so the run is supposed to end with exit 2. Exit 0
|
||||
# means the fuzzer stopped finding a bug that is still there; exit 1 means the
|
||||
# harness broke. The two are worth telling apart, which is why
|
||||
# --exit-on-violation exits 2 and not 1.
|
||||
#
|
||||
# folio on android is a health gate. It convicts in four runs out of five, which
|
||||
# is real evidence but not a gate: the fifth would report a regression it had
|
||||
# not found. Its budget is set so the conviction it usually gets is a bonus.
|
||||
#
|
||||
# the replay ui leg fuzzes sanderling's own replay UI, and any violation fails
|
||||
# it. Its properties are cross-panel agreements that hold for any trace, so none
|
||||
# of them needs recalibrating when the fixture changes.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
targets:
|
||||
description: which examples to fuzz
|
||||
type: choice
|
||||
options: [all, folio, android, ios, web, replay-ui]
|
||||
default: all
|
||||
seed:
|
||||
description: seed override (0 = each target's calibrated seed)
|
||||
default: "0"
|
||||
max-steps:
|
||||
description: step budget override (0 = each target's calibrated budget)
|
||||
default: "0"
|
||||
duration:
|
||||
description: wall-clock budget override (empty = each target's calibrated budget)
|
||||
default: ""
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
plan:
|
||||
runs-on: ubuntu-latest
|
||||
permissions: {}
|
||||
outputs:
|
||||
examples: ${{ steps.pick.outputs.examples }}
|
||||
steps:
|
||||
# The matrix is built here rather than written out under strategy.matrix
|
||||
# because a job-level `if:` cannot read the matrix context, so a static
|
||||
# matrix has no way to leave a leg out. jq -c keeps the value on one line,
|
||||
# which is what makes the $GITHUB_OUTPUT write below safe.
|
||||
- name: Pick the examples to fuzz
|
||||
id: pick
|
||||
run: |
|
||||
examples='[
|
||||
{"target":"android","name":"folio on android","app":"folio",
|
||||
"runs-on":"ubuntu-latest","timeout":90,"sanderling":"android",
|
||||
"seed":"9","max-steps":"200","duration":"20m","artifact":"folio-android"},
|
||||
{"target":"ios","name":"folio on ios","app":"folio",
|
||||
"runs-on":"macos-15","timeout":90,"sanderling":"ios",
|
||||
"seed":"7","max-steps":"240","duration":"20m","artifact":"folio-ios"},
|
||||
{"target":"web","name":"folio in the browser","app":"folio",
|
||||
"runs-on":"ubuntu-latest","timeout":60,"sanderling":"web","chrome":true,
|
||||
"seed":"3","max-steps":"240","duration":"20m","artifact":"folio-web"},
|
||||
{"target":"replay-ui","name":"the replay ui","app":"replay-ui",
|
||||
"runs-on":"ubuntu-latest","timeout":45,"sanderling":"web","chrome":true,
|
||||
"seed":"3","max-steps":"80","duration":"10m","artifact":"replay-ui-runs"}
|
||||
]'
|
||||
picked="$(jq -c --arg want "$TARGETS" \
|
||||
'map(select($want == "all" or .target == $want or .app == $want))' \
|
||||
<<<"$examples")"
|
||||
if [ "$picked" = "[]" ]; then
|
||||
echo "examples: '$TARGETS' selects no example, so this dispatch would run nothing" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "examples=$picked" >> "$GITHUB_OUTPUT"
|
||||
env:
|
||||
TARGETS: ${{ inputs.targets }}
|
||||
|
||||
fuzz:
|
||||
needs: plan
|
||||
name: fuzz ${{ matrix.name }}
|
||||
runs-on: ${{ matrix.runs-on }}
|
||||
timeout-minutes: ${{ matrix.timeout }}
|
||||
strategy:
|
||||
# Each leg is its own evidence. One target failing must not cancel the
|
||||
# others, which is how these ran as separate jobs.
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include: ${{ fromJSON(needs.plan.outputs.examples) }}
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || matrix.seed }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || matrix.max-steps }}
|
||||
DURATION: ${{ inputs.duration != '' && inputs.duration || matrix.duration }}
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
- name: Set up headless Chrome
|
||||
if: matrix.chrome
|
||||
uses: ./.github/actions/headless-chrome
|
||||
|
||||
- name: Build the folio app
|
||||
if: matrix.app == 'folio'
|
||||
uses: ./.github/actions/folio-app
|
||||
with:
|
||||
platform: ${{ matrix.target }}
|
||||
|
||||
# The UI the replay-ui spec drives is the one embedded in this binary, so
|
||||
# the build has to come after any change to replay-ui/src.
|
||||
- name: Build sanderling
|
||||
run: make "sanderling-$SANDERLING"
|
||||
env:
|
||||
SANDERLING: ${{ matrix.sanderling }}
|
||||
|
||||
- name: Put folio on the simulator
|
||||
if: matrix.target == 'ios'
|
||||
uses: ./.github/actions/folio-simulator
|
||||
|
||||
- name: Serve a trace to fuzz
|
||||
id: fixture
|
||||
if: matrix.target == 'replay-ui'
|
||||
uses: ./.github/actions/replay-ui-fixture
|
||||
|
||||
- name: Fuzz folio on an emulator
|
||||
if: matrix.target == 'android'
|
||||
uses: reactivecircus/android-emulator-runner@a421e43855164a8197daf9d8d40fe71c6996bb0d # v2.38.0
|
||||
with:
|
||||
api-level: 34
|
||||
target: google_apis
|
||||
arch: x86_64
|
||||
emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim
|
||||
disable-animations: true
|
||||
script: .github/scripts/folio-run.sh android
|
||||
|
||||
- name: Fuzz folio
|
||||
if: matrix.app == 'folio' && matrix.target != 'android'
|
||||
run: .github/scripts/folio-run.sh "$TARGET"
|
||||
env:
|
||||
TARGET: ${{ matrix.target }}
|
||||
|
||||
# Inputs go through env rather than into the script text: a `${{ }}` is
|
||||
# substituted before bash ever sees the line, so a seed of `$(id)` would
|
||||
# run as a command.
|
||||
- name: Fuzz the replay UI
|
||||
if: matrix.target == 'replay-ui'
|
||||
run: |
|
||||
./bin/sanderling test \
|
||||
--platform web \
|
||||
--spec replay-ui/sanderling/spec.ts \
|
||||
--bundle-id "$RUN_URL" \
|
||||
--duration "$DURATION" \
|
||||
--max-steps "$MAX_STEPS" \
|
||||
--seed "$SEED" \
|
||||
--exit-on-violation \
|
||||
--output runs/replay-ui
|
||||
env:
|
||||
RUN_URL: ${{ steps.fixture.outputs.url }}
|
||||
|
||||
# Exit 0 above means no property returned false. It does not mean any
|
||||
# property was ever evaluated against real content: they all decline to
|
||||
# judge when the elements they read are absent, so a run that never
|
||||
# rendered the step page is green and worthless. This step is what tells
|
||||
# the two apart, and it fails the job when nothing was judged. folio's
|
||||
# legs make the same call inside folio-run.sh, where the exit code it is
|
||||
# judging is in scope.
|
||||
- name: Classify the replay UI run
|
||||
if: ${{ always() && matrix.target == 'replay-ui' }}
|
||||
run: .github/scripts/replay-ui-summary.sh runs/replay-ui
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: ${{ matrix.artifact }}
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
@@ -1,277 +0,0 @@
|
||||
name: folio
|
||||
|
||||
# One spec, three platforms. Dispatch-only: each job boots a device or a
|
||||
# browser, builds the folio app for that platform, and runs
|
||||
# examples/folio/sanderling/spec.ts against it.
|
||||
#
|
||||
# web and ios are expect-the-bug jobs: folio double-submits a transaction on a
|
||||
# double tap, so the run is supposed to end with exit 2. Exit 0 means the fuzzer
|
||||
# stopped finding a bug that is still there; exit 1 means the harness broke. The
|
||||
# two are worth telling apart, which is why --exit-on-violation exits 2 and not
|
||||
# 1.
|
||||
#
|
||||
# android is a health gate. It convicts in four runs out of five, which is real
|
||||
# evidence but not a gate: the fifth would report a regression it had not found.
|
||||
# The budget is set so the conviction it usually gets is reported as a bonus.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
platforms:
|
||||
description: which legs to run
|
||||
type: choice
|
||||
options: [all, android, ios, web]
|
||||
default: all
|
||||
seed:
|
||||
description: seed override (0 = each job's calibrated seed)
|
||||
default: "0"
|
||||
duration:
|
||||
description: wall-clock budget per run
|
||||
default: 20m
|
||||
max-steps:
|
||||
description: step budget override (0 = each job's calibrated budget)
|
||||
default: "0"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
android:
|
||||
timeout-minutes: 90
|
||||
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'android' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up the JDKs
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
# The metro gradle plugin folio builds with needs a 21 runtime; the
|
||||
# sidecar toolchain pins 17. Both are installed so gradle can pick.
|
||||
java-version: |
|
||||
17
|
||||
21
|
||||
|
||||
- name: Set up Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
- name: Cache Gradle
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
|
||||
restore-keys: |
|
||||
folio-gradle-${{ runner.os }}-
|
||||
|
||||
# Without this the emulator falls back to software rendering and every
|
||||
# step costs several seconds.
|
||||
- name: Enable KVM
|
||||
run: |
|
||||
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \
|
||||
| sudo tee /etc/udev/rules.d/99-kvm4all.rules
|
||||
sudo udevadm control --reload-rules
|
||||
sudo udevadm trigger --name-match=kvm
|
||||
|
||||
- name: Build the folio APK
|
||||
run: ./gradlew :app:androidApp:assembleDebug
|
||||
working-directory: examples/folio
|
||||
|
||||
- name: Build sanderling
|
||||
run: make sanderling-android
|
||||
|
||||
- name: Fuzz folio on an emulator
|
||||
uses: reactivecircus/android-emulator-runner@v2
|
||||
with:
|
||||
api-level: 34
|
||||
target: google_apis
|
||||
arch: x86_64
|
||||
emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim
|
||||
disable-animations: true
|
||||
script: .github/scripts/folio-run.sh android
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || '9' }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '200' }}
|
||||
DURATION: ${{ inputs.duration }}
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: folio-android
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
|
||||
ios:
|
||||
timeout-minutes: 90
|
||||
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'ios' }}
|
||||
runs-on: macos-15
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
# idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so
|
||||
# it has to be named by its full tap path. xcodegen and just are core.
|
||||
- name: Install idb-companion, xcodegen and just
|
||||
run: brew install facebook/fb/idb-companion xcodegen just
|
||||
|
||||
- name: Set up the JDKs
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
# The metro gradle plugin folio builds with needs a 21 runtime; the
|
||||
# sidecar toolchain pins 17. Both are installed so gradle can pick.
|
||||
java-version: |
|
||||
17
|
||||
21
|
||||
|
||||
# The iOS app builds its Kotlin framework through the folio gradle
|
||||
# project, which configures :app:androidApp and so needs an Android SDK
|
||||
# even on this leg.
|
||||
- name: Set up Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
|
||||
# Both asset tarballs are built by the prepare scripts, and the runner
|
||||
# bundle is an xcodebuild of companion/Sources. Keyed on the scripts and
|
||||
# the versions the Makefile embeds, so a later run reuses them.
|
||||
- name: Cache the companion and runner bundles
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
internal/driver/ioscompanion/companionassets/assets
|
||||
internal/driver/ioscompanion/runnerassets/assets
|
||||
key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }}
|
||||
|
||||
- name: Build sanderling
|
||||
run: make sanderling-ios
|
||||
|
||||
- name: Boot a simulator
|
||||
run: |
|
||||
xcrun simctl boot "$IOS_DEVICE" || true
|
||||
xcrun simctl bootstatus "$IOS_DEVICE" -b
|
||||
env:
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
|
||||
- name: Build and install folio
|
||||
run: just ios
|
||||
working-directory: examples/folio
|
||||
env:
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
|
||||
# `just ios` leaves the app running, and the run's first act is to clear
|
||||
# its state. Stopping it here means the run always opens the same way.
|
||||
- name: Stop the app before the run
|
||||
run: xcrun simctl terminate booted app.folio || true
|
||||
|
||||
- name: Fuzz folio on the simulator
|
||||
run: .github/scripts/folio-run.sh ios
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || '7' }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }}
|
||||
DURATION: ${{ inputs.duration }}
|
||||
IOS_DEVICE: iPhone 16 Pro
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: folio-ios
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
|
||||
web:
|
||||
timeout-minutes: 60
|
||||
if: ${{ inputs.platforms == 'all' || inputs.platforms == 'web' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up the JDKs
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
# The metro gradle plugin folio builds with needs a 21 runtime; the
|
||||
# sidecar toolchain pins 17. Both are installed so gradle can pick.
|
||||
java-version: |
|
||||
17
|
||||
21
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
- name: Cache Gradle
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }}
|
||||
restore-keys: |
|
||||
folio-gradle-${{ runner.os }}-
|
||||
|
||||
- name: Set up Chrome
|
||||
uses: browser-actions/setup-chrome@v2
|
||||
with:
|
||||
chrome-version: stable
|
||||
|
||||
- name: Allow Chrome under unprivileged user namespaces
|
||||
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
|
||||
- name: Verify headless Chrome starts
|
||||
run: |
|
||||
chrome --version
|
||||
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
|
||||
--dump-dom 'data:text/html,<title>ok</title>'
|
||||
|
||||
- name: Build the folio wasmJs app
|
||||
run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution
|
||||
working-directory: examples/folio
|
||||
|
||||
- name: Build sanderling
|
||||
run: make sanderling-web
|
||||
|
||||
- name: Fuzz folio in the browser
|
||||
run: .github/scripts/folio-run.sh web
|
||||
env:
|
||||
SEED: ${{ inputs.seed != '0' && inputs.seed || '3' }}
|
||||
MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }}
|
||||
DURATION: ${{ inputs.duration }}
|
||||
|
||||
- name: Upload the run
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: folio-web
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
@@ -11,26 +11,55 @@ on:
|
||||
required: true
|
||||
type: string
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: release-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
resolve-tag:
|
||||
name: Resolve and validate the tag
|
||||
runs-on: ubuntu-latest
|
||||
permissions: {}
|
||||
outputs:
|
||||
tag: ${{ steps.tag.outputs.tag }}
|
||||
version: ${{ steps.tag.outputs.version }}
|
||||
steps:
|
||||
# A refname is attacker-controlled text and git permits backtick, `$`,
|
||||
# `(`, `;`, `&` and `|` in it, so it goes through env: a `${{ }}` is
|
||||
# substituted before bash ever sees the line. Every later job reads these
|
||||
# outputs rather than the refname, and nothing reaches a shell before it
|
||||
# has matched the pattern. The pattern is anchored and admits no newline,
|
||||
# which is what stops the value below forging a second $GITHUB_OUTPUT key.
|
||||
- name: Validate the tag
|
||||
id: tag
|
||||
run: |
|
||||
pattern='^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z]+(\.[0-9A-Za-z]+)*)?$'
|
||||
if [[ ! "$TAG" =~ $pattern ]]; then
|
||||
echo "release: refusing to publish from '$TAG'" >&2
|
||||
echo "release: a release tag is vMAJOR.MINOR.PATCH with an optional -prerelease, e.g. v0.1.0 or v0.0.1-rc1" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
|
||||
echo "version=${TAG#v}" >> "$GITHUB_OUTPUT"
|
||||
env:
|
||||
TAG: ${{ inputs.tag || github.ref_name }}
|
||||
|
||||
release-npm:
|
||||
name: Publish @sanderling/spec to npm
|
||||
needs: resolve-tag
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
ref: ${{ inputs.tag || github.ref }}
|
||||
|
||||
- name: Resolve version
|
||||
id: ver
|
||||
run: |
|
||||
raw="${{ inputs.tag || github.ref_name }}"
|
||||
echo "version=${raw#v}" >> "$GITHUB_OUTPUT"
|
||||
ref: ${{ needs.resolve-tag.outputs.tag }}
|
||||
# `npm ci` below runs dependency lifecycle scripts, and no step in
|
||||
# this job needs the git credential afterwards.
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Node 22
|
||||
uses: actions/setup-node@v7
|
||||
@@ -46,28 +75,36 @@ jobs:
|
||||
|
||||
- name: Stamp version
|
||||
working-directory: pkg/spec
|
||||
run: npm version ${{ steps.ver.outputs.version }} --no-git-tag-version --allow-same-version
|
||||
run: npm version "$VERSION" --no-git-tag-version --allow-same-version
|
||||
env:
|
||||
VERSION: ${{ needs.resolve-tag.outputs.version }}
|
||||
|
||||
- name: Publish
|
||||
working-directory: pkg/spec
|
||||
# npm tag pre-releases (e.g. 0.1.0-rc1) as "next" so npm install @sanderling/spec
|
||||
# keeps resolving the latest stable.
|
||||
run: |
|
||||
if [[ "${{ steps.ver.outputs.version }}" == *-* ]]; then
|
||||
if [[ "$VERSION" == *-* ]]; then
|
||||
npm publish --access public --tag next
|
||||
else
|
||||
npm publish --access public
|
||||
fi
|
||||
# The publish credential is scoped to the one step that publishes rather
|
||||
# than to the job, so no other step runs with it in reach.
|
||||
env:
|
||||
VERSION: ${{ needs.resolve-tag.outputs.version }}
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
release-cli:
|
||||
name: Publish sanderling CLI to GitHub Releases
|
||||
needs: resolve-tag
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
ref: ${{ inputs.tag || github.ref }}
|
||||
ref: ${{ needs.resolve-tag.outputs.tag }}
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Set up Go
|
||||
@@ -83,7 +120,7 @@ jobs:
|
||||
java-version: "17"
|
||||
|
||||
- name: Set up Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
|
||||
|
||||
- name: Cache Gradle
|
||||
uses: actions/cache@v6
|
||||
@@ -99,7 +136,7 @@ jobs:
|
||||
run: make sidecar
|
||||
|
||||
- name: Run GoReleaser
|
||||
uses: goreleaser/goreleaser-action@v7
|
||||
uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3
|
||||
with:
|
||||
version: "~> v2"
|
||||
args: release --clean
|
||||
|
||||
@@ -1,142 +0,0 @@
|
||||
name: replay-ui
|
||||
|
||||
# Sanderling fuzzing sanderling's own replay UI. Dispatch-only: it takes minutes
|
||||
# and it is a demo of the product loop, not a merge gate.
|
||||
#
|
||||
# The shape is: produce a real trace, serve it with `sanderling replay`, then run
|
||||
# a spec against that UI. Any violation fails the job. Six of the seven
|
||||
# properties in replay-ui/sanderling/spec.ts are cross-panel agreements that hold
|
||||
# for any trace; the seventh is the stock noUncaughtExceptions. None of them
|
||||
# needs recalibrating when the fixture changes.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
seed:
|
||||
description: seed for the dogfood run
|
||||
default: "3"
|
||||
max-steps:
|
||||
description: step budget for the dogfood run
|
||||
default: "80"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
dogfood:
|
||||
timeout-minutes: 45
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3.13"
|
||||
|
||||
# Pinned stable plus the AppArmor sysctl: the same setup ci.yml's browser
|
||||
# job needs to get headless Chrome up on ubuntu-latest.
|
||||
- name: Set up Chrome
|
||||
uses: browser-actions/setup-chrome@v2
|
||||
with:
|
||||
chrome-version: stable
|
||||
|
||||
- name: Allow Chrome under unprivileged user namespaces
|
||||
run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
|
||||
- name: Verify headless Chrome starts
|
||||
run: |
|
||||
chrome --version
|
||||
chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \
|
||||
--dump-dom 'data:text/html,<title>ok</title>'
|
||||
|
||||
# The UI the spec drives is the one embedded in this binary, so the build
|
||||
# has to come after any change to replay-ui/src.
|
||||
- name: Build sanderling
|
||||
run: make sanderling-web
|
||||
|
||||
# A trace with a violation and uncaught exceptions in it, so the UI has
|
||||
# something to render in every panel the spec looks at. No
|
||||
# --exit-on-violation here: the run is the fixture, and stopping it at the
|
||||
# first violation would leave a four-step trace to fuzz.
|
||||
- name: Record a fixture trace
|
||||
run: |
|
||||
python3 -m http.server 8792 --bind 127.0.0.1 \
|
||||
--directory test/browser/testdata/throwing &
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf http://127.0.0.1:8792/ >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "the fixture http server never answered on 127.0.0.1:8792" >&2
|
||||
exit 1
|
||||
fi
|
||||
./bin/sanderling test \
|
||||
--platform web \
|
||||
--spec test/browser/testdata/throwing/spec.ts \
|
||||
--bundle-id http://127.0.0.1:8792/ \
|
||||
--duration 5m --max-steps 25 --seed 7 \
|
||||
--output runs/fixture
|
||||
|
||||
- name: Serve the trace with sanderling replay
|
||||
run: |
|
||||
# Flags before the positional argument: Go's flag package stops
|
||||
# parsing at the first non-flag word.
|
||||
./bin/sanderling replay --port 8793 --no-open runs/fixture &
|
||||
ready=""
|
||||
for _ in $(seq 1 30); do
|
||||
curl -sf http://127.0.0.1:8793/api/runs >/dev/null && { ready=1; break; }
|
||||
sleep 1
|
||||
done
|
||||
if [ -z "$ready" ]; then
|
||||
echo "sanderling replay never served /api/runs on 127.0.0.1:8793" >&2
|
||||
exit 1
|
||||
fi
|
||||
run_id="$(ls runs/fixture | head -1)"
|
||||
echo "RUN_URL=http://127.0.0.1:8793/runs/$run_id/steps/1" >> "$GITHUB_ENV"
|
||||
curl -sf "http://127.0.0.1:8793/runs/$run_id/steps/1" >/dev/null
|
||||
|
||||
# Inputs go through env rather than into the script text: a `${{ }}` is
|
||||
# substituted before bash ever sees the line, so a seed of `$(id)` would
|
||||
# run as a command.
|
||||
- name: Fuzz the replay UI
|
||||
run: |
|
||||
./bin/sanderling test \
|
||||
--platform web \
|
||||
--spec replay-ui/sanderling/spec.ts \
|
||||
--bundle-id "$RUN_URL" \
|
||||
--duration 10m \
|
||||
--max-steps "$MAX_STEPS" \
|
||||
--seed "$SEED" \
|
||||
--exit-on-violation \
|
||||
--output runs/dogfood
|
||||
env:
|
||||
SEED: ${{ inputs.seed }}
|
||||
MAX_STEPS: ${{ inputs.max-steps }}
|
||||
|
||||
# Exit 0 above means no property returned false. It does not mean any
|
||||
# property was ever evaluated against real content: they all decline to
|
||||
# judge when the elements they read are absent, so a run that never
|
||||
# rendered the step page is green and worthless. This step is what tells
|
||||
# the two apart, and it fails the job when nothing was judged.
|
||||
- name: Summarise
|
||||
if: always()
|
||||
run: .github/scripts/replay-ui-summary.sh runs/dogfood
|
||||
env:
|
||||
SEED: ${{ inputs.seed }}
|
||||
MAX_STEPS: ${{ inputs.max-steps }}
|
||||
|
||||
- name: Upload runs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: replay-ui-runs
|
||||
path: runs/
|
||||
retention-days: 14
|
||||
Reference in new issue
Block a user