diff --git a/.github/workflows/examples.yml b/.github/workflows/examples.yml new file mode 100644 index 0000000..08780bc --- /dev/null +++ b/.github/workflows/examples.yml @@ -0,0 +1,193 @@ +name: examples + +# Every example sanderling ships, fuzzed the same way: build sanderling for a +# platform, bring the target up, run a spec against it, classify the trace it +# wrote, upload the run. Only the bring-up differs, and that lives in the +# per-target actions under .github/actions/. +# +# Dispatch-only: these take tens of minutes and they demonstrate the product +# loop, they do not gate a merge. +# +# folio on ios and in the browser expect the bug: folio double-submits a +# transaction on a double tap, so the run is supposed to end with exit 2. Exit 0 +# means the fuzzer stopped finding a bug that is still there; exit 1 means the +# harness broke. The two are worth telling apart, which is why +# --exit-on-violation exits 2 and not 1. +# +# folio on android is a health gate. It convicts in four runs out of five, which +# is real evidence but not a gate: the fifth would report a regression it had +# not found. Its budget is set so the conviction it usually gets is a bonus. +# +# the replay ui leg fuzzes sanderling's own replay UI, and any violation fails +# it. Its properties are cross-panel agreements that hold for any trace, so none +# of them needs recalibrating when the fixture changes. + +on: + workflow_dispatch: + inputs: + targets: + description: which examples to fuzz + type: choice + options: [all, folio, android, ios, web, replay-ui] + default: all + seed: + description: seed override (0 = each target's calibrated seed) + default: "0" + max-steps: + description: step budget override (0 = each target's calibrated budget) + default: "0" + duration: + description: wall-clock budget override (empty = each target's calibrated budget) + default: "" + +permissions: + contents: read + +jobs: + plan: + runs-on: ubuntu-latest + permissions: {} + outputs: + examples: ${{ steps.pick.outputs.examples }} + steps: + # The matrix is built here rather than written out under strategy.matrix + # because a job-level `if:` cannot read the matrix context, so a static + # matrix has no way to leave a leg out. jq -c keeps the value on one line, + # which is what makes the $GITHUB_OUTPUT write below safe. + - name: Pick the examples to fuzz + id: pick + run: | + examples='[ + {"target":"android","name":"folio on android","app":"folio", + "runs-on":"ubuntu-latest","timeout":90,"sanderling":"android", + "seed":"9","max-steps":"200","duration":"20m","artifact":"folio-android"}, + {"target":"ios","name":"folio on ios","app":"folio", + "runs-on":"macos-15","timeout":90,"sanderling":"ios", + "seed":"7","max-steps":"240","duration":"20m","artifact":"folio-ios"}, + {"target":"web","name":"folio in the browser","app":"folio", + "runs-on":"ubuntu-latest","timeout":60,"sanderling":"web","chrome":true, + "seed":"3","max-steps":"240","duration":"20m","artifact":"folio-web"}, + {"target":"replay-ui","name":"the replay ui","app":"replay-ui", + "runs-on":"ubuntu-latest","timeout":45,"sanderling":"web","chrome":true, + "seed":"3","max-steps":"80","duration":"10m","artifact":"replay-ui-runs"} + ]' + picked="$(jq -c --arg want "$TARGETS" \ + 'map(select($want == "all" or .target == $want or .app == $want))' \ + <<<"$examples")" + if [ "$picked" = "[]" ]; then + echo "examples: '$TARGETS' selects no example, so this dispatch would run nothing" >&2 + exit 1 + fi + echo "examples=$picked" >> "$GITHUB_OUTPUT" + env: + TARGETS: ${{ inputs.targets }} + + fuzz: + needs: plan + name: fuzz ${{ matrix.name }} + runs-on: ${{ matrix.runs-on }} + timeout-minutes: ${{ matrix.timeout }} + strategy: + # Each leg is its own evidence. One target failing must not cancel the + # others, which is how these ran as separate jobs. + fail-fast: false + matrix: + include: ${{ fromJSON(needs.plan.outputs.examples) }} + env: + SEED: ${{ inputs.seed != '0' && inputs.seed || matrix.seed }} + MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || matrix.max-steps }} + DURATION: ${{ inputs.duration != '' && inputs.duration || matrix.duration }} + IOS_DEVICE: iPhone 16 Pro + steps: + - uses: actions/checkout@v7 + + - name: Set up Go + uses: actions/setup-go@v7 + with: + go-version-file: go.mod + cache: true + + - name: Set up bun + uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0 + with: + bun-version: "1.3.13" + + - name: Set up headless Chrome + if: matrix.chrome + uses: ./.github/actions/headless-chrome + + - name: Build the folio app + if: matrix.app == 'folio' + uses: ./.github/actions/folio-app + with: + platform: ${{ matrix.target }} + + # The UI the replay-ui spec drives is the one embedded in this binary, so + # the build has to come after any change to replay-ui/src. + - name: Build sanderling + run: make "sanderling-$SANDERLING" + env: + SANDERLING: ${{ matrix.sanderling }} + + - name: Put folio on the simulator + if: matrix.target == 'ios' + uses: ./.github/actions/folio-simulator + + - name: Serve a trace to fuzz + id: fixture + if: matrix.target == 'replay-ui' + uses: ./.github/actions/replay-ui-fixture + + - name: Fuzz folio on an emulator + if: matrix.target == 'android' + uses: reactivecircus/android-emulator-runner@a421e43855164a8197daf9d8d40fe71c6996bb0d # v2.38.0 + with: + api-level: 34 + target: google_apis + arch: x86_64 + emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim + disable-animations: true + script: .github/scripts/folio-run.sh android + + - name: Fuzz folio + if: matrix.app == 'folio' && matrix.target != 'android' + run: .github/scripts/folio-run.sh "$TARGET" + env: + TARGET: ${{ matrix.target }} + + # Inputs go through env rather than into the script text: a `${{ }}` is + # substituted before bash ever sees the line, so a seed of `$(id)` would + # run as a command. + - name: Fuzz the replay UI + if: matrix.target == 'replay-ui' + run: | + ./bin/sanderling test \ + --platform web \ + --spec replay-ui/sanderling/spec.ts \ + --bundle-id "$RUN_URL" \ + --duration "$DURATION" \ + --max-steps "$MAX_STEPS" \ + --seed "$SEED" \ + --exit-on-violation \ + --output runs/replay-ui + env: + RUN_URL: ${{ steps.fixture.outputs.url }} + + # Exit 0 above means no property returned false. It does not mean any + # property was ever evaluated against real content: they all decline to + # judge when the elements they read are absent, so a run that never + # rendered the step page is green and worthless. This step is what tells + # the two apart, and it fails the job when nothing was judged. folio's + # legs make the same call inside folio-run.sh, where the exit code it is + # judging is in scope. + - name: Classify the replay UI run + if: ${{ always() && matrix.target == 'replay-ui' }} + run: .github/scripts/replay-ui-summary.sh runs/replay-ui + + - name: Upload the run + if: always() + uses: actions/upload-artifact@v7 + with: + name: ${{ matrix.artifact }} + path: runs/ + retention-days: 14 diff --git a/.github/workflows/folio.yml b/.github/workflows/folio.yml deleted file mode 100644 index 72facbc..0000000 --- a/.github/workflows/folio.yml +++ /dev/null @@ -1,277 +0,0 @@ -name: folio - -# One spec, three platforms. Dispatch-only: each job boots a device or a -# browser, builds the folio app for that platform, and runs -# examples/folio/sanderling/spec.ts against it. -# -# web and ios are expect-the-bug jobs: folio double-submits a transaction on a -# double tap, so the run is supposed to end with exit 2. Exit 0 means the fuzzer -# stopped finding a bug that is still there; exit 1 means the harness broke. The -# two are worth telling apart, which is why --exit-on-violation exits 2 and not -# 1. -# -# android is a health gate. It convicts in four runs out of five, which is real -# evidence but not a gate: the fifth would report a regression it had not found. -# The budget is set so the conviction it usually gets is reported as a bonus. - -on: - workflow_dispatch: - inputs: - platforms: - description: which legs to run - type: choice - options: [all, android, ios, web] - default: all - seed: - description: seed override (0 = each job's calibrated seed) - default: "0" - duration: - description: wall-clock budget per run - default: 20m - max-steps: - description: step budget override (0 = each job's calibrated budget) - default: "0" - -permissions: - contents: read - -jobs: - android: - timeout-minutes: 90 - if: ${{ inputs.platforms == 'all' || inputs.platforms == 'android' }} - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: go.mod - cache: true - - - name: Set up the JDKs - uses: actions/setup-java@v5 - with: - distribution: temurin - # The metro gradle plugin folio builds with needs a 21 runtime; the - # sidecar toolchain pins 17. Both are installed so gradle can pick. - java-version: | - 17 - 21 - - - name: Set up Android SDK - uses: android-actions/setup-android@v4 - - - name: Set up bun - uses: oven-sh/setup-bun@v2 - with: - bun-version: "1.3.13" - - - name: Cache Gradle - uses: actions/cache@v6 - with: - path: | - ~/.gradle/caches - ~/.gradle/wrapper - key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }} - restore-keys: | - folio-gradle-${{ runner.os }}- - - # Without this the emulator falls back to software rendering and every - # step costs several seconds. - - name: Enable KVM - run: | - echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \ - | sudo tee /etc/udev/rules.d/99-kvm4all.rules - sudo udevadm control --reload-rules - sudo udevadm trigger --name-match=kvm - - - name: Build the folio APK - run: ./gradlew :app:androidApp:assembleDebug - working-directory: examples/folio - - - name: Build sanderling - run: make sanderling-android - - - name: Fuzz folio on an emulator - uses: reactivecircus/android-emulator-runner@v2 - with: - api-level: 34 - target: google_apis - arch: x86_64 - emulator-options: -no-window -gpu swiftshader_indirect -no-snapshot -noaudio -no-boot-anim - disable-animations: true - script: .github/scripts/folio-run.sh android - env: - SEED: ${{ inputs.seed != '0' && inputs.seed || '9' }} - MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '200' }} - DURATION: ${{ inputs.duration }} - - - name: Upload the run - if: always() - uses: actions/upload-artifact@v7 - with: - name: folio-android - path: runs/ - retention-days: 14 - - ios: - timeout-minutes: 90 - if: ${{ inputs.platforms == 'all' || inputs.platforms == 'ios' }} - runs-on: macos-15 - steps: - - uses: actions/checkout@v7 - - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: go.mod - cache: true - - - name: Set up bun - uses: oven-sh/setup-bun@v2 - with: - bun-version: "1.3.13" - - # idb-companion is not in homebrew-core, only in facebook/homebrew-fb, so - # it has to be named by its full tap path. xcodegen and just are core. - - name: Install idb-companion, xcodegen and just - run: brew install facebook/fb/idb-companion xcodegen just - - - name: Set up the JDKs - uses: actions/setup-java@v5 - with: - distribution: temurin - # The metro gradle plugin folio builds with needs a 21 runtime; the - # sidecar toolchain pins 17. Both are installed so gradle can pick. - java-version: | - 17 - 21 - - # The iOS app builds its Kotlin framework through the folio gradle - # project, which configures :app:androidApp and so needs an Android SDK - # even on this leg. - - name: Set up Android SDK - uses: android-actions/setup-android@v4 - - # Both asset tarballs are built by the prepare scripts, and the runner - # bundle is an xcodebuild of companion/Sources. Keyed on the scripts and - # the versions the Makefile embeds, so a later run reuses them. - - name: Cache the companion and runner bundles - uses: actions/cache@v6 - with: - path: | - internal/driver/ioscompanion/companionassets/assets - internal/driver/ioscompanion/runnerassets/assets - key: ios-assets-${{ runner.os }}-${{ hashFiles('internal/driver/ioscompanion/companionassets/prepare.sh', 'companion/prepare.sh', 'companion/project.yml', 'companion/Sources/**') }} - - - name: Build sanderling - run: make sanderling-ios - - - name: Boot a simulator - run: | - xcrun simctl boot "$IOS_DEVICE" || true - xcrun simctl bootstatus "$IOS_DEVICE" -b - env: - IOS_DEVICE: iPhone 16 Pro - - - name: Build and install folio - run: just ios - working-directory: examples/folio - env: - IOS_DEVICE: iPhone 16 Pro - - # `just ios` leaves the app running, and the run's first act is to clear - # its state. Stopping it here means the run always opens the same way. - - name: Stop the app before the run - run: xcrun simctl terminate booted app.folio || true - - - name: Fuzz folio on the simulator - run: .github/scripts/folio-run.sh ios - env: - SEED: ${{ inputs.seed != '0' && inputs.seed || '7' }} - MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }} - DURATION: ${{ inputs.duration }} - IOS_DEVICE: iPhone 16 Pro - - - name: Upload the run - if: always() - uses: actions/upload-artifact@v7 - with: - name: folio-ios - path: runs/ - retention-days: 14 - - web: - timeout-minutes: 60 - if: ${{ inputs.platforms == 'all' || inputs.platforms == 'web' }} - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: go.mod - cache: true - - - name: Set up the JDKs - uses: actions/setup-java@v5 - with: - distribution: temurin - # The metro gradle plugin folio builds with needs a 21 runtime; the - # sidecar toolchain pins 17. Both are installed so gradle can pick. - java-version: | - 17 - 21 - - - name: Set up bun - uses: oven-sh/setup-bun@v2 - with: - bun-version: "1.3.13" - - - name: Cache Gradle - uses: actions/cache@v6 - with: - path: | - ~/.gradle/caches - ~/.gradle/wrapper - key: folio-gradle-${{ runner.os }}-${{ hashFiles('**/*.gradle*', '**/gradle-wrapper.properties') }} - restore-keys: | - folio-gradle-${{ runner.os }}- - - - name: Set up Chrome - uses: browser-actions/setup-chrome@v2 - with: - chrome-version: stable - - - name: Allow Chrome under unprivileged user namespaces - run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0 - - - name: Verify headless Chrome starts - run: | - chrome --version - chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \ - --dump-dom 'data:text/html,ok' - - - name: Build the folio wasmJs app - run: ./gradlew :app:webApp:wasmJsBrowserDevelopmentExecutableDistribution - working-directory: examples/folio - - - name: Build sanderling - run: make sanderling-web - - - name: Fuzz folio in the browser - run: .github/scripts/folio-run.sh web - env: - SEED: ${{ inputs.seed != '0' && inputs.seed || '3' }} - MAX_STEPS: ${{ inputs.max-steps != '0' && inputs.max-steps || '240' }} - DURATION: ${{ inputs.duration }} - - - name: Upload the run - if: always() - uses: actions/upload-artifact@v7 - with: - name: folio-web - path: runs/ - retention-days: 14 diff --git a/.github/workflows/replay-ui.yml b/.github/workflows/replay-ui.yml deleted file mode 100644 index 6fb4f4a..0000000 --- a/.github/workflows/replay-ui.yml +++ /dev/null @@ -1,142 +0,0 @@ -name: replay-ui - -# Sanderling fuzzing sanderling's own replay UI. Dispatch-only: it takes minutes -# and it is a demo of the product loop, not a merge gate. -# -# The shape is: produce a real trace, serve it with `sanderling replay`, then run -# a spec against that UI. Any violation fails the job. Six of the seven -# properties in replay-ui/sanderling/spec.ts are cross-panel agreements that hold -# for any trace; the seventh is the stock noUncaughtExceptions. None of them -# needs recalibrating when the fixture changes. - -on: - workflow_dispatch: - inputs: - seed: - description: seed for the dogfood run - default: "3" - max-steps: - description: step budget for the dogfood run - default: "80" - -permissions: - contents: read - -jobs: - dogfood: - timeout-minutes: 45 - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: go.mod - cache: true - - - name: Set up bun - uses: oven-sh/setup-bun@v2 - with: - bun-version: "1.3.13" - - # Pinned stable plus the AppArmor sysctl: the same setup ci.yml's browser - # job needs to get headless Chrome up on ubuntu-latest. - - name: Set up Chrome - uses: browser-actions/setup-chrome@v2 - with: - chrome-version: stable - - - name: Allow Chrome under unprivileged user namespaces - run: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0 - - - name: Verify headless Chrome starts - run: | - chrome --version - chrome --headless --no-sandbox --disable-gpu --disable-dev-shm-usage \ - --dump-dom 'data:text/html,ok' - - # The UI the spec drives is the one embedded in this binary, so the build - # has to come after any change to replay-ui/src. - - name: Build sanderling - run: make sanderling-web - - # A trace with a violation and uncaught exceptions in it, so the UI has - # something to render in every panel the spec looks at. No - # --exit-on-violation here: the run is the fixture, and stopping it at the - # first violation would leave a four-step trace to fuzz. - - name: Record a fixture trace - run: | - python3 -m http.server 8792 --bind 127.0.0.1 \ - --directory test/browser/testdata/throwing & - ready="" - for _ in $(seq 1 30); do - curl -sf http://127.0.0.1:8792/ >/dev/null && { ready=1; break; } - sleep 1 - done - if [ -z "$ready" ]; then - echo "the fixture http server never answered on 127.0.0.1:8792" >&2 - exit 1 - fi - ./bin/sanderling test \ - --platform web \ - --spec test/browser/testdata/throwing/spec.ts \ - --bundle-id http://127.0.0.1:8792/ \ - --duration 5m --max-steps 25 --seed 7 \ - --output runs/fixture - - - name: Serve the trace with sanderling replay - run: | - # Flags before the positional argument: Go's flag package stops - # parsing at the first non-flag word. - ./bin/sanderling replay --port 8793 --no-open runs/fixture & - ready="" - for _ in $(seq 1 30); do - curl -sf http://127.0.0.1:8793/api/runs >/dev/null && { ready=1; break; } - sleep 1 - done - if [ -z "$ready" ]; then - echo "sanderling replay never served /api/runs on 127.0.0.1:8793" >&2 - exit 1 - fi - run_id="$(ls runs/fixture | head -1)" - echo "RUN_URL=http://127.0.0.1:8793/runs/$run_id/steps/1" >> "$GITHUB_ENV" - curl -sf "http://127.0.0.1:8793/runs/$run_id/steps/1" >/dev/null - - # Inputs go through env rather than into the script text: a `${{ }}` is - # substituted before bash ever sees the line, so a seed of `$(id)` would - # run as a command. - - name: Fuzz the replay UI - run: | - ./bin/sanderling test \ - --platform web \ - --spec replay-ui/sanderling/spec.ts \ - --bundle-id "$RUN_URL" \ - --duration 10m \ - --max-steps "$MAX_STEPS" \ - --seed "$SEED" \ - --exit-on-violation \ - --output runs/dogfood - env: - SEED: ${{ inputs.seed }} - MAX_STEPS: ${{ inputs.max-steps }} - - # Exit 0 above means no property returned false. It does not mean any - # property was ever evaluated against real content: they all decline to - # judge when the elements they read are absent, so a run that never - # rendered the step page is green and worthless. This step is what tells - # the two apart, and it fails the job when nothing was judged. - - name: Summarise - if: always() - run: .github/scripts/replay-ui-summary.sh runs/dogfood - env: - SEED: ${{ inputs.seed }} - MAX_STEPS: ${{ inputs.max-steps }} - - - name: Upload runs - if: always() - uses: actions/upload-artifact@v7 - with: - name: replay-ui-runs - path: runs/ - retention-days: 14