Merge branch 'runner-uncertain-last-action' into pr-73-followups

This commit is contained in:
pj committed 2026-08-15 14:04:56 +05:30
commit df0cb96386
14 files changed
+428 -60

No files matched your search

+10 -3
View File
@@ -298,11 +298,18 @@ func Run(ctx context.Context, options Options) (Summary, error) {
logger.Warn("apply error; marking step transitional", "step", stepIndex, "err", err)
transitional = true
applySkipped = true
lastAction = nil
// The error says the call failed, not that the gesture never
// reached the app: a deadline that fires after dispatch leaves
// the effect committed. Reporting no action here would let a
// property convict the app for an effect with no cause, so the
// action is reported with its fate unknown instead.
unconfirmed := nextAction
lastAction = &unconfirmed
} else {
consecutiveApplyFailures = 0
actionCopy := nextAction
lastAction = &actionCopy
applied := nextAction
applied.Applied = true
lastAction = &applied
}
} else {
lastAction = nil
@@ -0,0 +1,150 @@
package runner
import (
"context"
"errors"
"fmt"
"path/filepath"
"sync/atomic"
"testing"
"time"
"github.com/priyanshujain/sanderling/internal/driver"
mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock"
)
// An apply error is not proof that nothing landed. An RPC deadline that fires
// after the tap was dispatched leaves the transaction committed, and a runner
// that reports "no action" for it hands
// submitCommitsOneTransactionPerAction a rise of one transaction against a
// window of zero submits: a conviction manufactured out of the runner's own
// uncertainty, on the property carrying most of the detection on android.
//
// The spec below is the real folio predicate pair, imported from the example,
// so what this asserts is the verdict the shipped property reaches.
const uncertainApplySpecTemplate = `
import { actions, always, extract, next, Tap } from "@sanderling/spec";
import {
committedTransactionsExceedSubmits,
countSubmitsInWindow,
} from "%s";
let submits = 0;
const submitsInWindow = extract("submitsInWindow", state => {
const window = countSubmitsInWindow({
previousCount: submits,
lastAction: state.lastAction,
fresh: true,
});
submits = window.next;
return window.reported;
});
const counts = extract("counts", state => {
const text = state.ax.find("id:TxnCount")?.text;
return text ? { Travel: parseInt(text, 10) } : null;
});
globalThis.properties = {
submitCommitsOneTransactionPerAction: always(
next(() =>
!committedTransactionsExceedSubmits({
countsBefore: counts.previous ?? null,
countsAfter: counts.current,
submitsInWindow: submitsInWindow.current,
}),
),
),
};
globalThis.actions = actions(() => [Tap({ on: "id:TxnSubmit" })]);
`
const homeWithTxnCount = `{"attributes":{"resource-id":"HomeScreen"},"children":[
{"attributes":{"resource-id":"TxnCount","text":"%d"},"children":[]},
{"attributes":{"resource-id":"TxnSubmit","bounds":"[40,80,240,160]"},"children":[],"clickable":true,"enabled":true}
]}`
// dispatchThenFailDriver is the device condition the runner cannot see through:
// the tap reaches the app and commits, then the call the runner is waiting on
// times out. Every later hierarchy read shows the committed transactions.
type dispatchThenFailDriver struct {
*mockdriver.Driver
commitsPerTap int64
committed atomic.Int64
}
func (d *dispatchThenFailDriver) Tap(context.Context, int, int) error {
return d.dispatchThenFail()
}
func (d *dispatchThenFailDriver) TapSelector(context.Context, string) error {
return d.dispatchThenFail()
}
func (d *dispatchThenFailDriver) dispatchThenFail() error {
d.committed.Add(d.commitsPerTap)
return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded")
}
func (d *dispatchThenFailDriver) Snapshot(context.Context) (string, driver.Image, error) {
return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil
}
func TestRunner_ApplyErrorAfterDispatchDoesNotConvictTheSubmitCountingProperty(t *testing.T) {
predicates, err := filepath.Abs("../../examples/folio/sanderling/predicates.ts")
if err != nil {
t.Fatal(err)
}
spec := fmt.Sprintf(uncertainApplySpecTemplate, predicates)
run := func(t *testing.T, commitsPerTap int64) []ViolationRecord {
t.Helper()
state := newHarnessWithSpec(t, spec)
device := &dispatchThenFailDriver{Driver: state.mock, commitsPerTap: commitsPerTap}
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
summary, err := Run(ctx, Options{
Duration: time.Hour,
IdleTimeout: 20 * time.Millisecond,
MaxSteps: 2,
Driver: device,
Verifier: state.verifier,
TraceWriter: state.writer,
})
if err != nil {
t.Fatalf("Run: %v", err)
}
if summary.Steps != 2 {
t.Fatalf("steps = %d, want 2; the run never reached the step that judges the pair", summary.Steps)
}
if got := device.committed.Load(); got != commitsPerTap*2 {
t.Fatalf("the device committed %d transaction(s), want %d; the taps never reached it",
got, commitsPerTap*2)
}
return summary.Violations
}
t.Run("one transaction per tap is not a double submit", func(t *testing.T) {
if violations := run(t, 1); len(violations) != 0 {
t.Errorf("the counting property convicted a healthy app: %v\n"+
"one transaction rose against a submit the runner dispatched but "+
"could not confirm, and the spec was told no action happened",
violations)
}
})
// The control. Without it a green above proves nothing: a property that
// never sees a comparable pair is silently vacuous and reports the same
// empty violation list.
t.Run("two transactions per tap still convicts", func(t *testing.T) {
violations := run(t, 2)
if len(violations) == 0 {
t.Fatal("the counting property missed a double submit; the harness never " +
"put the property in a position to fire, so the case above proves nothing")
}
if violations[0].Properties[0] != "submitCommitsOneTransactionPerAction" {
t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", violations[0].Properties)
}
})
}
+42 -1
View File
@@ -3,6 +3,7 @@ package runner
import (
"context"
"encoding/json"
"errors"
"testing"
"time"
@@ -71,7 +72,47 @@ func TestRunner_WebInstallsLastActionInThePage(t *testing.T) {
// Every later step carries what the runner actually applied. The shape is
// the goja host's (internal/verifier/marshal.go lastActionFields), pinned
// against it by TestLastAction_WebJSONMatchesTheGojaObject.
const want = `{"kind":"Tap","on":"id:TxnSubmit"}`
const want = `{"kind":"Tap","applied":true,"on":"id:TxnSubmit"}`
if web.installed[1] != want {
t.Errorf("step 2 installed %s, want %s", web.installed[1], want)
}
}
// failingTapWebDriver dispatches the tap and then fails the call, the shape an
// RPC deadline takes: the page has the click, the runner has an error.
type failingTapWebDriver struct {
*tappingWebDriver
}
func (d *failingTapWebDriver) Tap(context.Context, int, int) error {
return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded")
}
// The web leg of the same three states the goja host reports. "applied":null is
// not "no action": a property gated on the last action still sees the tap and
// decides for itself, which it cannot do if the page is handed a bare null.
func TestRunner_WebInstallsAnUnconfirmedActionWithItsFateUnknown(t *testing.T) {
state := newHarnessWithSpec(t, lastActionSpec)
web := &failingTapWebDriver{tappingWebDriver: &tappingWebDriver{Driver: state.mock}}
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
defer cancel()
if _, err := Run(ctx, Options{
Duration: time.Hour,
IdleTimeout: 20 * time.Millisecond,
MaxSteps: 2,
Driver: web,
Verifier: state.verifier,
TraceWriter: state.writer,
}); err != nil {
t.Fatalf("Run: %v", err)
}
if len(web.installed) < 2 {
t.Fatalf("the page was handed lastAction %d time(s); the web path never installed it",
len(web.installed))
}
const want = `{"kind":"Tap","applied":null,"on":"id:TxnSubmit"}`
if web.installed[1] != want {
t.Errorf("step 2 installed %s, want %s", web.installed[1], want)
}