Merge branch 'runner-uncertain-last-action' into pr-73-followups

This commit is contained in:
pj committed 2026-08-15 14:04:56 +05:30
commit df0cb96386
14 files changed
+428 -60

No files matched your search

+10 -3
View File
@@ -298,11 +298,18 @@ func Run(ctx context.Context, options Options) (Summary, error) {
logger.Warn("apply error; marking step transitional", "step", stepIndex, "err", err)
transitional = true
applySkipped = true
lastAction = nil
// The error says the call failed, not that the gesture never
// reached the app: a deadline that fires after dispatch leaves
// the effect committed. Reporting no action here would let a
// property convict the app for an effect with no cause, so the
// action is reported with its fate unknown instead.
unconfirmed := nextAction
lastAction = &unconfirmed
} else {
consecutiveApplyFailures = 0
actionCopy := nextAction
lastAction = &actionCopy
applied := nextAction
applied.Applied = true
lastAction = &applied
}
} else {
lastAction = nil
@@ -0,0 +1,150 @@
package runner
import (
"context"
"errors"
"fmt"
"path/filepath"
"sync/atomic"
"testing"
"time"
"github.com/priyanshujain/sanderling/internal/driver"
mockdriver "github.com/priyanshujain/sanderling/internal/driver/mock"
)
// An apply error is not proof that nothing landed. An RPC deadline that fires
// after the tap was dispatched leaves the transaction committed, and a runner
// that reports "no action" for it hands
// submitCommitsOneTransactionPerAction a rise of one transaction against a
// window of zero submits: a conviction manufactured out of the runner's own
// uncertainty, on the property carrying most of the detection on android.
//
// The spec below is the real folio predicate pair, imported from the example,
// so what this asserts is the verdict the shipped property reaches.
const uncertainApplySpecTemplate = `
import { actions, always, extract, next, Tap } from "@sanderling/spec";
import {
committedTransactionsExceedSubmits,
countSubmitsInWindow,
} from "%s";
let submits = 0;
const submitsInWindow = extract("submitsInWindow", state => {
const window = countSubmitsInWindow({
previousCount: submits,
lastAction: state.lastAction,
fresh: true,
});
submits = window.next;
return window.reported;
});
const counts = extract("counts", state => {
const text = state.ax.find("id:TxnCount")?.text;
return text ? { Travel: parseInt(text, 10) } : null;
});
globalThis.properties = {
submitCommitsOneTransactionPerAction: always(
next(() =>
!committedTransactionsExceedSubmits({
countsBefore: counts.previous ?? null,
countsAfter: counts.current,
submitsInWindow: submitsInWindow.current,
}),
),
),
};
globalThis.actions = actions(() => [Tap({ on: "id:TxnSubmit" })]);
`
const homeWithTxnCount = `{"attributes":{"resource-id":"HomeScreen"},"children":[
{"attributes":{"resource-id":"TxnCount","text":"%d"},"children":[]},
{"attributes":{"resource-id":"TxnSubmit","bounds":"[40,80,240,160]"},"children":[],"clickable":true,"enabled":true}
]}`
// dispatchThenFailDriver is the device condition the runner cannot see through:
// the tap reaches the app and commits, then the call the runner is waiting on
// times out. Every later hierarchy read shows the committed transactions.
type dispatchThenFailDriver struct {
*mockdriver.Driver
commitsPerTap int64
committed atomic.Int64
}
func (d *dispatchThenFailDriver) Tap(context.Context, int, int) error {
return d.dispatchThenFail()
}
func (d *dispatchThenFailDriver) TapSelector(context.Context, string) error {
return d.dispatchThenFail()
}
func (d *dispatchThenFailDriver) dispatchThenFail() error {
d.committed.Add(d.commitsPerTap)
return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded")
}
func (d *dispatchThenFailDriver) Snapshot(context.Context) (string, driver.Image, error) {
return fmt.Sprintf(homeWithTxnCount, d.committed.Load()), driver.Image{}, nil
}
func TestRunner_ApplyErrorAfterDispatchDoesNotConvictTheSubmitCountingProperty(t *testing.T) {
predicates, err := filepath.Abs("../../examples/folio/sanderling/predicates.ts")
if err != nil {
t.Fatal(err)
}
spec := fmt.Sprintf(uncertainApplySpecTemplate, predicates)
run := func(t *testing.T, commitsPerTap int64) []ViolationRecord {
t.Helper()
state := newHarnessWithSpec(t, spec)
device := &dispatchThenFailDriver{Driver: state.mock, commitsPerTap: commitsPerTap}
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
summary, err := Run(ctx, Options{
Duration: time.Hour,
IdleTimeout: 20 * time.Millisecond,
MaxSteps: 2,
Driver: device,
Verifier: state.verifier,
TraceWriter: state.writer,
})
if err != nil {
t.Fatalf("Run: %v", err)
}
if summary.Steps != 2 {
t.Fatalf("steps = %d, want 2; the run never reached the step that judges the pair", summary.Steps)
}
if got := device.committed.Load(); got != commitsPerTap*2 {
t.Fatalf("the device committed %d transaction(s), want %d; the taps never reached it",
got, commitsPerTap*2)
}
return summary.Violations
}
t.Run("one transaction per tap is not a double submit", func(t *testing.T) {
if violations := run(t, 1); len(violations) != 0 {
t.Errorf("the counting property convicted a healthy app: %v\n"+
"one transaction rose against a submit the runner dispatched but "+
"could not confirm, and the spec was told no action happened",
violations)
}
})
// The control. Without it a green above proves nothing: a property that
// never sees a comparable pair is silently vacuous and reports the same
// empty violation list.
t.Run("two transactions per tap still convicts", func(t *testing.T) {
violations := run(t, 2)
if len(violations) == 0 {
t.Fatal("the counting property missed a double submit; the harness never " +
"put the property in a position to fire, so the case above proves nothing")
}
if violations[0].Properties[0] != "submitCommitsOneTransactionPerAction" {
t.Errorf("violated %v, want submitCommitsOneTransactionPerAction", violations[0].Properties)
}
})
}
+42 -1
View File
@@ -3,6 +3,7 @@ package runner
import (
"context"
"encoding/json"
"errors"
"testing"
"time"
@@ -71,7 +72,47 @@ func TestRunner_WebInstallsLastActionInThePage(t *testing.T) {
// Every later step carries what the runner actually applied. The shape is
// the goja host's (internal/verifier/marshal.go lastActionFields), pinned
// against it by TestLastAction_WebJSONMatchesTheGojaObject.
const want = `{"kind":"Tap","on":"id:TxnSubmit"}`
const want = `{"kind":"Tap","applied":true,"on":"id:TxnSubmit"}`
if web.installed[1] != want {
t.Errorf("step 2 installed %s, want %s", web.installed[1], want)
}
}
// failingTapWebDriver dispatches the tap and then fails the call, the shape an
// RPC deadline takes: the page has the click, the runner has an error.
type failingTapWebDriver struct {
*tappingWebDriver
}
func (d *failingTapWebDriver) Tap(context.Context, int, int) error {
return errors.New("rpc error: code = DeadlineExceeded desc = context deadline exceeded")
}
// The web leg of the same three states the goja host reports. "applied":null is
// not "no action": a property gated on the last action still sees the tap and
// decides for itself, which it cannot do if the page is handed a bare null.
func TestRunner_WebInstallsAnUnconfirmedActionWithItsFateUnknown(t *testing.T) {
state := newHarnessWithSpec(t, lastActionSpec)
web := &failingTapWebDriver{tappingWebDriver: &tappingWebDriver{Driver: state.mock}}
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
defer cancel()
if _, err := Run(ctx, Options{
Duration: time.Hour,
IdleTimeout: 20 * time.Millisecond,
MaxSteps: 2,
Driver: web,
Verifier: state.verifier,
TraceWriter: state.writer,
}); err != nil {
t.Fatalf("Run: %v", err)
}
if len(web.installed) < 2 {
t.Fatalf("the page was handed lastAction %d time(s); the web path never installed it",
len(web.installed))
}
const want = `{"kind":"Tap","applied":null,"on":"id:TxnSubmit"}`
if web.installed[1] != want {
t.Errorf("step 2 installed %s, want %s", web.installed[1], want)
}
+14 -2
View File
@@ -344,7 +344,19 @@ func lastActionFields(action *Action) []actionField {
point := func(x, y int) []actionField {
return []actionField{{key: "x", value: x}, {key: "y", value: y}}
}
fields := []actionField{{key: "kind", value: string(action.Kind)}}
// An action whose apply call failed is not an action that did not happen:
// the dispatch may have landed before the error. That is unknown, and
// unknown is null here for the same reason every other absence in the spec
// surface is, so a property decides for itself instead of being handed a
// "nothing happened" the runner cannot vouch for.
var applied any
if action.Applied {
applied = true
}
fields := []actionField{
{key: "kind", value: string(action.Kind)},
{key: "applied", value: applied},
}
if action.On != "" {
fields = append(fields, actionField{key: "on", value: action.On})
}
@@ -397,7 +409,7 @@ func objectFromFields(runtime *goja.Runtime, fields []actionField) *goja.Object
// has no Go-side state object to read: the runner pushes this JSON into the
// page before each extractor evaluation. A nil action encodes as JSON null,
// the same value the goja host reports on the first step of a run and after a
// step whose action was never applied.
// step whose action was never dispatched.
func EncodeLastAction(action *Action) json.RawMessage {
if action == nil {
return json.RawMessage("null")
+40
View File
@@ -168,6 +168,7 @@ func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) {
}{
{"nil", nil},
{"Tap", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", X: 12, Y: 34}},
{"TapApplied", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}},
{"TapWithoutSelector", &Action{Kind: ActionKindTap, X: 12, Y: 34}},
{"DoubleTap", &Action{Kind: ActionKindDoubleTap, On: `desc:say "hi" <b>`}},
{"InputText", &Action{Kind: ActionKindInputText, On: "id:field", Text: "50"}},
@@ -193,3 +194,42 @@ func TestLastAction_WebJSONMatchesTheGojaObject(t *testing.T) {
})
}
}
// A spec has to be able to tell three things apart: no action ran, an action
// ran, and an action was dispatched whose fate the runner cannot vouch for.
// The third used to be reported as the first, which is how a property that
// reasons "an effect landed with no action to cause it" convicts an app over
// an RPC deadline.
func TestLastAction_SeparatesNoActionFromAnActionOfUnknownFate(t *testing.T) {
verifier := newVerifier(t)
mustLoad(t, verifier, `
globalThis.fate = __sanderling__.extract(state =>
state.lastAction === null ? "no action"
: state.lastAction.applied === true ? "applied"
: state.lastAction.applied === null ? "unknown"
: "unreadable");
`)
for _, testCase := range []struct {
name string
action *Action
want string
}{
{"nothing ran", nil, "no action"},
{"dispatch confirmed", &Action{Kind: ActionKindTap, On: "id:TxnSubmit", Applied: true}, "applied"},
{"dispatch unconfirmed", &Action{Kind: ActionKindTap, On: "id:TxnSubmit"}, "unknown"},
} {
t.Run(testCase.name, func(t *testing.T) {
if err := verifier.PushSnapshot(SnapshotInput{
Snapshots: Snapshots{},
LastAction: testCase.action,
}); err != nil {
t.Fatal(err)
}
handle := verifier.runtime.GlobalObject().Get("fate").ToObject(verifier.runtime)
if got := handle.Get("current").String(); got != testCase.want {
t.Errorf("the spec read %q, want %q", got, testCase.want)
}
})
}
}
+5
View File
@@ -33,6 +33,11 @@ type Action struct {
// Direction is the scroll direction for ActionKindScroll: one of "up",
// "down", "left", "right". Empty for every other kind.
Direction string
// Applied is meaningful only on the action a step reports to the spec as
// state.lastAction: true when the runner saw the dispatch succeed, false
// when the apply call failed and nothing can say whether the action
// reached the app. The spec is told which of the two it is.
Applied bool
}
// LogEntry mirrors a logcat line captured between steps.