diff --git a/packages/ad-replay/src/index.ts b/packages/ad-replay/src/index.ts index 4a4df28df6..2f533795f9 100644 --- a/packages/ad-replay/src/index.ts +++ b/packages/ad-replay/src/index.ts @@ -60,7 +60,7 @@ * and the classification/guard/binding-evidence/verification-routing shapes * `verifyAndDispatchStep` exchanges with the daemon's `AdReplayStepRuntime` * implementation (`AdReplayVerificationEntry`, `AdReplayTargetClassification`, - * `AdReplayTargetBindingEvidence`, `AdReplayDispatchGuard`, + * `AdReplayTargetObservation`, `AdReplayTargetBindingEvidence`, `AdReplayDispatchGuard`, * `AdReplayDispatchOutcome`) ARE named here: the * daemon builds/reads real values of these shapes directly now (routing in * `session-replay-target-verification.ts`, wire-narrowing in @@ -93,6 +93,7 @@ export type { AdReplayStepRuntime, AdReplayTargetBindingEvidence, AdReplayTargetClassification, + AdReplayTargetObservation, AdReplayVarSources, AdReplayVerificationEntry, } from './internal/runtime-port-types.ts'; diff --git a/packages/ad-replay/src/internal/__tests__/step-loop.test.ts b/packages/ad-replay/src/internal/__tests__/step-loop.test.ts index 0cd7ee6abf..bb2ce6a98f 100644 --- a/packages/ad-replay/src/internal/__tests__/step-loop.test.ts +++ b/packages/ad-replay/src/internal/__tests__/step-loop.test.ts @@ -63,11 +63,8 @@ function createFakeRuntime(params: { isRepairArmed?: () => boolean } = {}): { let armCount = 0; const runtime: AdReplayStepRuntime = { beginTargetVerification: () => ({ kind: 'inactive' }), - captureObservation: async () => { - throw new Error('captureObservation: not used by this fixture (no targetEvidence)'); - }, - classifyTarget: () => { - throw new Error('classifyTarget: not used by this fixture (no targetEvidence)'); + observeTarget: async () => { + throw new Error('observeTarget: not used by this fixture (no targetEvidence)'); }, async dispatchStep(dispatchedAction, _resolvedAction, _index, artifactPaths) { dispatched.push(dispatchedAction.command); @@ -175,13 +172,8 @@ test("a post-dispatch target-binding mismatch reports the pre-step artifact snap // called for it — routed to the #1349 deferred-landmark path, which // dispatches with a guard WITHOUT any capture/classify round trip. beginTargetVerification: () => ({ kind: 'post-resolution', isSelectorWait: true }), - captureObservation: async () => { - throw new Error( - 'captureObservation: not used — deferred-landmark skips straight to dispatch', - ); - }, - classifyTarget: () => { - throw new Error('classifyTarget: not used — deferred-landmark skips straight to dispatch'); + observeTarget: async () => { + throw new Error('observeTarget: not used — deferred-landmark skips straight to dispatch'); }, async dispatchStep(dispatchedAction, _resolvedAction, _index, artifactPaths, _guard) { if (dispatchedAction.command === 'open') { diff --git a/packages/ad-replay/src/internal/runtime-port-types.ts b/packages/ad-replay/src/internal/runtime-port-types.ts index 3d21be54bb..502c8ce457 100644 --- a/packages/ad-replay/src/internal/runtime-port-types.ts +++ b/packages/ad-replay/src/internal/runtime-port-types.ts @@ -97,9 +97,9 @@ export type AdReplayVerifiedTargetGuard = Readonly<{ matchCount: number; }>; -/** `captureObservation`'s neutral result: nodes for classification, or why a capture was not available. */ -export type AdReplayObservation = Readonly< - | { readonly state: 'available'; readonly nodes: readonly SnapshotNode[] } +/** `observeTarget`'s neutral result: the recorded target's classification, or why no capture was available. */ +export type AdReplayTargetObservation = Readonly< + | { readonly state: 'classified'; readonly classification: AdReplayTargetClassification } | { readonly state: 'unavailable'; readonly reason: string; readonly hint?: string } >; @@ -120,7 +120,7 @@ export type AdReplayVerificationEntry = Readonly< } >; -/** `classifyTarget`'s result: a verified guard, or the divergence evidence a target-binding failure reports. */ +/** The recorded target's classification: a verified guard, or the divergence evidence a target-binding failure reports. */ export type AdReplayTargetClassification = Readonly< | { readonly verified: true; readonly guard: AdReplayVerifiedTargetGuard } | Readonly<{ @@ -220,26 +220,20 @@ export type AdReplayStepRuntime = Readonly<{ targetRole?: 'source' | 'destination', ): AdReplayVerificationEntry; /** - * Captures a fresh snapshot for classification or for a divergence's - * `screen` — daemon authority (`SessionStore`, the capture pipeline, the - * #1385 launch-race retry). + * Captures a fresh snapshot (with the #1385 launch-race retry) and resolves + * the recorded target against it using the SAME lookup/matching a real + * dispatch would — daemon authority (`SessionStore`, the capture pipeline, + * tree helpers and the selectors package engine). A step whose dispatch + * would wait for its target under a readiness budget re-captures while the + * target does not match yet, under that same budget, so this gate never + * refuses a target the dispatch itself would have waited for. The last + * capture is the divergence's `screen`. */ - captureObservation( - action: SessionAction, - index: number, - options: { retryLaunchRace: boolean }, - ): Promise; - /** - * Resolves the recorded target against `nodes` using the SAME - * lookup/matching a real dispatch would — daemon authority (tree helpers and - * the selectors package engine). - */ - classifyTarget(params: { + observeTarget(params: { action: SessionAction; index: number; token: string; - nodes: readonly SnapshotNode[]; - }): AdReplayTargetClassification; + }): Promise; /** * Dispatches the action, optionally carrying a pre-action identity guard, * and detects the guard-mismatch / wait-landmark-mismatch post-resolution @@ -279,7 +273,7 @@ export type AdReplayStepRuntime = Readonly<{ ): Promise; /** * Builds a target-binding divergence from `evidence`, reusing the LAST - * `captureObservation` result for its `screen` (the pre-dispatch capture + * `observeTarget` capture for its `screen` (the pre-dispatch capture * and classification/capture-failure evidence share one capture) — * daemon authority. `artifactPaths` is the pre-step snapshot, as above; * `scrubVars` as above. diff --git a/packages/ad-replay/src/internal/verify-dispatch.ts b/packages/ad-replay/src/internal/verify-dispatch.ts index 99e6ad7b75..e82303bc5d 100644 --- a/packages/ad-replay/src/internal/verify-dispatch.ts +++ b/packages/ad-replay/src/internal/verify-dispatch.ts @@ -8,10 +8,10 @@ import { import type { AdReplayDispatchGuard, AdReplayDispatchOutcome, - AdReplayObservation, AdReplayScrubValue, AdReplayStepOutcome, AdReplayStepRuntime, + AdReplayTargetObservation, AdReplayVerifiedTargetGuard, } from './runtime-port-types.ts'; @@ -25,8 +25,8 @@ import type { * those four functions live here — engine-private, never re-exported by the * façade — and the daemon side is the narrow `AdReplayStepRuntime` * capabilities this function drives: routing (`beginTargetVerification`), - * capture (`captureObservation`), classification (`classifyTarget`), - * dispatch (`dispatchStep`), and wire-building the resulting divergence + * capture-and-classification (`observeTarget`), dispatch (`dispatchStep`), + * and wire-building the resulting divergence * (`buildRecordedUnverifiableFailure`, `buildTargetBindingFailure`, * `buildPostDispatchTargetBindingFailure`). `./step-loop.ts`'s `runAdReplay` * is this module's one caller. @@ -121,11 +121,8 @@ export async function verifyAndDispatchStep( } const token = preDispatchPlan.token; - // #1385: this is the pre-dispatch gate a step right after `open --relaunch` - // can race — the app may still be launching/mounting when this capture - // lands. Bounded retry rides out that transition (`retryLaunchRace`). - const observation = await runtime.captureObservation(action, index, { retryLaunchRace: true }); - if (observation.state !== 'available') { + const observation = await runtime.observeTarget({ action, index, token }); + if (observation.state !== 'classified') { return { status: 'failed', failure: await runtime.buildTargetBindingFailure( @@ -138,7 +135,7 @@ export async function verifyAndDispatchStep( }; } - const classification = runtime.classifyTarget({ action, index, token, nodes: observation.nodes }); + const { classification } = observation; if (classification.verified) { return dispatchWithGuard(runtime, scrubVars, action, resolvedAction, index, artifactPaths, { kind: 'target', @@ -213,10 +210,12 @@ async function verifyAndDispatchMultiTargetStep( }; } - const observation = await runtime.captureObservation(endpointAction, index, { - retryLaunchRace: true, + const observation = await runtime.observeTarget({ + action: endpointAction, + index, + token: plan.token, }); - if (observation.state !== 'available') { + if (observation.state !== 'classified') { return { status: 'failed', failure: await runtime.buildTargetBindingFailure( @@ -229,12 +228,7 @@ async function verifyAndDispatchMultiTargetStep( }; } - const classification = runtime.classifyTarget({ - action: endpointAction, - index, - token: plan.token, - nodes: observation.nodes, - }); + const { classification } = observation; if (!classification.verified) { return { status: 'failed', @@ -268,7 +262,7 @@ async function verifyAndDispatchMultiTargetStep( } function captureUnavailableEvidence( - observation: Extract, + observation: Extract, ) { return { kind: 'identity-unverifiable' as const, diff --git a/packages/ad-script/src/internal/target-annotation-identity.ts b/packages/ad-script/src/internal/target-annotation-identity.ts index 80ba39047d..9126e32444 100644 --- a/packages/ad-script/src/internal/target-annotation-identity.ts +++ b/packages/ad-script/src/internal/target-annotation-identity.ts @@ -52,7 +52,7 @@ type IdentityTreeNode = Pick; * (`@agent-device/selectors/target-evidence`), replay-time verification * (`packages/replay-port/src/daemon-port/session-replay-target-verification.ts`), and the * dispatch-side post-resolution guard - * (`src/commands/interaction/runtime/resolution.ts`), so all three compute + * (`src/commands/interaction/runtime/replay-target-guard.ts`), so all three compute * a node's identity with byte-identical semantics. */ export function readNodeLocalIdentity( diff --git a/packages/capture-kit/package.json b/packages/capture-kit/package.json index f22fb90cc9..556ef65fb3 100644 --- a/packages/capture-kit/package.json +++ b/packages/capture-kit/package.json @@ -46,6 +46,10 @@ "types": "./src/capture-admission/durable-capture-runtime-recovery.ts", "default": "./src/capture-admission/durable-capture-runtime-recovery.ts" }, + "./observe-until": { + "types": "./src/observe-until.ts", + "default": "./src/observe-until.ts" + }, "./perf-capture-admission-ledger": { "types": "./src/capture-admission/perf-capture-admission-ledger.ts", "default": "./src/capture-admission/perf-capture-admission-ledger.ts" diff --git a/packages/capture-kit/src/observe-until.test.ts b/packages/capture-kit/src/observe-until.test.ts new file mode 100644 index 0000000000..3a1426e622 --- /dev/null +++ b/packages/capture-kit/src/observe-until.test.ts @@ -0,0 +1,338 @@ +import assert from 'node:assert/strict'; +import { describe, test } from 'vitest'; +import { AppError } from '@agent-device/kernel/errors'; +import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; +import { observeUntil, type ObservationClock } from './observe-until.ts'; + +/** A clock that advances only when the loop sleeps or a capture declares its own cost. */ +function fakeClock(): ObservationClock & { advance(ms: number): void; slept: number[] } { + let nowMs = 1_000; + const slept: number[] = []; + return { + slept, + now: () => nowMs, + advance: (ms) => { + nowMs += ms; + }, + sleep: async (ms) => { + slept.push(ms); + nowMs += ms; + }, + }; +} + +/** No per-capture deadline: the engine default. */ +const UNBOUNDED_SCHEDULE = { intervalMs: 200, budgetMs: 1_000 } as const; +const SCHEDULE = { ...UNBOUNDED_SCHEDULE, captureDeadline: 'cancel' } as const; + +/** The Android helper's content verdict: the capture ran but held no readable app content. */ +function unreadableContent(): AppError { + return new AppError('COMMAND_FAILED', 'no readable content', { + androidSnapshotHelperFailureReason: 'system-window-only', + }); +} + +describe('observeUntil', () => { + test('answers done on the poll whose verdict accepts, with the timeline', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + clock.advance(50); + return captures; + }, + verdict: (latest) => + latest >= 3 ? { kind: 'done', result: `seen ${latest}` } : { kind: 'continue' }, + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.equal(observed.kind === 'done' && observed.result, 'seen 3'); + assert.equal(observed.polls.length, 3); + assert.deepEqual(clock.slept, [200, 200]); + assert.equal(observed.waitedMs, 550); + }); + + test('expires when the budget runs out while the verdict still continues, keeping the last value', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: () => ({ kind: 'continue' }), + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.kind === 'expired' && observed.last, 6); + assert.equal(observed.polls.length, 6); + assert.equal(observed.waitedMs, 1_000); + }); + + test('rides out only the errors the caller classifies, and keeps polling', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + if (captures === 1) throw unreadableContent(); + return captures; + }, + verdict: (latest) => ({ kind: 'done', result: latest }), + schedule: SCHEDULE, + rideOut: isUnreadableCaptureContentError, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['rode-out', 'observed'], + ); + }); + + test('fails on the first error the caller does not ride out', async () => { + const clock = fakeClock(); + const failure = new AppError('COMMAND_FAILED', 'wedged'); + const observed = await observeUntil({ + capture: async () => { + throw failure; + }, + verdict: () => ({ kind: 'continue' }), + schedule: SCHEDULE, + rideOut: isUnreadableCaptureContentError, + clock, + }); + assert.equal(observed.kind, 'failed'); + assert.equal(observed.kind === 'failed' && observed.error, failure); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['failed'], + ); + }); + + test('reports the last ridden-out error when every poll was ridden out', async () => { + const clock = fakeClock(); + const failures: AppError[] = []; + const observed = await observeUntil({ + capture: async () => { + const failure = unreadableContent(); + failures.push(failure); + throw failure; + }, + verdict: (latest) => ({ kind: 'done', result: latest }), + schedule: SCHEDULE, + rideOut: isUnreadableCaptureContentError, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.kind === 'expired' && observed.last, undefined); + assert.equal(observed.kind === 'expired' && observed.lastError, failures.at(-1)); + assert.ok(observed.polls.every((poll) => poll.outcome === 'rode-out')); + }); + + test('cancels and joins a capture still in flight at the deadline', async () => { + let aborted = false; + const observed = await observeUntil({ + capture: (signal) => + new Promise((resolve) => { + signal.addEventListener('abort', () => { + aborted = true; + setTimeout(() => resolve(1), 5); + }); + }), + verdict: () => ({ kind: 'done', result: true }), + schedule: { intervalMs: 5, budgetMs: 20, captureDeadline: 'cancel' }, + }); + assert.equal(observed.kind, 'stalled'); + assert.equal(aborted, true); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['stalled'], + ); + }); + + test('always completes minPolls even past the budget, so a quiet pair can form', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + clock.advance(1_000); + return captures; + }, + verdict: (latest, previous) => + previous !== undefined + ? { kind: 'done', result: [previous, latest] } + : { kind: 'continue' }, + schedule: { ...UNBOUNDED_SCHEDULE, minPolls: 2 }, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.deepEqual(observed.kind === 'done' && observed.result, [1, 2]); + }); + + test('a verdict can raise the budget but never lower it', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: (latest) => + latest === 1 ? { kind: 'continue', budgetMs: 2_000 } : { kind: 'continue', budgetMs: 100 }, + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.waitedMs, 2_000); + }); + + test('judges an initial value before spending a poll', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: (latest) => ({ kind: 'done', result: latest }), + schedule: SCHEDULE, + initial: 42, + clock, + }); + assert.equal(observed.kind === 'done' && observed.value, 42); + assert.equal(captures, 0); + assert.equal(observed.polls.length, 0); + }); +}); + +describe('observeUntil cancellation', () => { + test('a signal aborted during the sleep ends the loop canceled before another capture', async () => { + const clock = fakeClock(); + const controller = new AbortController(); + let captures = 0; + await assert.rejects( + observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: () => ({ kind: 'continue' }), + schedule: UNBOUNDED_SCHEDULE, + signal: controller.signal, + clock: { + now: clock.now, + sleep: async (ms) => { + await clock.sleep(ms); + controller.abort(); + }, + }, + }), + (error) => error instanceof AppError && error.details?.reason === 'request_canceled', + ); + assert.equal(captures, 1); + }); +}); + +describe('observeUntil budgetFrom first-capture', () => { + test('never bounds the first capture and spends the budget only on retries', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + if (captures === 1) clock.advance(5_000); + return captures; + }, + verdict: (latest) => (latest === 3 ? { kind: 'done', result: latest } : { kind: 'continue' }), + schedule: { + intervalMs: 200, + budgetMs: 1_000, + budgetFrom: 'first-capture', + captureDeadline: 'cancel', + }, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.equal(observed.polls.length, 3); + assert.equal(observed.polls[0]?.durationMs, 5_000); + assert.equal(observed.waitedMs, 5_400); + }); + + test('sleeps the interval between an initial observation and the first poll', async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: async () => 2, + verdict: (latest, previous) => + previous === undefined + ? { kind: 'continue' } + : { kind: 'done', result: [previous, latest] }, + schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2, captureDeadline: 'cancel' }, + initial: 1, + clock, + }); + assert.deepEqual(observed.kind === 'done' && observed.result, [1, 2]); + assert.deepEqual(clock.slept, [200]); + assert.equal(observed.polls.length, 1); + }); +}); + +describe('observeUntil captureDeadline', () => { + /** A second capture that returns only after the remaining budget is spent, ignoring its signal. */ + function lateSecondCapture(clock: ReturnType) { + let captures = 0; + return async () => { + captures += 1; + clock.advance(captures === 1 ? 100 : 1_500); + return captures; + }; + } + + test('by default judges a capture that finishes past the budget and can end done', async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: lateSecondCapture(clock), + verdict: (latest) => (latest === 2 ? { kind: 'done', result: latest } : { kind: 'continue' }), + schedule: UNBOUNDED_SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['observed', 'observed'], + ); + }); + + test("by default ends expired, never stalled, when the late capture's verdict continues", async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: lateSecondCapture(clock), + verdict: () => ({ kind: 'continue' }), + schedule: UNBOUNDED_SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.kind === 'expired' && observed.last, 2); + assert.equal(observed.polls.length, 2); + }); + + test("'cancel' ends stalled on a capture that finishes past its deadline", async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: lateSecondCapture(clock), + verdict: (latest) => (latest === 2 ? { kind: 'done', result: latest } : { kind: 'continue' }), + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'stalled'); + assert.equal(observed.kind === 'stalled' && observed.last, 1); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['observed', 'stalled'], + ); + }); +}); diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts new file mode 100644 index 0000000000..89d7bdfbc9 --- /dev/null +++ b/packages/capture-kit/src/observe-until.ts @@ -0,0 +1,321 @@ +import { createRequestCanceledError } from '@agent-device/kernel/errors'; +import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; + +/** + * The one observation loop: capture until a verdict accepts, on a schedule, under a budget. Owns + * cadence, the deadline, abort-and-join of an in-flight capture under `captureDeadline: 'cancel'`, + * which capture errors are ridden out, and the per-poll timeline. Owns nothing about WHAT is + * observed: the verdict is the caller's, and so is the equality rule behind it (digest, signature, + * pixel). + */ + +/** + * How an observation loop ended. + * + * - `done` — the verdict accepted a capture. + * - `expired` — the budget ran out while the verdict still said continue. + * - `stalled` — a capture was still in flight at its deadline; it was cancelled and joined, so no + * late result can mutate state after the loop returned. + * - `failed` — a capture error the loop does not ride out. + */ +export type ObservationEnd = 'done' | 'expired' | 'stalled' | 'failed'; + +export type ObservationSchedule = Readonly<{ + /** Delay between polls, clamped to the remaining budget. */ + intervalMs: number; + /** + * Wall-clock budget from the first capture. It bounds when a poll may start; under + * `captureDeadline: 'cancel'` a capture that starts inside it may overrun it by at most one + * interval. The sample after the last sleep is always taken, so a verdict that reads elapsed time + * against a cap sees a poll at or past it. + */ + budgetMs: number; + /** Observations (an `initial` counts) the loop always completes before the budget may end it. */ + minPolls?: number; + /** + * `'start'` (default) bounds every capture, the first included. `'first-capture'` leaves the first + * capture unbounded and spends the budget only on retries, so a caller whose first attempt is a + * one-shot pays nothing on its success path. + */ + budgetFrom?: 'start' | 'first-capture'; + /** + * Whether a capture is bounded by its own deadline. `'cancel'` arms each capture with the + * remaining budget (at least one interval) as an abort signal, and a capture that ends at or past + * that deadline ends the loop `stalled`: declare it only when the capture honors the signal. + * `'none'` hands the capture no deadline; a capture that finishes past the budget is still judged, + * so the loop ends `done` on an accepting verdict and `expired` otherwise. Default `'none'`, so a + * capture that ignores the signal cannot end the loop `stalled` by omission. + */ + captureDeadline?: 'cancel' | 'none'; +}>; + +export type ObservationClock = Readonly<{ + now(): number; + sleep(ms: number): Promise; +}>; + +export type ObservationVerdict = + | Readonly<{ kind: 'done'; result: R }> + /** Keep polling; `budgetMs` raises the budget measured from the first capture (never lowers it). */ + | Readonly<{ kind: 'continue'; budgetMs?: number }>; + +/** `failed` is a capture error the loop does not ride out; it always ends the loop. */ +export type ObservationPollOutcome = 'observed' | 'rode-out' | 'stalled' | 'failed'; + +export type ObservationPoll = Readonly<{ + startedMs: number; + durationMs: number; + outcome: ObservationPollOutcome; +}>; + +export type ObservationEvidence = Readonly<{ + polls: readonly ObservationPoll[]; + waitedMs: number; +}>; + +type ObservedEnd = + | Readonly<{ kind: 'done'; result: R; value: T }> + | Readonly<{ kind: 'expired'; last: T | undefined; lastError: unknown }> + | Readonly<{ kind: 'stalled'; last: T | undefined; lastError: unknown; error: unknown }> + | Readonly<{ kind: 'failed'; last: T | undefined; error: unknown }>; + +export type Observed = ObservationEvidence & ObservedEnd; + +export type ObserveUntilParams = Readonly<{ + capture: (signal: AbortSignal) => Promise; + /** Judges the latest capture; `previous` is the last observed value, for quiet-pair predicates. */ + verdict: (latest: T, previous: T | undefined, polls: number) => ObservationVerdict; + schedule: ObservationSchedule; + /** + * True keeps polling past this capture error. Default: no error is ridden out. The last + * ridden-out error is reported as `lastError` when the loop expires or stalls. + */ + rideOut?: (error: unknown) => boolean; + signal?: AbortSignal; + clock?: ObservationClock; + /** A capture the caller already holds; judged first, without a poll. */ + initial?: T; + /** Diagnostic phase for the end-of-loop line; nothing is logged without it. */ + phase?: string; +}>; + +const REAL_CLOCK: ObservationClock = { + now: () => Date.now(), + sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)), +}; + +type LoopState = { + readonly startedMs: number; + budgetStartedMs: number; + budgetMs: number; + observations: number; + readonly polls: ObservationPoll[]; + previous: T | undefined; + last: T | undefined; + lastError: unknown; +}; + +type Loop = Readonly<{ + params: ObserveUntilParams; + clock: ObservationClock; + state: LoopState; +}>; + +export async function observeUntil( + params: ObserveUntilParams, +): Promise> { + const clock = params.clock ?? REAL_CLOCK; + const startedMs = clock.now(); + const loop: Loop = { + params, + clock, + state: { + startedMs, + budgetStartedMs: startedMs, + budgetMs: params.schedule.budgetMs, + observations: 0, + polls: [], + previous: undefined, + last: undefined, + lastError: undefined, + }, + }; + if (params.initial !== undefined) { + loop.state.observations += 1; + const done = judge(loop, params.initial); + if (done) return done; + } + while (true) { + if (params.signal?.aborted) throw createRequestCanceledError(); + const expired = await waitBeforePoll(loop); + if (expired) return expired; + if (params.signal?.aborted) throw createRequestCanceledError(); + const ended = await pollOnce(loop); + if (ended) return ended; + } +} + +function remainingMs({ state, clock }: Loop): number { + return state.budgetStartedMs + state.budgetMs - clock.now(); +} + +function mustPoll({ params, state }: Loop): boolean { + return state.observations < (params.schedule.minPolls ?? 1); +} + +/** Sleeps one interval between observations; ends the loop when the budget is spent first. */ +async function waitBeforePoll(loop: Loop): Promise | undefined> { + const { state, clock, params } = loop; + if (state.observations === 0) return undefined; + const forced = mustPoll(loop); + if (!forced && remainingMs(loop) <= 0) + return finish(loop, { kind: 'expired', last: state.last, lastError: state.lastError }); + const { intervalMs } = params.schedule; + await clock.sleep(forced ? intervalMs : Math.max(0, Math.min(intervalMs, remainingMs(loop)))); + return undefined; +} + +async function pollOnce(loop: Loop): Promise | undefined> { + const { params, clock, state } = loop; + const unbounded = params.schedule.budgetFrom === 'first-capture' && state.polls.length === 0; + const pollStartedMs = clock.now(); + const bounded = !unbounded && params.schedule.captureDeadline === 'cancel'; + const poll = await captureWithin( + bounded ? Math.max(remainingMs(loop), params.schedule.intervalMs) : undefined, + params.signal, + params.capture, + clock, + ); + state.observations += 1; + if (unbounded) state.budgetStartedMs = clock.now(); + const outcome = pollOutcome(loop, poll); + state.polls.push({ + startedMs: pollStartedMs - state.startedMs, + durationMs: clock.now() - pollStartedMs, + outcome, + }); + if (poll.kind === 'observed') return judge(loop, poll.value); + if (outcome === 'rode-out') { + state.lastError = poll.error; + return undefined; + } + if (poll.kind === 'stalled') { + return finish(loop, { + kind: 'stalled', + last: state.last, + lastError: state.lastError, + error: poll.error, + }); + } + return finish(loop, { kind: 'failed', last: state.last, error: poll.error }); +} + +function pollOutcome( + loop: Loop, + poll: CaptureWithinOutcome, +): ObservationPollOutcome { + if (poll.kind === 'stalled') return 'stalled'; + if (poll.kind === 'failed') return loop.params.rideOut?.(poll.error) ? 'rode-out' : 'failed'; + return 'observed'; +} + +function judge(loop: Loop, value: T): Observed | undefined { + const { params, state } = loop; + const verdict = params.verdict(value, state.previous, state.polls.length); + state.previous = value; + state.last = value; + if (verdict.kind === 'done') return finish(loop, { kind: 'done', result: verdict.result, value }); + if (verdict.budgetMs !== undefined) state.budgetMs = Math.max(state.budgetMs, verdict.budgetMs); + return undefined; +} + +function finish(loop: Loop, end: ObservedEnd): Observed { + const { params, state, clock } = loop; + const observed: Observed = { + ...end, + polls: state.polls, + waitedMs: clock.now() - state.startedMs, + }; + if (params.phase) { + emitDiagnostic({ + level: 'debug', + phase: params.phase, + data: { + end: observed.kind satisfies ObservationEnd, + polls: state.polls.length, + waitedMs: observed.waitedMs, + }, + }); + } + return observed; +} + +type CaptureWithinOutcome = + | Readonly<{ kind: 'observed'; value: T }> + | Readonly<{ kind: 'failed'; error: unknown }> + | Readonly<{ kind: 'stalled'; error: unknown }>; + +/** + * Runs one capture under the remaining budget by cancellation, then waits for it to quiesce. + * Deliberately not race-and-abandon: a late capture could otherwise mutate session state or keep a + * platform helper after the loop has returned. + */ +async function captureWithin( + remainingMs: number | undefined, + parent: AbortSignal | undefined, + capture: (signal: AbortSignal) => Promise, + clock: ObservationClock, +): Promise> { + const deadline = armDeadline(remainingMs, clock); + const signal = parent ? AbortSignal.any([parent, deadline.signal]) : deadline.signal; + try { + const value = await capture(signal); + return endOfCapture(parent, deadline.expired(), undefined) ?? { kind: 'observed', value }; + } catch (error) { + return endOfCapture(parent, deadline.expired(), error) ?? { kind: 'failed', error }; + } finally { + deadline.dispose(); + } +} + +/** A capture that ended after the deadline stalled; one ended by the caller's signal was canceled. */ +function endOfCapture( + parent: AbortSignal | undefined, + deadlineExpired: boolean, + error: unknown, +): CaptureWithinOutcome | undefined { + if (parent?.aborted) throw createRequestCanceledError(); + return deadlineExpired ? { kind: 'stalled', error } : undefined; +} + +/** The deadline has passed once its timer fired or the loop's clock reached it, whichever is first. */ +function armDeadline( + remainingMs: number | undefined, + clock: ObservationClock, +): { + signal: AbortSignal; + expired: () => boolean; + dispose: () => void; +} { + const controller = new AbortController(); + const deadlineAtMs = remainingMs === undefined ? undefined : clock.now() + remainingMs; + let fired = false; + const timer = + remainingMs === undefined + ? undefined + : setTimeout( + () => { + fired = true; + controller.abort(new DOMException('Observation deadline exceeded', 'TimeoutError')); + }, + Math.max(0, remainingMs), + ); + timer?.unref(); + return { + signal: controller.signal, + expired: () => fired || (deadlineAtMs !== undefined && clock.now() >= deadlineAtMs), + dispose: () => { + if (timer !== undefined) clearTimeout(timer); + }, + }; +} diff --git a/packages/command-registry/src/registry.ts b/packages/command-registry/src/registry.ts index 3dfa4738fe..7957180c08 100644 --- a/packages/command-registry/src/registry.ts +++ b/packages/command-registry/src/registry.ts @@ -1301,6 +1301,7 @@ export const RAW_COMMAND_DESCRIPTORS = [ }, timeoutPolicy: postActionObservationTimeoutPolicy('click', PRESERVE_DAEMON_TIMEOUT_POLICY), postActionObservation: postActionObservation('click'), + targetReadiness: 'budgeted', responseDataTransform: TOUCH_INTERACTION_RESPONSE_DATA_TRANSFORM, batchable: true, platformExecution: { kind: 'device-runtime', uses: clickRuntimeUses }, @@ -1330,6 +1331,7 @@ export const RAW_COMMAND_DESCRIPTORS = [ envelopeMs: 210_000, }, postActionObservation: postActionObservation('longpress'), + targetReadiness: 'budgeted', batchable: true, platformExecution: { kind: 'device-runtime', uses: longPressRuntimeUses }, }, @@ -1358,6 +1360,7 @@ export const RAW_COMMAND_DESCRIPTORS = [ frameworkTier: 'core', timeoutPolicy: postActionObservationTimeoutPolicy('press', PRESERVE_DAEMON_TIMEOUT_POLICY), postActionObservation: postActionObservation('press'), + targetReadiness: 'budgeted', responseDataTransform: TOUCH_INTERACTION_RESPONSE_DATA_TRANSFORM, batchable: true, platformExecution: { kind: 'device-runtime', uses: pressRuntimeUses }, @@ -1972,7 +1975,12 @@ function readCatalogKey(descriptor: { } const TIMEOUT_POLICY_BY_COMMAND: ReadonlyMap = new Map( - commandDescriptors.map((descriptor) => [descriptor.name, descriptor.timeoutPolicy]), + Array.from(COMMAND_DESCRIPTOR_BY_NAME.values(), (descriptor) => [ + descriptor.name, + descriptor.targetReadiness + ? { ...descriptor.timeoutPolicy, targetReadiness: descriptor.targetReadiness } + : descriptor.timeoutPolicy, + ]), ); const DEVICE_CLAIM_POLICY_BY_COMMAND: ReadonlyMap = new Map( @@ -1999,6 +2007,11 @@ export function commandSupportsSettleObservation(command: string | undefined): b return resolveCommandPostActionObservationSupport(command) !== undefined; } +export function commandAcceptsReadinessBudget(command: string | undefined): boolean { + if (command === undefined) return false; + return COMMAND_DESCRIPTOR_BY_NAME.get(command)?.targetReadiness === 'budgeted'; +} + export function commandSupportsVerifyEvidence(command: string | undefined): boolean { return resolveCommandPostActionObservationSupport(command) === 'settle-and-verify'; } diff --git a/packages/command-registry/src/timeout-policy.ts b/packages/command-registry/src/timeout-policy.ts index 30e6f58f61..587c59315c 100644 --- a/packages/command-registry/src/timeout-policy.ts +++ b/packages/command-registry/src/timeout-policy.ts @@ -64,7 +64,12 @@ type BoundedTimeoutPolicy = CommandTimeoutPolicy & { envelopeMs: number }; type FlagTimeoutBudget = Extract; type RequestTimeoutInput = Readonly<{ positionals?: string[]; - flags?: Readonly<{ timeoutMs?: number; settle?: boolean; waitMs?: number }>; + flags?: Readonly<{ + timeoutMs?: number; + settle?: boolean; + waitMs?: number; + readinessTimeoutMs?: number; + }>; }>; /** Resolves the request envelope from its declared policy and user-supplied budget. */ @@ -74,11 +79,32 @@ export function resolveCommandRequestTimeoutMs( ): number | undefined { if (policy.envelopeMs === 'unbounded') return undefined; const boundedPolicy: BoundedTimeoutPolicy = { ...policy, envelopeMs: policy.envelopeMs }; - return ( + const envelopeMs = resolvePositionalBudgetTimeoutMs(boundedPolicy, input.positionals ?? []) ?? resolveFlagBudgetTimeoutMs(boundedPolicy, input.flags) ?? - boundedPolicy.envelopeMs - ); + boundedPolicy.envelopeMs; + return envelopeMs + readinessBudgetMs(policy, input.flags); +} + +/** + * The ceiling of a `targetReadiness: 'budgeted'` command's readiness budget: the promotedTarget + * selector row's poll ceiling, pinned to it by test. + */ +export const READINESS_BUDGET_MAX_MS = 2_000; + +/** + * A readiness budget is target-poll time spent before the action itself, so it extends whatever + * envelope the command otherwise has; the daemon's own poll must never outlive the client's clock. + * The poll never runs past the promotedTarget row's ceiling, so neither does the widening. + */ +function readinessBudgetMs( + policy: CommandTimeoutPolicy, + flags: RequestTimeoutInput['flags'], +): number { + if (policy.targetReadiness !== 'budgeted') return 0; + const budgetMs = flags?.readinessTimeoutMs; + if (typeof budgetMs !== 'number' || !Number.isInteger(budgetMs) || budgetMs <= 0) return 0; + return Math.min(budgetMs, READINESS_BUDGET_MAX_MS); } function resolvePositionalBudgetTimeoutMs( diff --git a/packages/command-registry/src/types.ts b/packages/command-registry/src/types.ts index 934bd0bc38..16911f71aa 100644 --- a/packages/command-registry/src/types.ts +++ b/packages/command-registry/src/types.ts @@ -75,8 +75,21 @@ export type CommandTimeoutPolicy = { budget: CommandTimeoutBudget; envelopeMs: number | 'unbounded'; onTimeout: 'preserve-daemon' | 'reset-daemon'; + /** + * The descriptor's {@link CommandTargetReadiness} trait, attached by `resolveCommandTimeoutPolicy` + * so the envelope can widen by the readiness budget. Descriptors declare it as `targetReadiness`, + * never on their `timeoutPolicy`. + */ + targetReadiness?: CommandTargetReadiness; }; +/** + * `budgeted`: the command's target resolution may poll for a target that does not exist yet, under + * a caller-supplied `readinessTimeoutMs`, capped at `READINESS_BUDGET_MAX_MS` (`timeout-policy.ts`). + * Must match the commands the `targetReadiness` interaction guarantee cells mark `runtime`. + */ +export type CommandTargetReadiness = 'budgeted'; + /** * #1320 "Command descriptor policy": what a command may do with the host-global * device claim store. REQUIRED on every descriptor (no default), and read by the @@ -199,6 +212,9 @@ export type TargetIdentityVerification = 'pre-dispatch' | 'post-resolution'; * commands that support `--settle`/`--verify`; consumed by * command surfaces and timeout policy instead of repeated * command-name lists. + * - `targetReadiness` — optional; the commands whose target resolution accepts a + * `readinessTimeoutMs` budget. Read by replay (which supplies a + * default budget) and by the request envelope (which widens by it). * - `responseDataTransform` — optional public response data shaping rules for * command-owned fields in daemon responses. This keeps * response shaping on the same descriptor surface as other @@ -221,6 +237,7 @@ type CommandDescriptorBase = { */ deviceClaimPolicy: DeviceClaimPolicy; postActionObservation?: PostActionObservationSupport; + targetReadiness?: CommandTargetReadiness; responseDataTransform?: CommandResponseDataTransform; catalog: CommandCatalogFacet; /** Required iff `catalog.group === 'public'`; see {@link CommandFrameworkTier}. */ diff --git a/packages/contracts/src/client-gesture.ts b/packages/contracts/src/client-gesture.ts index 5928b727a9..efa8556e64 100644 --- a/packages/contracts/src/client-gesture.ts +++ b/packages/contracts/src/client-gesture.ts @@ -35,11 +35,20 @@ export type SettleCommandOptions = { timeoutMs?: number; }; +/** + * How long a tap-shaped interaction may poll for a target that does not exist yet, capped at the + * promotedTarget row's maxTimeoutMs. Never model- or CLI-writable; omitted means one attempt. + */ +export type ReadinessBudgetOptions = { + readinessTimeoutMs?: number; +}; + export type ClickOptions = DeviceCommandBaseOptions & SelectorSnapshotCommandOptions & InteractionTarget & RepeatedPressOptions & - SettleCommandOptions & { + SettleCommandOptions & + ReadinessBudgetOptions & { button?: ClickButton; /** * Opt-in (#1047): return cheap post-action evidence (AX digest, node counts, @@ -53,14 +62,16 @@ export type PressOptions = DeviceCommandBaseOptions & SelectorSnapshotCommandOptions & InteractionTarget & RepeatedPressOptions & - SettleCommandOptions & { + SettleCommandOptions & + ReadinessBudgetOptions & { verify?: boolean; }; export type LongPressOptions = DeviceCommandBaseOptions & SelectorSnapshotCommandOptions & InteractionTarget & - SettleCommandOptions & { + SettleCommandOptions & + ReadinessBudgetOptions & { durationMs?: number; }; diff --git a/packages/contracts/src/command-flags.ts b/packages/contracts/src/command-flags.ts index f7777c1bd5..a5ab9ab429 100644 --- a/packages/contracts/src/command-flags.ts +++ b/packages/contracts/src/command-flags.ts @@ -35,6 +35,11 @@ export type CommandFlags = Omit & { kind?: string; maestro?: MaestroRuntimeFlags; postGestureStabilization?: boolean; + /** + * Readiness budget for press/click/longpress, capped at the promotedTarget row's maxTimeoutMs. + * No CliFlags counterpart: never CLI- or model-writable. + */ + readinessTimeoutMs?: number; snapshotIncludeHiddenContentHints?: boolean; leaseProvider?: string; provider?: string; diff --git a/packages/contracts/src/facades/client.ts b/packages/contracts/src/facades/client.ts index 8cd5d9dad2..f7ec417f78 100644 --- a/packages/contracts/src/facades/client.ts +++ b/packages/contracts/src/facades/client.ts @@ -55,6 +55,7 @@ export type { PanOptions, PinchOptions, PressOptions, + ReadinessBudgetOptions, RepeatedPressOptions, RotateGestureOptions, ScrollOptions, diff --git a/packages/contracts/src/interaction-guarantees.ts b/packages/contracts/src/interaction-guarantees.ts index b18f46be48..57b56a54a9 100644 --- a/packages/contracts/src/interaction-guarantees.ts +++ b/packages/contracts/src/interaction-guarantees.ts @@ -73,6 +73,11 @@ export const INTERACTION_GUARANTEES = [ // unique/disambiguated/exact/label-fallback/not-observed provenance, // pre-action diagnostics only — never ref-issued or MCP-pinned. 'resolutionDisclosure', + // How the path waits for the target to exist and become actionable before acting: a declared + // budget under which a plain no-match keeps retrying, versus a target that must already exist at + // dispatch time. Distinct from occlusion/offscreen/nonHittable, which judge a target already + // found; this is about whether one is found at all. + 'targetReadiness', ] as const; export type InteractionGuarantee = (typeof INTERACTION_GUARANTEES)[number]; @@ -149,6 +154,17 @@ const SHARED_RESPONSE_CONSTRUCTION: GuaranteeEnforcement = { via: 'src/daemon/interaction/internal/interaction-touch-response.ts#buildInteractionResponseData', }; +// Both Maestro-compatible fast paths (src/daemon/interaction/internal/interaction-touch-direct-ios.ts) +// dispatch ONE fused XCTest runner request (`command: 'tap'` with a selectorKey/selectorValue) built +// in packages/platform-apple/src/interactions.ts#tapElementSelector — not the separate +// packages/maestro/ flow-script engine, which drives standalone `.yaml` Maestro flows and never +// participates in an ordinary daemon click. +const DIRECT_IOS_SINGLE_QUERY_READINESS: GuaranteeEnforcement = { + kind: 'waived', + reason: + "Intentional: the fused runner request performs a single XCTest selector/frame query per dispatch and does not itself retry for an as-yet-nonexistent target; a miss is a runner failure (see errorTaxonomy) rather than a wait, and delegation-on-error (ADR 0011) is a different path's guarantee, not a retry of this one.", +}; + // The two runtime tree paths (selector and ref resolution) run the SAME shared // guard/observation implementations; only how the target is found // (disambiguation) and how failures are described (errorTaxonomy) differ. @@ -194,7 +210,7 @@ const RUNTIME_TREE_SHARED_GUARANTEES = { // them is undetected. offscreen: { kind: 'runtime', - via: 'src/commands/interaction/runtime/resolution.ts#throwIfOffscreenInteractionTarget', + via: 'src/commands/interaction/runtime/target-visibility-stages.ts#throwIfOffscreenInteractionTarget', }, // Promotion runs only for rows that declare it (#1656); the retarget itself // is still resolveActionableTouchResolution. @@ -238,6 +254,14 @@ export const INTERACTION_DISPATCH_PATHS: Record & durationMs?: number; holdMs?: number; jitterPx?: number; + /** + * The readiness budget a tap-shaped interaction (press/click/longpress) may spend polling for a + * target that does not exist yet, capped at the promotedTarget row's maxTimeoutMs. Never model- + * or CLI-writable; absent means one attempt. + */ + readinessTimeoutMs?: number; pixels?: number; /** Scroll: repeat passes until this selector is visible on screen. */ until?: string; diff --git a/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts b/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts index 52651408fb..c6834d9bcb 100644 --- a/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts +++ b/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts @@ -46,3 +46,27 @@ test('native replay failure metadata keeps machine fields and daemon-owned paths expect(response.error.details).toHaveProperty('reason'); expect(response.error.details).not.toHaveProperty('reas'); }); + +test('a replay divergence carries the readiness evidence of an exhausted target wait', () => { + const readiness = { polls: 11, waitedMs: 2_004, end: 'expired' }; + const response = buildReplayDivergenceFailureResponseFromDescriptor({ + error: { + code: 'COMMAND_FAILED', + message: 'Selector did not match', + details: { reason: 'selector_not_found', readiness }, + }, + actionLabel: 'press label=Continue', + action: 'press', + positionals: ['label=Continue'], + step: 3, + replayPath: '/tmp/flows/checkout.ad', + artifactPaths: [], + divergence: {}, + scrubVars: [], + }); + + expect(response.ok).toBe(false); + if (response.ok) return; + expect(response.error.code).toBe('REPLAY_DIVERGENCE'); + expect(response.error.details).toMatchObject({ reason: 'selector_not_found', readiness }); +}); diff --git a/packages/replay-port/src/daemon-port/command-types.ts b/packages/replay-port/src/daemon-port/command-types.ts index 5b765daa4f..e34948b0f2 100644 --- a/packages/replay-port/src/daemon-port/command-types.ts +++ b/packages/replay-port/src/daemon-port/command-types.ts @@ -10,6 +10,7 @@ import type { ReplayTestAttemptStepSink } from '@agent-device/replay-test'; import type { DaemonResponse, SessionRuntimeHints } from '@agent-device/kernel/contracts'; import type { DeviceInfo } from '@agent-device/kernel/device'; import type { SnapshotState } from '@agent-device/kernel/snapshot'; +import type { ObservationClock } from '@agent-device/capture-kit/observe-until'; /** * The slice of the daemon's live session record replay reads. The daemon passes its full @@ -127,6 +128,11 @@ export type ReplayInvoke = (request: ReplayDispatchRequest) => Promise SessionScope; + /** + * The time source the pre-dispatch target-readiness wait paces itself by. Absent: wall clock. + * The daemon never sets it; tests inject one, as `AgentDeviceRuntime.clock` does for the dispatch. + */ + clock?: ObservationClock; }>; export type ReplayCommand = Readonly<{ diff --git a/packages/replay-port/src/daemon-port/native-command.ts b/packages/replay-port/src/daemon-port/native-command.ts index 50ba57430a..80b28f2ab3 100644 --- a/packages/replay-port/src/daemon-port/native-command.ts +++ b/packages/replay-port/src/daemon-port/native-command.ts @@ -1,4 +1,4 @@ -import { asAppError } from '@agent-device/kernel/errors'; +import { normalizeError } from '@agent-device/kernel/errors'; import { runAdReplay } from '@agent-device/ad-replay'; import type { SnapshotTimingSample } from '@agent-device/contracts/capture'; import { summarizeSnapshotTimingSamples } from '@agent-device/contracts/capture'; @@ -183,12 +183,15 @@ export async function runReplayCommand(command: ReplayCommand): Promise 0 ? { artifactPaths: [...artifactPaths] } : undefined, - ); + const normalized = normalizeError(error); + if (artifactPaths.size === 0) return { ok: false, error: normalized }; + return { + ok: false, + error: { + ...normalized, + details: { ...normalized.details, artifactPaths: [...artifactPaths] }, + }, + }; } } diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index 3221deb61c..e826dc7cc5 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -6,6 +6,8 @@ import type { ReplayInvoke, } from '@agent-device/replay-port/command-types'; import { mergeParentFlags } from '@agent-device/command-registry/batch'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import type { ReadinessSchedule } from '@agent-device/selectors/selector-pipeline-policy'; import { AppError, normalizeError } from '@agent-device/kernel/errors'; import { gesturePayloadFromPositionals, @@ -44,6 +46,11 @@ export async function invokeReplayAction(params: { /** The isolation scope the daemon already resolved for the request, when it did. */ resolvedSessionScope: SessionScope | undefined; dependencies: ReplayDaemonDependencies; + /** + * What is left of the step's readiness budget after the pre-dispatch target gate waited; 0 makes + * the dispatch resolve its target once. Absent: the step's full budget. + */ + readinessTimeoutMs?: number; }): Promise { const { req, @@ -85,6 +92,7 @@ export async function invokeReplayAction(params: { invoke, resolvedSessionScope, dependencies, + readinessTimeoutMs: params.readinessTimeoutMs, }); } catch (error) { // Only an expected AppError dispatch failure (e.g. a selector-miss) gets @@ -142,10 +150,12 @@ async function invokeResolvedReplayAction(params: { invoke: ReplayInvoke; resolvedSessionScope: SessionScope | undefined; dependencies: ReplayDaemonDependencies; + readinessTimeoutMs: number | undefined; }): Promise { const { req, sessionName, resolved, sourceAction, invoke, resolvedSessionScope, dependencies } = params; - const flags = buildReplayActionFlags(req.flags, resolved.flags); + const flags = buildReplayActionFlags(req.flags, resolved.flags, resolved.command); + if (params.readinessTimeoutMs !== undefined) flags.readinessTimeoutMs = params.readinessTimeoutMs; const recordedInputVariable = sourceAction.command === 'fill' ? readRecordedInputVariableName(inferFillText(sourceAction)) @@ -215,9 +225,37 @@ function readResponseTiming(data: unknown): Record | undefined ); } +/** + * A replayed step of a readiness-budgeted command carries no budget of its own; replay supplies one + * so a step recorded against a loading screen can land. + */ +const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; + +/** + * The readiness schedule a replayed step's dispatch polls its target under, or undefined when the + * step's command does not wait for its target. The pre-dispatch target gate polls under this same + * schedule, built by the selector policy's `readinessScheduleFor` as the dispatch builds its own. + */ +export async function replayStepReadinessSchedule( + parentFlags: CommandFlags | undefined, + action: SessionAction, +): Promise { + const { readinessTimeoutMs } = buildReplayActionFlags(parentFlags, action.flags, action.command); + if (readinessTimeoutMs === undefined) return undefined; + // Loaded only by a step that waits, so the replay entry does not evaluate the selector policy. + const { SELECTOR_PIPELINE_POLICIES, readinessScheduleFor } = + await import('@agent-device/selectors/selector-pipeline-policy'); + return readinessScheduleFor(SELECTOR_PIPELINE_POLICIES.promotedTarget.poll, readinessTimeoutMs); +} + function buildReplayActionFlags( parentFlags: CommandFlags | undefined, actionFlags: SessionAction['flags'] | undefined, + command: string, ): CommandFlags { - return mergeParentFlags(parentFlags, { ...(actionFlags ?? {}) }); + const flags = mergeParentFlags(parentFlags, { ...(actionFlags ?? {}) }); + if (commandAcceptsReadinessBudget(command) && flags.readinessTimeoutMs === undefined) { + flags.readinessTimeoutMs = REPLAY_DEFAULT_READINESS_TIMEOUT_MS; + } + return flags; } diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts index 0cfbf22d3b..d46a0cdf09 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts @@ -17,8 +17,17 @@ import { import type { SnapshotTimingSample } from '@agent-device/contracts/capture'; import { withReplayFailureDiagnostics } from './session-replay-runtime-failure.ts'; -import { invokeReplayAction } from './session-replay-action-runtime.ts'; -import type { AdReplayStepFailure, AdReplayStepRuntime } from '@agent-device/ad-replay'; +import { + invokeReplayAction, + replayStepReadinessSchedule, +} from './session-replay-action-runtime.ts'; +import type { + AdReplayStepFailure, + AdReplayStepRuntime, + AdReplayTargetObservation, +} from '@agent-device/ad-replay'; +import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; +import type { ObservationEvidence } from '@agent-device/capture-kit/observe-until'; import { collectReplayActionArtifactPaths } from '@agent-device/replay-port/session-replay-runtime-artifacts'; import { applyReplayDispatchGuard, @@ -79,7 +88,7 @@ import type { ReplayTestAttemptStepSink } from '@agent-device/replay-test'; * engine itself never touches it. * * `lastObservation` is the analogous side-map for `buildTargetBindingFailure` - * — it reuses the SAME capture `captureObservation` just took (for its + * — it reuses the SAME capture `observeTarget` last took (for its * `screen`), mirroring the pre-R3 code's single-capture-serves-both-paths * invariant instead of taking a second, possibly-different snapshot. * @@ -88,14 +97,14 @@ import type { ReplayTestAttemptStepSink } from '@agent-device/replay-test'; * `createAdReplayStepRuntime` call covers every step), so an un-reset * `lastObservation` would silently carry a PREVIOUS step's capture into a * step that somehow reached `buildTargetBindingFailure` without its own - * `captureObservation` call first — the `?? { reason: 'observation-missing' + * `observeTarget` call first — the `?? { reason: 'observation-missing' * }` fallback below exists to name that condition, but could never actually * fire for it; it would instead attach a stale, wrong-step screen. `armStep` * runs exactly once per step, before any of this step's capabilities do — * clearing `lastObservation` there makes the fallback message correct for * ANY future call ordering, not just the current one where every * `buildTargetBindingFailure` call site happens to be preceded by this same - * step's own `captureObservation`. + * step's own `observeTarget`. */ export function createAdReplayStepRuntime(params: { ctx: ReplayStepContext; @@ -115,6 +124,8 @@ export function createAdReplayStepRuntime(params: { const { ctx, req, artifactPaths, onStep, armSaveScript } = params; let lastResponse: DaemonResponse | undefined; let lastObservation: DivergenceObservation | undefined; + /** This step's pre-dispatch readiness wait, when the gate polled more than once. */ + let gateWait: GateReadinessWait | undefined; /** * The `TargetBindingDivergenceContext` every wire-builder needs — built @@ -157,6 +168,31 @@ export function createAdReplayStepRuntime(params: { ); }; + // #1385: the pre-dispatch gate a step right after `open --relaunch` can + // race — the app may still be launching/mounting when this capture lands, + // producing a transient `capture-failed` / `sparse-snapshot` verdict that is + // not a real divergence. Bounded retry (`retryLaunchRace`) rides out that + // transition instead of failing closed on the first unlucky capture. + const captureTargetObservation = async ( + action: SessionAction, + ): Promise => { + const session = ctx.observationStore.get(); + if (!session) { + return { + state: 'unavailable', + reason: 'no-session', + hint: 'The session closed before a screen could be captured to verify the recorded target.', + }; + } + return await captureDivergenceObservation({ + session, + observationStore: ctx.observationStore, + logPath: ctx.logPath, + action, + retryLaunchRace: true, + }); + }; + const runtime: AdReplayStepRuntime = { beginTargetVerification(action, resolvedAction, _index, targetRole) { return resolveTargetVerificationEntry({ @@ -167,46 +203,60 @@ export function createAdReplayStepRuntime(params: { }); }, - async captureObservation(action, _index, options) { - const session = ctx.observationStore.get(); - // #1385: this is the pre-dispatch gate a step right after `open - // --relaunch` can race — the app may still be launching/mounting when - // this capture lands, producing a transient `capture-failed` / - // `sparse-snapshot` verdict that is not a real divergence. Bounded - // retry (`retryLaunchRace`, engine-driven) rides out that transition - // instead of failing closed on the first unlucky capture. - const observation: DivergenceObservation = session - ? await captureDivergenceObservation({ - session, - observationStore: ctx.observationStore, - logPath: ctx.logPath, + async observeTarget({ action, token }) { + const observeOnce = async (): Promise => { + const observation = await captureTargetObservation(action); + lastObservation = observation; + if (observation.state !== 'available') { + return { state: 'unavailable', reason: observation.reason, hint: observation.hint }; + } + const session = ctx.sessionStore.get(); + return { + state: 'classified', + classification: classifyPreDispatchTarget({ + // An available capture implies an active session, and the engine + // only calls this for an action carrying `targetEvidence`. + recorded: action.targetEvidence!, + token, action, - retryLaunchRace: options.retryLaunchRace, - }) - : { - state: 'unavailable', - reason: 'no-session', - hint: 'The session closed before a screen could be captured to verify the recorded target.', - }; - lastObservation = observation; - return observation.state === 'available' - ? { state: 'available', nodes: observation.nodes } - : { state: 'unavailable', reason: observation.reason, hint: observation.hint }; - }, - - classifyTarget({ action, token, nodes }) { - const session = ctx.sessionStore.get(); - return classifyPreDispatchTarget({ - // Only ever called right after a successful `captureObservation`, - // which itself only reaches `state: 'available'` when a session is - // active — `action.targetEvidence`/`session` are always defined here - // in practice. - recorded: action.targetEvidence!, - token, - action, - nodes: [...nodes], - platform: session!.device.platform, + nodes: [...observation.nodes], + platform: session!.device.platform, + }), + }; + }; + // The gate waits only where the dispatch would: a readiness-budgeted + // command resolving a selector (a `@ref` resolves without a wait). + const readiness = token.startsWith('@') + ? undefined + : await replayStepReadinessSchedule(ctx.replayReq.flags, action); + if (!readiness) return await observeOnce(); + const { observeUntil } = await import('@agent-device/capture-kit/observe-until'); + const observed = await observeUntil({ + capture: observeOnce, + verdict: (latest) => + isTargetNotRenderedYet(latest) ? { kind: 'continue' } : { kind: 'done', result: latest }, + schedule: readiness, + ...(ctx.signal ? { signal: ctx.signal } : {}), + ...(ctx.dependencies.clock ? { clock: ctx.dependencies.clock } : {}), }); + if (observed.polls.length > 1) { + gateWait = { + remainingBudgetMs: Math.max(0, readiness.budgetMs - budgetSpentMs(observed)), + readiness: { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: observed.kind, + }, + }; + emitDiagnostic({ + level: 'debug', + phase: 'interaction_target_readiness', + data: { ...gateWait.readiness, command: action.command }, + }); + } + if (observed.kind === 'done') return observed.result; + if (observed.last !== undefined) return observed.last; + throw observed.kind === 'expired' ? observed.lastError : observed.error; }, // `_stepArtifactPaths` (the pre-step snapshot) is unused here — dispatch @@ -215,6 +265,11 @@ export function createAdReplayStepRuntime(params: { async dispatchStep(action, resolvedAction, index, _stepArtifactPaths, guard) { const sourceLine = ctx.actionLines[index] ?? 1; const response = await invokeReplayAction({ + ...(gateWait + ? { + readinessTimeoutMs: gateWait.remainingBudgetMs, + } + : {}), req: applyReplayDispatchGuard(ctx.replayReq, guard), sessionName: ctx.sessionName, action, @@ -263,7 +318,11 @@ export function createAdReplayStepRuntime(params: { evidence, observation, ); - return recordFailure(response); + return recordFailure( + evidence.kind === 'selector-miss' && gateWait + ? withReadinessDetail(response, gateWait.readiness) + : response, + ); }, async buildPostDispatchTargetBindingFailure( @@ -316,6 +375,7 @@ export function createAdReplayStepRuntime(params: { // capabilities — the natural per-step boundary to clear the previous // step's capture (see this factory's own header). lastObservation = undefined; + gateWait = undefined; armSaveScript(); }, isRepairArmed: () => ctx.coordinator.view()?.repairBoundary !== undefined, @@ -453,3 +513,37 @@ function readSessionSnapshotSamplesSince( ): SnapshotTimingSample[] { return sessionStore.get()?.snapshotDiagnostics?.samples.slice(start) ?? []; } + +/** A selector miss on a readiness-budgeted step: the target may still be rendering. */ +function isTargetNotRenderedYet(observation: AdReplayTargetObservation): boolean { + return ( + observation.state === 'classified' && + !observation.classification.verified && + observation.classification.kind === 'selector-miss' + ); +} + +type GateReadinessWait = { + /** The step's readiness budget the gate left for the dispatch. */ + remainingBudgetMs: number; + /** The same evidence a dispatched readiness wait reports (`SelectorReadinessDetails`). */ + readiness: { polls: number; waitedMs: number; end: string }; +}; + +/** The budget a `budgetFrom: 'first-capture'` wait spent: its time after the first capture ended. */ +function budgetSpentMs(observed: ObservationEvidence): number { + const first = observed.polls[0]; + return first ? observed.waitedMs - (first.startedMs + first.durationMs) : 0; +} + +/** Carries the gate's wait where a dispatched step's target-not-found failure carries its own. */ +function withReadinessDetail( + response: DaemonResponse, + readiness: GateReadinessWait['readiness'], +): DaemonResponse { + if (response.ok) return response; + return { + ...response, + error: { ...response.error, details: { ...response.error.details, readiness } }, + }; +} diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts index fc705aebe2..d62d42f93d 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts @@ -122,6 +122,7 @@ function readStringDetail( } const SAFE_CAUSE_DETAIL_KEYS = [ + 'readiness', 'reason', 'recovery', 'retriable', diff --git a/packages/replay-port/src/daemon-port/session-replay-target-verification.ts b/packages/replay-port/src/daemon-port/session-replay-target-verification.ts index 0bf3feb9a4..2da5832542 100644 --- a/packages/replay-port/src/daemon-port/session-replay-target-verification.ts +++ b/packages/replay-port/src/daemon-port/session-replay-target-verification.ts @@ -73,7 +73,7 @@ import { extractReplayTargetToken, readRefLabel } from '@agent-device/replay-por // // - `resolveTargetVerificationEntry` — routing (registry lookup, session // read, wait-form parse, token extraction) for `beginTargetVerification`. -// - `classifyPreDispatchTarget` — tree matching for `classifyTarget`. +// - `classifyPreDispatchTarget` — tree matching for `observeTarget`. // - `buildRecordedUnverifiableFailureResponse` / // `buildTargetBindingFailureResponse` / // `buildPostDispatchTargetBindingFailureResponse` — capture + wire-shaping @@ -409,7 +409,7 @@ export function resolveTargetVerificationEntry(params: { } // --------------------------------------------------------------------------- -// `classifyTarget`: resolves the recorded target against an already-captured +// `observeTarget`'s classification: resolves the recorded target against an already-captured // tree using the SAME lookup/matching a real dispatch would. // // #1555 structural-quality review ("unify on the engine's types"): returns diff --git a/packages/selectors/src/interaction-error.ts b/packages/selectors/src/interaction-error.ts index 35970844c0..593173ebaf 100644 --- a/packages/selectors/src/interaction-error.ts +++ b/packages/selectors/src/interaction-error.ts @@ -12,4 +12,10 @@ export const INTERACTION_ERROR_REASONS = { refUnlabeled: 'ref_unlabeled', /** The target names a node with no usable centre to touch: a missing, non-finite, or negative rect. */ targetBoundsInvalid: 'target_bounds_invalid', + /** + * A readiness poll captured a tree the snapshot-quality verdict calls sparse and the selector did + * not resolve in it: the tree is untrustworthy, so absence proves nothing and the wait ends now. + * `details.snapshotQuality` carries the verdict. + */ + captureSparse: 'capture_sparse', } as const; diff --git a/packages/selectors/src/selector-pipeline-policy.test.ts b/packages/selectors/src/selector-pipeline-policy.test.ts new file mode 100644 index 0000000000..c074c1a223 --- /dev/null +++ b/packages/selectors/src/selector-pipeline-policy.test.ts @@ -0,0 +1,30 @@ +import { describe, expect, test } from 'vitest'; +import { SELECTOR_PIPELINE_POLICIES, readinessScheduleFor } from './selector-pipeline-policy.ts'; + +const promotedPoll = SELECTOR_PIPELINE_POLICIES.promotedTarget.poll; + +describe('readinessScheduleFor', () => { + test('polls at the row cadence with the supplied budget, counted from the first capture', () => { + expect(readinessScheduleFor(promotedPoll, 800)).toEqual({ + intervalMs: promotedPoll.intervalMs, + budgetMs: 800, + budgetFrom: 'first-capture', + }); + }); + + test('caps the supplied budget at the row ceiling', () => { + expect(readinessScheduleFor(promotedPoll, promotedPoll.maxTimeoutMs + 5_000)?.budgetMs).toBe( + promotedPoll.maxTimeoutMs, + ); + }); + + test('a row that resolves against one capture has no schedule', () => { + expect(readinessScheduleFor(SELECTOR_PIPELINE_POLICIES.resolvedTarget.poll, 800)).toBe( + undefined, + ); + }); + + test.each([undefined, 0, -1, 1.5])('a %s budget takes the one-attempt path', (budget) => { + expect(readinessScheduleFor(promotedPoll, budget)).toBe(undefined); + }); +}); diff --git a/packages/selectors/src/selector-pipeline-policy.ts b/packages/selectors/src/selector-pipeline-policy.ts index bdd4829678..aed75591b7 100644 --- a/packages/selectors/src/selector-pipeline-policy.ts +++ b/packages/selectors/src/selector-pipeline-policy.ts @@ -2,6 +2,7 @@ import { SELECTOR_RESOLUTION_POLICIES, type SelectorResolutionPolicy, } from '@agent-device/selectors'; +import type { ObservationSchedule } from '@agent-device/capture-kit/observe-until'; /** * The structural half of the per-caller selector policy (#1656), companion to @@ -38,7 +39,7 @@ export type SelectorOcclusionStage = 'exclude-and-refuse' | 'refuse' | 'ignore'; /** * Consumed by `throwIfOffscreenInteractionTarget` - * (commands/interaction/runtime/resolution.ts), the end-to-end enforcement + * (commands/interaction/runtime/target-visibility-stages.ts), the end-to-end enforcement * point including the iOS live-rect rescue probe (#1542). * * - `refuse` — a target whose tap point lies outside the viewport refuses @@ -63,8 +64,12 @@ export type SelectorOffscreenStage = 'refuse' | 'ignore'; */ export type SelectorPromotionStage = 'hittable-ancestor' | 'hittable-ancestor-below-root' | 'none'; -/** The poll budget of a row that polls, read by `selectorPollBudget`. */ -export type SelectorPollBudget = { +/** + * `wait`/`findWait`: the row supplies its own default deadline, spent whenever the caller passes + * no explicit timeout. Read by `selectorPollBudget`, which `createWaitPolling` derives its deadline + * and sleep from. + */ +export type WaitPollBudget = { /** Used when the caller passes no explicit timeout. */ defaultTimeoutMs: number; /** Delay between polls, clamped to the remaining budget. */ @@ -72,10 +77,28 @@ export type SelectorPollBudget = { }; /** - * Consumed by `selectorPollBudget`, which `createWaitPolling` derives its - * deadline and sleep from. `'none'` is an answer, not an omission: a row that - * resolves against one capture has no polling contract, so asking for its - * budget is a caller bug — never a place to default one in. + * `promotedTarget`: the row states only a ceiling and a cadence, never a default — a miss with no + * caller-supplied budget takes the one-attempt path instead of polling (`resolveSelectorInteractionTarget`). + * When the caller does supply one (`readinessTimeoutMs`, never model- or + * CLI-writable), it is capped at `maxTimeoutMs` before it bounds the readiness loop. + */ +export type ReadinessPollBudget = { + /** The ceiling a caller-supplied readiness timeout may not exceed. */ + maxTimeoutMs: number; + /** Delay between polls, clamped to the remaining budget. */ + intervalMs: number; +}; + +/** The poll budget of a row that polls: `wait`/`findWait`'s own default, or `promotedTarget`'s cap. */ +export type SelectorPollBudget = WaitPollBudget | ReadinessPollBudget; + +/** + * Consumed by `selectorPollBudget` (wait-shaped rows only) and — for `promotedTarget` — by + * `resolveSelectorInteractionTarget`'s target-readiness loop. `'none'` is an answer, not an + * omission: a row that resolves against one capture has no polling contract, so asking for its + * budget is a caller bug — never a place to default one in. Acting rows are not all `'none'`: + * `promotedTarget` polls for the target to exist, while `resolvedTarget` and every read row still + * resolve against a single capture. */ export type SelectorPollStage = SelectorPollBudget | 'none'; @@ -105,16 +128,23 @@ export type SelectorPipelinePolicy = SelectorListPolicy & { }; /** Shared by both wait loops; they differ only in what each poll resolves. */ -const WAIT_POLL_BUDGET: SelectorPollBudget = { defaultTimeoutMs: 10_000, intervalMs: 300 }; +const WAIT_POLL_BUDGET: WaitPollBudget = { defaultTimeoutMs: 10_000, intervalMs: 300 }; export const SELECTOR_PIPELINE_POLICIES = { - /** `click`/`press`/`longpress`: the tap lands on the actionable owner of the match. */ + /** + * `click`/`press`/`longpress`: the tap lands on the actionable owner of the + * match. When the caller supplies a readiness budget, polls for the target to + * appear and become resolvable (a rect), capped at this row's `maxTimeoutMs`, before refusing — + * resolution only; occlusion, off-screen, and promotion still run once, against the winning + * capture, after the loop ends. A miss with no caller-supplied budget takes the one-attempt path + * (`resolveSelectorInteractionTarget`). + */ promotedTarget: { resolution: SELECTOR_RESOLUTION_POLICIES.act, occlusion: 'exclude-and-refuse', offscreen: 'refuse', promotion: 'hittable-ancestor', - poll: 'none', + poll: { maxTimeoutMs: 2_000, intervalMs: 200 }, }, /** * `fill`/`focus`/`scroll`/gesture endpoints, and the native-ref preflight — @@ -202,6 +232,34 @@ export const SELECTOR_PIPELINE_POLICIES = { export type SelectorPipelinePolicyName = keyof typeof SELECTOR_PIPELINE_POLICIES; +/** + * The schedule a readiness wait polls its target under. The budget counts from the end of the + * first capture: that capture is the one-attempt lookup a step pays without any wait, so the + * replay target gate and the dispatch spend the budget only on retries. + */ +export type ReadinessSchedule = Readonly< + Pick & { budgetFrom: 'first-capture' } +>; + +/** + * The one owner of a readiness schedule, for the replay target gate and the dispatch alike: + * `undefined` for a row that resolves against one capture or when no positive integer budget was + * supplied, otherwise the row's cadence and the supplied budget capped at the row's `maxTimeoutMs`. + */ +export function readinessScheduleFor( + poll: ReadinessPollBudget | 'none', + readinessTimeoutMs: number | undefined, +): ReadinessSchedule | undefined { + if (poll === 'none') return undefined; + if (readinessTimeoutMs === undefined || !Number.isInteger(readinessTimeoutMs)) return undefined; + if (readinessTimeoutMs <= 0) return undefined; + return { + intervalMs: poll.intervalMs, + budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs), + budgetFrom: 'first-capture', + }; +} + /** * The two questions a row asks the engine, derived from its ambiguity contract * rather than from a hand-kept list of row names: `reject-candidates` rows go diff --git a/packages/selectors/src/selector-pipeline.test.ts b/packages/selectors/src/selector-pipeline.test.ts index b3be5a4845..1300459d06 100644 --- a/packages/selectors/src/selector-pipeline.test.ts +++ b/packages/selectors/src/selector-pipeline.test.ts @@ -316,18 +316,31 @@ test('the uniqueness rows still refuse matches that are not one wrapper chain', } }); -test('the poll stage answers only for the rows that poll', () => { +test('the poll stage answers only for the rows that poll a caller-default budget', () => { for (const row of ['wait', 'findWait'] as const) { assert.deepEqual(selectorPollBudget(SELECTOR_PIPELINE_POLICIES[row]), { defaultTimeoutMs: 10_000, intervalMs: 300, }); } - for (const row of NODE_STAGE_ROWS.filter((name) => name !== 'wait' && name !== 'findWait')) { + for (const row of NODE_STAGE_ROWS.filter( + (name) => name !== 'wait' && name !== 'findWait' && name !== 'promotedTarget', + )) { assert.throws(() => selectorPollBudget(SELECTOR_PIPELINE_POLICIES[row]), /no poll budget/, row); } }); +test('promotedTarget states a caller-capped readiness budget, not a caller-default one', () => { + assert.deepEqual(SELECTOR_PIPELINE_POLICIES.promotedTarget.poll, { + maxTimeoutMs: 2_000, + intervalMs: 200, + }); + assert.throws( + () => selectorPollBudget(SELECTOR_PIPELINE_POLICIES.promotedTarget), + /caller-capped readiness budget/, + ); +}); + test('a listing row cannot be handed the node stages it does not declare', () => { // The `@ts-expect-error` directives ARE the assertion: a listing has no // single target to retarget, keep on screen, or wait for, so widening @@ -390,7 +403,7 @@ test('the documented per-caller pipelines are the ones declared', () => { }), ), { - promotedTarget: ['exclude-and-refuse', 'refuse', 'hittable-ancestor', 'no-poll'], + promotedTarget: ['exclude-and-refuse', 'refuse', 'hittable-ancestor', 'poll'], resolvedTarget: ['exclude-and-refuse', 'refuse', 'none', 'no-poll'], coveredDiagnosis: ['refuse', 'ignore', 'none', 'no-poll'], readText: ['ignore', 'ignore', 'none', 'no-poll'], diff --git a/packages/selectors/src/selector-pipeline.ts b/packages/selectors/src/selector-pipeline.ts index f7d8518cba..f43ce7d408 100644 --- a/packages/selectors/src/selector-pipeline.ts +++ b/packages/selectors/src/selector-pipeline.ts @@ -14,9 +14,9 @@ import type { CandidateSetPipelinePolicy, SelectorListPolicy, SelectorPipelinePolicy, - SelectorPollBudget, SelectorPromotionStage, SingleTargetPipelinePolicy, + WaitPollBudget, } from './selector-pipeline-policy.ts'; /** @@ -262,12 +262,21 @@ async function guardOffscreen( return await hooks.offscreen(node, nodes); } -/** The poll stage: the budget `createWaitPolling` runs the row under. */ -export function selectorPollBudget(policy: SelectorPipelinePolicy): SelectorPollBudget { +/** + * The poll stage: the caller-default budget `createWaitPolling` runs a wait-shaped row under. + * `promotedTarget`'s row polls too, but under a caller-supplied, row-capped budget rather than a + * row-owned default — it has no `defaultTimeoutMs` to hand back, so it is not a legal input here. + */ +export function selectorPollBudget(policy: SelectorPipelinePolicy): WaitPollBudget { if (policy.poll === 'none') { throw new Error( 'selector pipeline row resolves against one capture and declares no poll budget', ); } + if (!('defaultTimeoutMs' in policy.poll)) { + throw new Error( + 'selector pipeline row polls under a caller-capped readiness budget, not a caller-default one', + ); + } return policy.poll; } diff --git a/scripts/layering/package-boundaries.test.ts b/scripts/layering/package-boundaries.test.ts index ea6b4f56ef..3f41c6c817 100644 --- a/scripts/layering/package-boundaries.test.ts +++ b/scripts/layering/package-boundaries.test.ts @@ -408,6 +408,7 @@ test('the real tree parses, declares, and passes R11', () => { '@agent-device/capture-kit/ios-snapshot-runtime', '@agent-device/capture-kit/ios-snapshot-tree', '@agent-device/capture-kit/mobile-snapshot-semantics', + '@agent-device/capture-kit/observe-until', '@agent-device/capture-kit/perf-capture-admission-ledger', '@agent-device/capture-kit/perf-capture-recovery', '@agent-device/capture-kit/perf-capture-resource-store', @@ -652,6 +653,7 @@ test('the real tree parses, declares, and passes R11', () => { 'AdReplayStepRuntime', 'AdReplayTargetBindingEvidence', 'AdReplayTargetClassification', + 'AdReplayTargetObservation', 'AdReplayVarSources', 'AdReplayVerificationEntry', 'inspectAdReplay', diff --git a/src/__tests__/client-interactions-readiness.test.ts b/src/__tests__/client-interactions-readiness.test.ts new file mode 100644 index 0000000000..9b77ccf024 --- /dev/null +++ b/src/__tests__/client-interactions-readiness.test.ts @@ -0,0 +1,19 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { createAgentDeviceClient } from '../agent-device-client.ts'; +import { createTransport } from './client-transport-fixture.ts'; + +// The readiness budget must reach `req.flags` for press/click/longpress exactly like +// `button`/`durationMs` do. It is never CLI- or model-writable; the SDK client option is its route. +test('readinessTimeoutMs on press/click/longpress reaches the request flags', async () => { + const setup = createTransport(async () => ({ ok: true, data: {} })); + const client = createAgentDeviceClient(setup.config, { transport: setup.transport }); + + await client.interactions.press({ selector: 'label=Foo', readinessTimeoutMs: 2_000 }); + await client.interactions.click({ selector: 'label=Foo', readinessTimeoutMs: 1_500 }); + await client.interactions.longPress({ selector: 'label=Foo', readinessTimeoutMs: 900 }); + + assert.equal(setup.calls[0]?.flags?.readinessTimeoutMs, 2_000); + assert.equal(setup.calls[1]?.flags?.readinessTimeoutMs, 1_500); + assert.equal(setup.calls[2]?.flags?.readinessTimeoutMs, 900); +}); diff --git a/src/__tests__/command-descriptor-timeout-policy.test.ts b/src/__tests__/command-descriptor-timeout-policy.test.ts index 03f49a5b83..692d6bc8c4 100644 --- a/src/__tests__/command-descriptor-timeout-policy.test.ts +++ b/src/__tests__/command-descriptor-timeout-policy.test.ts @@ -2,12 +2,16 @@ import { test } from 'vitest'; import assert from 'node:assert/strict'; import { PUBLIC_COMMANDS } from '@agent-device/command-registry/catalog'; import { + commandAcceptsReadinessBudget, commandDescriptors, resolveCommandPostActionObservationSupport, resolveCommandTimeoutPolicy, } from '@agent-device/command-registry/registry'; +import { INTERACTION_DISPATCH_PATHS } from '@agent-device/contracts/interaction-guarantees'; +import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import { DEFAULT_TIMEOUT_POLICY, + READINESS_BUDGET_MAX_MS, resolveCommandRequestTimeoutMs, } from '@agent-device/command-registry/timeout-policy'; import { DEFAULT_STABLE_TIMEOUT_MS } from '../commands/interaction/runtime/stable-capture.ts'; @@ -403,3 +407,57 @@ test('open and prepare startup budgets keep a client-envelope margin over the da 90_000, ); }); + +test('a readiness budget widens the request envelope on top of the settle envelope', () => { + const press = resolveCommandTimeoutPolicy('press'); + const withoutReadiness = resolveCommandRequestTimeoutMs(press, { flags: {} }); + assert.equal(withoutReadiness, 90_000); + assert.equal( + resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 2_000 } }), + 92_000, + ); + assert.equal( + resolveCommandRequestTimeoutMs(press, { flags: { settle: true, readinessTimeoutMs: 2_000 } }), + 90_000 + 10_000 + 30_000 + 2_000, + ); + assert.equal( + resolveCommandRequestTimeoutMs(resolveCommandTimeoutPolicy('longpress'), { + flags: { readinessTimeoutMs: 2_000 }, + }), + 212_000, + ); + assert.equal(resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 0 } }), 90_000); + assert.equal( + resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 999_000 } }), + 90_000 + 2_000, + ); +}); + +test('the envelope readiness cap is the promotedTarget row poll ceiling', () => { + assert.equal( + READINESS_BUDGET_MAX_MS, + SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs, + ); + assert.equal( + resolveCommandRequestTimeoutMs(resolveCommandTimeoutPolicy('fill'), { + flags: { readinessTimeoutMs: 2_000 }, + }), + 90_000, + ); +}); + +test('the readiness-budgeted commands are the ones a runtime targetReadiness cell enforces', () => { + const declared = commandDescriptors + .map((descriptor) => descriptor.name) + .filter((command) => commandAcceptsReadinessBudget(command)) + .sort(); + const enforced = [ + ...new Set( + Object.values(INTERACTION_DISPATCH_PATHS).flatMap((path) => { + const cell = path.guarantees.targetReadiness; + return cell.kind === 'runtime' ? (cell.appliesTo ?? path.commands) : []; + }), + ), + ].sort(); + assert.deepEqual(declared, enforced); +}); diff --git a/src/commands/command-flags.ts b/src/commands/command-flags.ts index 4699c2c0bc..f7b44a4618 100644 --- a/src/commands/command-flags.ts +++ b/src/commands/command-flags.ts @@ -87,6 +87,7 @@ function buildFlags(options: InternalRequestOptions): CommandFlags { durationMs: options.durationMs, holdMs: options.holdMs, jitterPx: options.jitterPx, + readinessTimeoutMs: options.readinessTimeoutMs, pixels: options.pixels, until: options.until, doubleTap: options.doubleTap, diff --git a/src/commands/command-input.ts b/src/commands/command-input.ts index 1fa4b59ebb..675e1b9d84 100644 --- a/src/commands/command-input.ts +++ b/src/commands/command-input.ts @@ -412,6 +412,7 @@ export function readFieldInput( ); const commonInput = readCommonInput(record, { readTargetAlias: !Object.hasOwn(fields, 'target'), + readinessBudgetDeclared: Object.hasOwn(fields, 'readinessTimeoutMs'), }); return compactRecord({ ...commonInput, diff --git a/src/commands/common-input-fields.ts b/src/commands/common-input-fields.ts index e53672c5c9..ba88f9e0f2 100644 --- a/src/commands/common-input-fields.ts +++ b/src/commands/common-input-fields.ts @@ -32,7 +32,11 @@ export type CommonCommandInput = Pick< noRecord?: boolean; }; -export type CommonInputReadOptions = { readTargetAlias?: boolean }; +export type CommonInputReadOptions = { + readTargetAlias?: boolean; + /** The command's own fields declare `readinessTimeoutMs` (`interaction/metadata.ts`). */ + readinessBudgetDeclared?: boolean; +}; /** * `cli-grammar/common.ts`'s two flag-derived projections a row can join: @@ -212,6 +216,11 @@ const COMMON_INPUT_FIELDS = { flagKey: 'noRecord', flagIn: ['input', 'selection'], }, + readinessTimeoutMs: { + // No schema and no value: a command whose descriptor declares `targetReadiness: 'budgeted'` + // carries and reads the key through its own fields; every other command refuses it here. + read: refuseUndeclaredReadinessBudget, + }, daemonBaseUrl: { schema: { type: 'string', description: 'Remote daemon base URL.' }, read: (record) => optionalString(record, 'daemonBaseUrl'), @@ -246,7 +255,10 @@ const COMMON_INPUT_FIELDS = { schema: { type: 'boolean', description: 'Enable debug diagnostics.' }, read: (record) => optionalBoolean(record, 'debug'), }, -} as const satisfies Record; +} as const satisfies Record< + keyof CommonCommandInput | 'target' | 'readinessTimeoutMs', + CommonInputFieldSpec +>; const COMMON_INPUT_ROWS: ReadonlyArray = Object.entries(COMMON_INPUT_FIELDS); @@ -324,3 +336,16 @@ function readDeviceTarget( } return deviceTarget ?? targetAlias; } + +function refuseUndeclaredReadinessBudget( + record: Record, + options: CommonInputReadOptions, +): undefined { + if (options.readinessBudgetDeclared === true || !Object.hasOwn(record, 'readinessTimeoutMs')) { + return undefined; + } + throw new AppError( + 'INVALID_ARGS', + 'readinessTimeoutMs applies only to commands that wait for their target: press, click, and longpress.', + ); +} diff --git a/src/commands/interaction/index.ts b/src/commands/interaction/index.ts index 6ab009896b..e17f46c9e3 100644 --- a/src/commands/interaction/index.ts +++ b/src/commands/interaction/index.ts @@ -667,6 +667,7 @@ function toClickOptions(input: ClickInput): ClickOptions { ...toRepeatedOptions(input), button: input.button, verify: input.verify, + readinessTimeoutMs: input.readinessTimeoutMs, ...toSettleOptions(input), }; } @@ -678,6 +679,7 @@ function toPressOptions(input: PressInput): PressOptions { ...toSelectorSnapshotOptions(input), ...toRepeatedOptions(input), verify: input.verify, + readinessTimeoutMs: input.readinessTimeoutMs, ...toSettleOptions(input), }; } @@ -701,6 +703,7 @@ function toLongPressOptions(input: LongPressInput): LongPressOptions { ...toClientInteractionTarget(input.target), ...toSelectorSnapshotOptions(input), durationMs: input.durationMs, + readinessTimeoutMs: input.readinessTimeoutMs, ...toSettleOptions(input), }; } diff --git a/src/commands/interaction/metadata.test.ts b/src/commands/interaction/metadata.test.ts index 182265e17b..7d108661e5 100644 --- a/src/commands/interaction/metadata.test.ts +++ b/src/commands/interaction/metadata.test.ts @@ -1,5 +1,7 @@ import { expect, test } from 'vitest'; import { AppError } from '@agent-device/kernel/errors'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import { findCommandMetadata, listCommandMetadata } from '../command-metadata.ts'; import { interactionCommandMetadata } from './metadata.ts'; // `fill ""` is the clear-field primitive (#2063): before it existed, emptying an input @@ -50,3 +52,34 @@ test('type keeps refusing an empty text: appending nothing is not a clear', () = expect((error as AppError).code).toBe('INVALID_ARGS'); } }); + +const SELECTOR_TARGET = { kind: 'selector', selector: 'label=Continue' }; + +test('a command advertises readinessTimeoutMs exactly when its descriptor declares the budget', () => { + for (const metadata of listCommandMetadata()) { + const properties = metadata.inputSchema.properties ?? {}; + expect('readinessTimeoutMs' in properties, metadata.name).toBe( + commandAcceptsReadinessBudget(metadata.name), + ); + } +}); + +test('press reads readinessTimeoutMs', () => { + const input = findCommandMetadata('press').readInput({ + target: SELECTOR_TARGET, + readinessTimeoutMs: 2_000, + }) as { readinessTimeoutMs?: number }; + expect(input.readinessTimeoutMs).toBe(2_000); +}); + +test('fill refuses readinessTimeoutMs with an input error, and reads without it', () => { + const fill = findCommandMetadata('fill'); + expect(() => fill.readInput({ target: SELECTOR_TARGET, text: 'hi' })).not.toThrow(); + let refusal: unknown; + try { + fill.readInput({ target: SELECTOR_TARGET, text: 'hi', readinessTimeoutMs: 2_000 }); + } catch (error) { + refusal = error; + } + expect(refusal instanceof AppError && refusal.code).toBe('INVALID_ARGS'); +}); diff --git a/src/commands/interaction/metadata.ts b/src/commands/interaction/metadata.ts index 6faba1ad95..62a5b4ff26 100644 --- a/src/commands/interaction/metadata.ts +++ b/src/commands/interaction/metadata.ts @@ -21,6 +21,7 @@ import { SWIPE_REPETITION_MAX, } from '@agent-device/contracts/scroll-gesture'; import { FIND_LOCATORS } from '@agent-device/selectors'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; import { booleanField, elementTargetField, @@ -28,6 +29,7 @@ import { integerField, interactionTargetField, numberField, + operatorField, pointField, repeatedFields, requiredField, @@ -78,12 +80,37 @@ const interactionCommandDescriptions = { type InteractionCommandName = keyof typeof interactionCommandDescriptions; +/** + * The input field a command's `targetReadiness: 'budgeted'` descriptor trait entitles it to. Only + * those commands declare `readinessTimeoutMs`; the common input reader refuses the key for every + * command whose fields do not (`common-input-fields.ts`). Fails closed at module load for a command + * without the trait, so a field map cannot advertise a budget its runtime never polls under. + */ +function targetReadinessFields(command: InteractionCommandName) { + if (!commandAcceptsReadinessBudget(command)) { + throw new Error(`${command} does not declare targetReadiness: 'budgeted'`); + } + return { + readinessTimeoutMs: operatorField( + integerField( + "Operator-only: how long the command may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", + { min: 1 }, + ), + { + operatorPath: + 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', + }, + ), + }; +} + const clickFields = { target: requiredField(interactionTargetField()), button: enumField(CLICK_BUTTONS, 'Pointer button for platforms that support mouse buttons.'), ...selectorSnapshotFields(), ...repeatedFields(), ...postActionObservationFields('click'), + ...targetReadinessFields('click'), }; const pressFields = { @@ -91,6 +118,7 @@ const pressFields = { ...selectorSnapshotFields(), ...repeatedFields(), ...postActionObservationFields('press'), + ...targetReadinessFields('press'), }; const fillFields = { @@ -113,6 +141,7 @@ const longPressFields = { durationMs: integerField('Long press duration in milliseconds.', { min: 0 }), ...selectorSnapshotFields(), ...postActionObservationFields('longpress'), + ...targetReadinessFields('longpress'), }; const hoverFields = { diff --git a/src/commands/interaction/runtime/__tests__/test-utils/index.ts b/src/commands/interaction/runtime/__tests__/test-utils/index.ts index c6013d40aa..83a48df40e 100644 --- a/src/commands/interaction/runtime/__tests__/test-utils/index.ts +++ b/src/commands/interaction/runtime/__tests__/test-utils/index.ts @@ -342,6 +342,8 @@ export function createInteractionDevice( > & { platform?: AgentDeviceBackend['platform']; sessionMetadata?: Record; + /** An advancing clock, for interactions that poll (the promotedTarget readiness loop). */ + clock?: { now: () => number; sleep: (ms: number) => Promise }; } = {}, ) { return createAgentDevice({ @@ -378,6 +380,7 @@ export function createInteractionDevice( { name: 'default', snapshot, metadata: overrides.sessionMetadata }, ]), policy: localCommandPolicy(), + ...(overrides.clock ? { clock: overrides.clock } : {}), }); } diff --git a/src/commands/interaction/runtime/gestures.ts b/src/commands/interaction/runtime/gestures.ts index a4b045e79e..df71ea3095 100644 --- a/src/commands/interaction/runtime/gestures.ts +++ b/src/commands/interaction/runtime/gestures.ts @@ -27,13 +27,13 @@ import { } from './post-action-observation.ts'; import { assertSupportedInteractionSurface, - captureInteractionSnapshot, - dispatchNativeRefInteraction, resolveInteractionTarget, - type ExpectedResolvedTarget, type InteractionTarget, type ResolvedInteractionTarget, } from './resolution.ts'; +import { dispatchNativeRefInteraction } from './native-ref-interaction.ts'; +import type { ExpectedResolvedTarget } from './interaction-resolution-request.ts'; +import { captureInteractionSnapshot } from './interaction-snapshot-capture.ts'; import { resolveVisibleSnapshotViewport } from './viewport.ts'; type DragRecordingTarget = { @@ -95,6 +95,11 @@ export type FocusCommandResult = ResolvedInteractionTarget & BackendResultEnvelo export type LongPressCommandOptions = CommandContext & { target: InteractionTarget; durationMs?: number; + /** + * Polls for the target to exist and become actionable, capped at the promotedTarget row's + * maxTimeoutMs. Absent takes one attempt. + */ + readinessTimeoutMs?: number; /** ADR 0012 step 4: replay-only post-resolution guard; see resolution.ts. */ expectedResolvedTarget?: ExpectedResolvedTarget; } & SettlePostActionObservationOptions; @@ -142,6 +147,7 @@ export const longPressCommand: RuntimeCommand< pipeline: SELECTOR_PIPELINE_POLICIES.promotedTarget, captureEvidenceBaseline: observation.needsPreActionBaseline, expectedResolvedTarget: options.expectedResolvedTarget, + readinessTimeoutMs: options.readinessTimeoutMs, }); if (!runtime.backend.longPress) { throw new AppError('UNSUPPORTED_OPERATION', 'longPress is not supported by this backend'); diff --git a/src/commands/interaction/runtime/interaction-resolution-request.ts b/src/commands/interaction/runtime/interaction-resolution-request.ts new file mode 100644 index 0000000000..1a2507c776 --- /dev/null +++ b/src/commands/interaction/runtime/interaction-resolution-request.ts @@ -0,0 +1,73 @@ +import type { ActingPipelinePolicy } from '@agent-device/selectors/selector-pipeline-policy'; +import type { PreresolvedInteractionTarget } from '@agent-device/contracts/interaction'; +import type { ReplayTargetGuardDenotation } from '@agent-device/contracts/replay'; + +/** + * ADR 0012 migration step 4, post-resolution guard: the LOCAL identity AND + * the STRUCTURAL denotation (pre-order document index + same-parent sibling + * ordinal) of the element replay's pre-action verification isolated. Set ONLY + * by the replay step loop (via `DaemonRequest.internal.replayTargetGuard`) for + * annotated verified actions — never on live interactive commands. + * + * Local identity alone is insufficient: ADR path 6 isolates ONE member among + * several nodes that share the same `{id, role, label}` using sibling / + * region-scoped viewportOrder. If verification isolates duplicate A but + * dispatch's occlusion/visibility filtering selects duplicate B with the same + * local identity, a local-identity-only guard would pass and tap the wrong + * element. The structural denotation is the discriminator that catches that + * split BEFORE the device action. + */ +export type ExpectedResolvedTarget = ReplayTargetGuardDenotation; + +export type InteractionAction = + | 'click' + | 'press' + | 'fill' + | 'focus' + | 'longPress' + | 'hover' + | 'scroll' + | 'swipe' + | 'pinch' + | 'pan' + | 'drag' + | 'fling' + | 'rotate' + | 'transform'; + +export type ResolveInteractionTargetParams = { + action: InteractionAction; + requireInteractive: boolean; + /** + * The structural pipeline this action runs (#1656): occlusion, off-screen, + * and hittable-ancestor promotion are the row's decisions. `promotedTarget` + * for tap-shaped actions, `resolvedTarget` for the actions that must keep + * the element they resolved. + */ + pipeline: ActingPipelinePolicy; + /** + * How long a `promotedTarget` row may poll for a target that does not exist yet (never model- or + * CLI-writable); `resolveSelectorInteractionTarget` caps it at the row's `maxTimeoutMs`. Anything + * other than a positive integer, and every `resolvedTarget` row, takes one attempt. + */ + readinessTimeoutMs?: number; + /** + * `--verify` (#1047): also capture the pre-action node set for a `point` target + * so `changedFromBefore` evidence has a baseline. Ref/selector targets already + * capture a snapshot to resolve the target, so this is a no-op cost for them — + * their nodes are attached below regardless of this flag. For point targets, + * which normally skip capture entirely, this opts into one extra capture, only + * when the caller explicitly asked for verify evidence. Defaults to false. + */ + captureEvidenceBaseline?: boolean; + /** ADR 0012 step 4 post-resolution guard; see `ExpectedResolvedTarget`. */ + expectedResolvedTarget?: ExpectedResolvedTarget; + /** Identifies one endpoint when a multi-target replay guard refuses. */ + replayTargetRole?: 'source' | 'destination'; + /** + * #1654: the caller already resolved this `@ref` against its own capture, so + * the ref branch adopts that node instead of looking the ref up again. Ref + * targets only — a selector target has nothing pre-resolved to adopt. + */ + preresolvedTarget?: PreresolvedInteractionTarget; +}; diff --git a/src/commands/interaction/runtime/interaction-snapshot-capture.ts b/src/commands/interaction/runtime/interaction-snapshot-capture.ts new file mode 100644 index 0000000000..5ec85592d0 --- /dev/null +++ b/src/commands/interaction/runtime/interaction-snapshot-capture.ts @@ -0,0 +1,43 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotState } from '@agent-device/kernel/snapshot'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { now, toBackendContext } from '../../runtime-common.ts'; + +/** + * A leaf shared by every interaction runtime module that needs a fresh (or session-cached) tree: + * `resolution.ts`'s ref/selector resolution, `selector-readiness.ts`'s readiness poll, + * `gestures.ts`'s viewport resolution, and `post-action-observation.ts`'s `--verify` baseline. + * Deliberately self-contained — it imports no sibling of this directory, so those siblings can + * import it without ever risking a cycle back. + */ + +export type InteractionSnapshot = { + snapshot: SnapshotState; +}; + +export async function captureInteractionSnapshot( + runtime: AgentDeviceRuntime, + options: CommandContext, + interactiveOnly: boolean, +): Promise { + if (!runtime.backend.captureSnapshot) { + throw new AppError('UNSUPPORTED_OPERATION', 'snapshot is not supported by this backend'); + } + const sessionName = options.session ?? 'default'; + const session = await runtime.sessions.get(sessionName); + if (!session) throw new AppError('SESSION_NOT_FOUND', 'No active session. Run open first.'); + const result = await runtime.backend.captureSnapshot(toBackendContext(runtime, options), { + interactiveOnly, + includeRects: true, + }); + const snapshot = + result.snapshot ?? + ({ + nodes: result.nodes ?? [], + truncated: result.truncated, + backend: result.backend as SnapshotState['backend'], + createdAt: now(runtime), + } satisfies SnapshotState); + await runtime.sessions.set({ ...session, snapshot }); + return { snapshot }; +} diff --git a/src/commands/interaction/runtime/interaction-verb.ts b/src/commands/interaction/runtime/interaction-verb.ts index 6e2809514e..31dbe7fe4a 100644 --- a/src/commands/interaction/runtime/interaction-verb.ts +++ b/src/commands/interaction/runtime/interaction-verb.ts @@ -1,4 +1,4 @@ -import type { InteractionAction } from './resolution.ts'; +import type { InteractionAction } from './interaction-resolution-request.ts'; /** * The passive verb a refusal uses for the action it declined, so every acting guard that refuses diff --git a/src/commands/interaction/runtime/interactions.ts b/src/commands/interaction/runtime/interactions.ts index 5f5c18afe9..890cc7e603 100644 --- a/src/commands/interaction/runtime/interactions.ts +++ b/src/commands/interaction/runtime/interactions.ts @@ -19,12 +19,9 @@ import { planPostActionObservation, type PostActionObservationOptions, } from './post-action-observation.ts'; -import { - dispatchNativeRefInteraction, - resolveInteractionTarget, - type ExpectedResolvedTarget, - type InteractionTarget, -} from './resolution.ts'; +import { resolveInteractionTarget, type InteractionTarget } from './resolution.ts'; +import { dispatchNativeRefInteraction } from './native-ref-interaction.ts'; +import type { ExpectedResolvedTarget } from './interaction-resolution-request.ts'; export { focusCommand, hoverCommand, longPressCommand } from './gestures.ts'; export type { @@ -41,6 +38,11 @@ export type PressCommandOptions = CommandContext & RepeatedInput & { target: InteractionTarget; button?: ClickButton; + /** + * Polls for the target to exist and become actionable, capped at the promotedTarget row's + * maxTimeoutMs. Absent takes one attempt. + */ + readinessTimeoutMs?: number; /** ADR 0012 step 4: replay-only post-resolution guard; see resolution.ts. */ expectedResolvedTarget?: ExpectedResolvedTarget; /** #1654: a mutating `find`'s already-resolved node; see resolution.ts. */ @@ -152,6 +154,7 @@ async function tapCommand( captureEvidenceBaseline: observation.needsPreActionBaseline, expectedResolvedTarget: options.expectedResolvedTarget, preresolvedTarget: options.preresolvedTarget, + readinessTimeoutMs: options.readinessTimeoutMs, }); if (!runtime.backend.tap) { throw new AppError('UNSUPPORTED_OPERATION', 'tap is not supported by this backend'); diff --git a/src/commands/interaction/runtime/keyboard-occlusion.ts b/src/commands/interaction/runtime/keyboard-occlusion.ts index b48bb2e10e..edb28812bf 100644 --- a/src/commands/interaction/runtime/keyboard-occlusion.ts +++ b/src/commands/interaction/runtime/keyboard-occlusion.ts @@ -13,7 +13,7 @@ import { type KeyboardSurface, } from '@agent-device/contracts/tap-keyboard-occlusion'; import { interactionVerb } from './interaction-verb.ts'; -import type { InteractionAction } from './resolution.ts'; +import type { InteractionAction } from './interaction-resolution-request.ts'; /** * The keyboard guard every acting path runs against the tree it already holds. diff --git a/src/commands/interaction/runtime/native-ref-interaction.ts b/src/commands/interaction/runtime/native-ref-interaction.ts new file mode 100644 index 0000000000..85fac76355 --- /dev/null +++ b/src/commands/interaction/runtime/native-ref-interaction.ts @@ -0,0 +1,135 @@ +import type { SnapshotNode } from '@agent-device/kernel/snapshot'; +import { normalizeRef } from '@agent-device/kernel/snapshot'; +import { resolveRectCenter } from '@agent-device/kernel/rect-center'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; +import { surfaceScopedNodes } from './post-action-surface.ts'; +import type { + InteractionTarget, + ResolvedInteractionTarget, + SurfaceScopedNodes, +} from '@agent-device/contracts/interaction'; +import type { + BackendActionResult, + BackendCommandContext, + BackendRefTarget, +} from '../../../backend.ts'; +import { toBackendContext } from '../../runtime-common.ts'; +import { toBackendResult } from '../../runtime-types.ts'; +import type { InteractionAction } from './interaction-resolution-request.ts'; +import { tryResolveRefNode } from './ref-target-resolution.ts'; +import { EXACT_REF_RESOLUTION, describeNonHittableTarget } from './resolution-disclosure.ts'; +import { + assertVisibleRefTarget, + runInteractionPipelineStages, +} from './target-visibility-stages.ts'; + +/** + * ADR 0011 native-ref preflight: `click @ref` / `fill @ref` fast paths + * dispatch straight to `backend.tapTarget`/`fillTarget`, and a backend fast + * path can silently "succeed" — delegation-on-error never triggers there. The + * ref came from the stored session snapshot, so the node is already in hand: + * run the SAME shared guards the runtime path uses against it before the + * backend call — occlusion (`isSnapshotNodeInteractionBlocked` via + * `assertInteractionNotBlocked`) and offscreen (the snapshot visibility resolver via + * `assertVisibleRefTarget`) ERROR with the runtime path's exact shapes, and + * the non-hittable annotation is returned for the fast-path result. + * + * Zero extra round trips by construction on the accept path: no session, no + * stored snapshot, an unresolvable/invalid ref, or a node without a usable + * rect all make the preflight a no-op and the fast path proceeds exactly as + * before. Promotion to a hittable ancestor stays a runtime-path behavior — + * the preflight never changes which element the backend acts on. Exception: + * a would-be off-screen refusal may spend one extra iOS runner round trip + * (#1542's double-check) before erroring — cost only on the path that was + * about to fail anyway. + * + * Exported as an ADR 0011 registry anchor (interaction-guarantees.ts `via` + * symbol, imported dynamically by the gate test); production callers reach + * it through `dispatchNativeRefInteraction`. + */ +// fallow-ignore-next-line unused-export +export async function preflightNativeRefInteraction( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, + action: InteractionAction, +): Promise<{ + targetHittable?: boolean; + hint?: string; + node?: SnapshotNode; + preAction?: SurfaceScopedNodes; +}> { + const session = await runtime.sessions.get(options.session ?? 'default'); + const storedSnapshot = session?.snapshot; + const nodes = storedSnapshot?.nodes; + if (!storedSnapshot || !nodes || normalizeRef(target.ref) === null) return {}; + const outcome = tryResolveRefNode(nodes, target.ref, { + fallbackLabel: target.fallbackLabel ?? '', + }); + if (outcome.kind !== 'resolved') return {}; + const { resolved } = outcome; + // `resolvedTarget` whatever the command: its `none` promotion is what holds + // ADR 0011's "the preflight never changes which element the backend acts on". + const pipeline = SELECTOR_PIPELINE_POLICIES.resolvedTarget; + // #1542: dispatches by REF, not coordinate, so no point is re-derived for the dispatch — but + // evidence/annotation below still describes the returned (visible) node. + const { node: visibleNode } = await runInteractionPipelineStages({ + policy: pipeline, + nodes, + node: resolved.node, + action, + label: `Ref ${target.ref}`, + hooks: { + offscreen: async (node, tree) => + await assertVisibleRefTarget(runtime, options, node, tree, target.ref, { + action, + pipeline, + }), + }, + resolveTapPoint: (node) => resolveRectCenter(node.rect), + }); + return { + ...describeNonHittableTarget(visibleNode, action), + // ADR 0012 decision 3: the guard lookup above doubles as the record-time + // evidence source for the fast path, at zero extra capture cost. + node: visibleNode, + preAction: surfaceScopedNodes(storedSnapshot), + }; +} + +/** + * ADR 0011 native-ref dispatch, shared by click/fill/hover @ref: run the + * preflight guards against the stored node, hand the ref to the backend as + * its own element handle, and return the exact-ref result envelope. Callers + * decide WHEN the path applies (backend capability, no non-default options, + * no replay guard, no settle baseline); this owns only the dispatch itself so + * the three commands cannot drift on preflight or disclosure. + */ +export async function dispatchNativeRefInteraction( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, + action: InteractionAction, + dispatch: ( + context: BackendCommandContext, + refTarget: BackendRefTarget, + ) => Promise, +): Promise< + Extract & { backendResult?: Record } +> { + const preflight = await preflightNativeRefInteraction(runtime, options, target, action); + const backendResult = await dispatch(toBackendContext(runtime, options), { + kind: 'ref', + ref: target.ref, + ...(target.fallbackLabel ? { fallbackLabel: target.fallbackLabel } : {}), + }); + const formattedBackendResult = toBackendResult(backendResult); + return { + kind: 'ref', + target: { kind: 'ref', ref: target.ref }, + resolution: EXACT_REF_RESOLUTION, + ...preflight, + ...(formattedBackendResult ? { backendResult: formattedBackendResult } : {}), + }; +} diff --git a/src/commands/interaction/runtime/post-action-observation.ts b/src/commands/interaction/runtime/post-action-observation.ts index 68f78de53b..a0b513f192 100644 --- a/src/commands/interaction/runtime/post-action-observation.ts +++ b/src/commands/interaction/runtime/post-action-observation.ts @@ -5,7 +5,7 @@ import type { SettleObservation, SettleParams, } from '@agent-device/contracts/interaction'; -import { captureInteractionSnapshot } from './resolution.ts'; +import { captureInteractionSnapshot } from './interaction-snapshot-capture.ts'; import { summarizePostActionEvidence, surfaceScopedNodes } from './post-action-surface.ts'; import { settleAfterInteraction, settleEvidence } from './settle.ts'; diff --git a/src/commands/interaction/runtime/ref-target-resolution.test.ts b/src/commands/interaction/runtime/ref-target-resolution.test.ts new file mode 100644 index 0000000000..a8e942e409 --- /dev/null +++ b/src/commands/interaction/runtime/ref-target-resolution.test.ts @@ -0,0 +1,126 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { ref } from './selector-read-utils.ts'; +import { tryResolveRefNode } from './ref-target-resolution.ts'; +import { STALE_REF_HINT } from '@agent-device/selectors'; +import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; +import type { Point } from '@agent-device/kernel/snapshot'; +import { createInteractionDevice, selectorSnapshot } from './__tests__/test-utils/index.ts'; + +test('runtime ref interactions fail closed when the authorized ref has no usable bounds (ADR 0014)', async () => { + const staleSnapshot = makeSnapshotState([ + { + index: 0, + depth: 0, + type: 'Button', + label: 'Continue', + hittable: true, + }, + ]); + const calls: Point[] = []; + let captures = 0; + const device = createInteractionDevice(staleSnapshot, { + captureSnapshot: async () => { + captures += 1; + return { snapshot: selectorSnapshot() }; + }, + tap: async (_context, point) => { + calls.push(point); + }, + }); + + // ADR 0014: the authorized frame's @e1 has no usable rect, so it FAILS rather + // than recapturing and accepting the same index from a newer tree by + // positional coincidence. + await assert.rejects( + () => device.interactions.click(ref('@e1'), { session: 'default' }), + (error: unknown) => { + assert.match((error as Error).message, /Ref @e1 has no usable bounds/); + assert.deepEqual( + (error as { details?: Record }).details, + { reason: 'target_bounds_invalid', ref: 'e1', hint: STALE_REF_HINT }, + 'the frame lists @e1, so the refusal names the bounds, not a missing ref', + ); + return true; + }, + ); + assert.equal(captures, 0); + assert.deepEqual(calls, []); +}); + +test('runtime ref interactions refuse a ref the authorized frame does not list with ref_not_found', async () => { + const calls: Point[] = []; + let captures = 0; + const device = createInteractionDevice(selectorSnapshot(), { + captureSnapshot: async () => { + captures += 1; + return { snapshot: selectorSnapshot() }; + }, + tap: async (_context, point) => { + calls.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.click(ref('@e9'), { session: 'default' }), + (error: unknown) => { + assert.match((error as Error).message, /Ref @e9 not found/); + assert.deepEqual((error as { details?: Record }).details, { + reason: 'ref_not_found', + ref: 'e9', + hint: STALE_REF_HINT, + }); + return true; + }, + ); + assert.equal(captures, 0); + assert.deepEqual(calls, []); +}); + +test('tryResolveRefNode discloses exact for a resolved ref and label-fallback for label recovery', () => { + const nodes = selectorSnapshot().nodes; + + const exact = tryResolveRefNode(nodes, '@e1', { fallbackLabel: '' }); + assert.equal(exact.kind, 'resolved'); + if (exact.kind !== 'resolved') throw new Error('unreachable'); + assert.equal(exact.resolved.node.label, 'Continue'); + assert.deepEqual(exact.resolved.resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'exact', + }); + + const recovered = tryResolveRefNode(nodes, '@e9', { fallbackLabel: 'Continue' }); + assert.equal(recovered.kind, 'resolved'); + if (recovered.kind !== 'resolved') throw new Error('unreachable'); + assert.equal(recovered.resolved.node.label, 'Continue'); + assert.deepEqual(recovered.resolved.resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'label-fallback', + }); + + assert.deepEqual(tryResolveRefNode(nodes, '@e9', { fallbackLabel: '' }), { kind: 'missing' }); +}); + +test('tryResolveRefNode tells a listed node without a usable centre from a missing one', () => { + const unusable = makeSnapshotState([ + { index: 0, depth: 0, type: 'Button', label: 'Continue', hittable: true }, + ]).nodes; + + const byRef = tryResolveRefNode(unusable, '@e1', { fallbackLabel: '' }); + assert.equal(byRef.kind, 'unusable'); + if (byRef.kind !== 'unusable') throw new Error('unreachable'); + assert.equal(byRef.node.label, 'Continue'); + + const byLabel = tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Continue' }); + assert.equal( + byLabel.kind, + 'unusable', + 'the trailing-label recovery found the node, so it is not missing', + ); + + assert.deepEqual(tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Elsewhere' }), { + kind: 'missing', + }); +}); diff --git a/src/commands/interaction/runtime/ref-target-resolution.ts b/src/commands/interaction/runtime/ref-target-resolution.ts new file mode 100644 index 0000000000..cd4333d7fa --- /dev/null +++ b/src/commands/interaction/runtime/ref-target-resolution.ts @@ -0,0 +1,262 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { + SnapshotKeyboardBandFact, + SnapshotNode, + SnapshotState, +} from '@agent-device/kernel/snapshot'; +import { findNodeByRef, normalizeRef } from '@agent-device/kernel/snapshot'; +import { resolveRectCenter } from '@agent-device/kernel/rect-center'; +import type { + AgentDeviceRuntime, + CommandContext, + CommandSessionRecord, +} from '../../../runtime-contract.ts'; +import { STALE_REF_HINT } from '@agent-device/selectors'; +import type { InteractionSnapshot } from './interaction-snapshot-capture.ts'; +import { requireSnapshotSession } from './selector-read-shared.ts'; +import { findNodeByLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; +import { surfaceScopedNodes } from './post-action-surface.ts'; +import type { + InteractionTarget, + PreresolvedInteractionTarget, + ResolvedInteractionTarget, + SurfaceScopedNodes, +} from '@agent-device/contracts/interaction'; +import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { localIdentitiesEqual, readNodeLocalIdentity } from '@agent-device/ad-script'; +import type { ResolveInteractionTargetParams } from './interaction-resolution-request.ts'; +import { assertReplayTargetResolution } from './replay-target-guard.ts'; +import { + buildRefResolution, + describeResolvedInteractionNode, + type ResolvedRefNode, +} from './resolution-disclosure.ts'; +import { + assertVisibleRefTarget, + runInteractionPipelineStages, +} from './target-visibility-stages.ts'; +import { resolveNodeTouchPoint } from './resolution-touch-point.ts'; + +/** The node a ref target acts on, plus the tree the shared guards read it against. */ +type RefResolution = { + tree: SurfaceScopedNodes; + resolved: ResolvedRefNode; + /** The keyboard band the capture of `tree.nodes` measured, when it measured one (#2660). */ + keyboard?: SnapshotKeyboardBandFact; +}; + +/** + * #1654: adopt the node the caller already resolved instead of resolving the + * same `@ref` a second time. This replaces the LOOKUP only — every guard below + * still runs, against the caller's tree, at the symbols the ADR 0011 + * `runtime-ref` cells name. + * + * `exact` is truthful only when all three pieces of carried provenance agree: + * the positional ref, the payload ref, and the node's own ref. Fail closed if + * future internal plumbing lets them drift. + */ +function adoptPreresolvedRefTarget( + target: Extract, + preresolved: PreresolvedInteractionTarget, +): RefResolution { + const ref = normalizeRef(target.ref); + if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${target.ref}`); + const carriedRef = normalizeRef(preresolved.ref); + const nodeRef = preresolved.node.ref ? normalizeRef(preresolved.node.ref) : null; + if (carriedRef !== ref || nodeRef !== ref || !preresolved.nodes.includes(preresolved.node)) { + throw new AppError( + 'COMMAND_FAILED', + 'Internal find target provenance does not match the interaction ref', + ); + } + return { + tree: { + nodes: preresolved.nodes, + ...(preresolved.iosSystemSurfaceBundleId + ? { surfaceBundleId: preresolved.iosSystemSurfaceBundleId } + : {}), + }, + ...(preresolved.keyboard ? { keyboard: preresolved.keyboard } : {}), + resolved: buildRefResolution(ref, preresolved.node, 'exact'), + }; +} + +async function readRefResolution( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, +): Promise { + const capture = await resolveSnapshotForRef(runtime, options, target); + return { + tree: surfaceScopedNodes(capture.snapshot), + ...(capture.snapshot.keyboard ? { keyboard: capture.snapshot.keyboard } : {}), + resolved: capture.resolved, + }; +} + +export async function resolveRefInteractionTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, + params: ResolveInteractionTargetParams, +): Promise { + const { tree, keyboard, resolved } = params.preresolvedTarget + ? adoptPreresolvedRefTarget(target, params.preresolvedTarget) + : await readRefResolution(runtime, options, target); + const nodes = tree.nodes; + // #1542: point/response read from the returned (possibly rescue-patched) node. + const { node: visibleNode, tapPoint: point } = await runInteractionPipelineStages({ + policy: params.pipeline, + nodes, + ...(keyboard ? { keyboard } : {}), + node: resolved.node, + action: params.action, + label: `Ref ${target.ref}`, + hooks: { + onResolved: (node, tree) => assertReplayTargetResolution(node, tree, params), + offscreen: async (node, tree) => + await assertVisibleRefTarget(runtime, options, node, tree, target.ref, params), + }, + resolveTapPoint: (node) => + resolveNodeTouchPoint(node, nodes, { + invalidMessage: `Ref ${target.ref} has no usable bounds`, + blockedTargetLabel: `Ref ${target.ref}`, + blockedTargetDetails: { ref: `@${normalizeRef(target.ref) ?? node.ref}` }, + }), + }); + return { + kind: 'ref', + point, + target: { kind: 'ref', ref: `@${resolved.ref}` }, + ...describeResolvedInteractionNode( + runtime, + visibleNode, + tree, + params.action, + resolved.resolution, + ), + }; +} + +async function resolveSnapshotForRef( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, +): Promise { + const { session, snapshot: frameTree } = await requireSnapshotSession(runtime, options.session); + + const fallbackLabel = target.fallbackLabel ?? ''; + const outcome = tryResolveRefNode(frameTree.nodes, target.ref, { + fallbackLabel, + }); + // ADR 0014: missing authorized-frame evidence FAILS. It must not fall through + // to a fresh capture and accept the same ref body from a newer tree — that is + // exactly the positional-coincidence retarget the frame model forbids. A stale + // read is observable and recoverable; a stale mutation can act on the wrong + // element. The caller re-observes (snapshot) or uses a selector. + if (outcome.kind !== 'resolved') throw refMissRefusal(outcome, target.ref); + return reconcileFreshObservation({ + session, + frameTree, + target, + fallbackLabel, + authorized: outcome.resolved, + }); +} + +/** + * ADR 0014 step 5: decouple Android freshness from ref authorization. The frame + * tree names WHICH node `@eN` authorizes. When a freshness (or other read-only) + * capture has advanced the operational observation past the frame, adopt the + * observation's node — its fresh on-screen coordinates — ONLY when its local + * identity still matches the authorized node. That covers the legitimate case of + * an element that merely moved. If the identity differs (a different element now + * sits at that index) or the ref is absent from the observation, keep the + * authorized frame node so a positional coincidence cannot retarget the action. + */ +function reconcileFreshObservation(params: { + session: CommandSessionRecord; + frameTree: SnapshotState; + target: Extract; + fallbackLabel: string; + authorized: ResolvedRefNode; +}): InteractionSnapshot & { resolved: ResolvedRefNode } { + const { session, frameTree, target, fallbackLabel, authorized } = params; + const observation = session.snapshot; + if (!observation || observation === frameTree) { + return { snapshot: frameTree, resolved: authorized }; + } + const observed = tryResolveRefNode(observation.nodes, target.ref, { fallbackLabel }); + if ( + observed.kind === 'resolved' && + localIdentitiesEqual( + readNodeLocalIdentity(authorized.node), + readNodeLocalIdentity(observed.resolved.node), + ) + ) { + return { snapshot: observation, resolved: observed.resolved }; + } + return { snapshot: frameTree, resolved: authorized }; +} + +/** The runtime-ref resolver: `exact` for a resolved `@ref`, `label-fallback` for trailing-label recovery. */ +/** + * What one tree makes of a ref: the node it authorizes (exact, or the trailing-label recovery), a + * node it lists (by ref or by that label) that has no usable centre, or no node at all. The two + * misses are distinct outcomes so a caller can name a stale ref and an unactionable target apart. + */ +export type RefResolutionOutcome = + | { kind: 'resolved'; resolved: ResolvedRefNode } + | { kind: 'unusable'; node: SnapshotNode } + | { kind: 'missing' }; + +export function tryResolveRefNode( + nodes: SnapshotState['nodes'], + refInput: string, + options: { + fallbackLabel: string; + }, +): RefResolutionOutcome { + const ref = normalizeRef(refInput); + if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${refInput}`); + const refNode = findNodeByRef(nodes, ref); + if (isUsableResolvedNode(refNode)) { + return { kind: 'resolved', resolved: buildRefResolution(ref, refNode, 'exact') }; + } + const fallbackNode = + options.fallbackLabel.length > 0 ? findNodeByLabel(nodes, options.fallbackLabel) : null; + if (isUsableResolvedNode(fallbackNode)) { + return { kind: 'resolved', resolved: buildRefResolution(ref, fallbackNode, 'label-fallback') }; + } + const found = refNode ?? fallbackNode; + return found ? { kind: 'unusable', node: found } : { kind: 'missing' }; +} + +/** + * The refusal for a ref the frame could not authorize: a ref no node carries is stale or was never + * issued (`ref_not_found`); a ref whose node is listed but has no usable centre is present and + * unactionable (`target_bounds_invalid`). Both recover the same way, a fresh observation, so both + * carry the stale-ref hint; `details.ref` is the bare ref body either way. + */ +function refMissRefusal( + miss: Exclude, + refInput: string, +): AppError { + const ref = normalizeRef(refInput) ?? refInput; + return miss.kind === 'unusable' + ? new AppError('COMMAND_FAILED', `Ref ${refInput} has no usable bounds`, { + reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, + ref, + hint: STALE_REF_HINT, + }) + : new AppError('COMMAND_FAILED', `Ref ${refInput} not found`, { + reason: INTERACTION_ERROR_REASONS.refNotFound, + ref, + hint: STALE_REF_HINT, + }); +} + +function isUsableResolvedNode(node: SnapshotNode | null | undefined): node is SnapshotNode { + if (!node) return false; + return resolveRectCenter(node.rect) !== null; +} diff --git a/src/commands/interaction/runtime/replay-target-guard.ts b/src/commands/interaction/runtime/replay-target-guard.ts new file mode 100644 index 0000000000..ba4cff4ece --- /dev/null +++ b/src/commands/interaction/runtime/replay-target-guard.ts @@ -0,0 +1,65 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotNode, SnapshotState } from '@agent-device/kernel/snapshot'; +import { + localIdentitiesEqual, + readNodeLocalIdentity, + readNodeStructuralDenotation, + structuralDenotationsEqual, +} from '@agent-device/ad-script'; +import { REPLAY_TARGET_GUARD_MISMATCH_REASON } from '@agent-device/contracts/replay'; +import type { + ExpectedResolvedTarget, + ResolveInteractionTargetParams, +} from './interaction-resolution-request.ts'; + +/** + * Compares the resolution winner (pre-promotion: hittable-ancestor promotion + * deliberately retargets to the same LEAF's actionable container and must not + * trip the guard — duplicates are distinct leaves with distinct structural + * denotations, so comparing the leaf is exactly right) against the verified + * member's local identity AND structural denotation; throws pre-action when + * EITHER differs. + */ +export function assertExpectedResolvedTarget( + node: SnapshotNode, + nodes: SnapshotState['nodes'], + expected: ExpectedResolvedTarget | undefined, + action: string, + targetRole?: 'source' | 'destination', +): void { + if (!expected) return; + const observedIdentity = readNodeLocalIdentity(node); + const observedStructural = readNodeStructuralDenotation(node, nodes); + if ( + localIdentitiesEqual(observedIdentity, expected.identity) && + structuralDenotationsEqual(observedStructural, expected.structural) + ) { + return; + } + throw new AppError( + 'COMMAND_FAILED', + `${action} resolved to a different element than replay verification isolated; the action was not sent`, + { + reason: REPLAY_TARGET_GUARD_MISMATCH_REASON, + observed: observedIdentity, + observedStructural, + expected: expected.identity, + expectedStructural: expected.structural, + ...(targetRole ? { targetRole } : {}), + }, + ); +} + +export function assertReplayTargetResolution( + node: SnapshotNode, + nodes: SnapshotState['nodes'], + params: ResolveInteractionTargetParams, +): void { + assertExpectedResolvedTarget( + node, + nodes, + params.expectedResolvedTarget, + params.action, + params.replayTargetRole, + ); +} diff --git a/src/commands/interaction/runtime/resolution-disclosure.test.ts b/src/commands/interaction/runtime/resolution-disclosure.test.ts new file mode 100644 index 0000000000..bb3daff2d7 --- /dev/null +++ b/src/commands/interaction/runtime/resolution-disclosure.test.ts @@ -0,0 +1,19 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { buildRefResolution } from './resolution-disclosure.ts'; +import { selectorSnapshot } from './__tests__/test-utils/index.ts'; + +test('buildRefResolution is the shared exact and label-fallback disclosure constructor', () => { + const node = selectorSnapshot().nodes[0]!; + + assert.deepEqual(buildRefResolution('e1', node, 'exact').resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'exact', + }); + assert.deepEqual(buildRefResolution('e1', node, 'label-fallback').resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'label-fallback', + }); +}); diff --git a/src/commands/interaction/runtime/resolution-disclosure.ts b/src/commands/interaction/runtime/resolution-disclosure.ts new file mode 100644 index 0000000000..34b25c298e --- /dev/null +++ b/src/commands/interaction/runtime/resolution-disclosure.ts @@ -0,0 +1,179 @@ +import type { SnapshotNode, SnapshotState } from '@agent-device/kernel/snapshot'; +import type { AgentDeviceRuntime } from '../../../runtime-contract.ts'; +import { type SelectorResolution, buildSelectorChainForNode } from '@agent-device/selectors'; +import { resolvePressRecordingTarget } from '@agent-device/selectors/press-retarget'; +import { resolveRefLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; +import { normalizeType } from '@agent-device/contracts/snapshot'; +import { truncateUtf8 } from './truncate-utf8.ts'; +import type { + RecordingTargetOverride, + ResolutionDiagnosticEntry, + ResolutionDisclosure, + SurfaceScopedNodes, +} from '@agent-device/contracts/interaction'; +import type { InteractionAction } from './interaction-resolution-request.ts'; + +export type ResolvedRefNode = { + ref: string; + node: SnapshotNode; + resolution: ResolutionDisclosure; +}; + +// ADR 0012 decision 2 bounds: diagnostic strings and losing alternatives. +const RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP = 256; +const MAX_RESOLUTION_ALTERNATIVES = 5; + +/** + * A successful `@ref` lookup names exactly one node; label recovery discloses label-fallback instead. + * ADR 0011 registry anchor: interaction-guarantees.ts cites it as a `via` symbol. + */ +export const EXACT_REF_RESOLUTION: ResolutionDisclosure = { + source: 'ref', + phase: 'pre-action', + kind: 'exact', +}; + +const LABEL_FALLBACK_REF_RESOLUTION: ResolutionDisclosure = { + source: 'ref', + phase: 'pre-action', + kind: 'label-fallback', +}; + +/** Shared construction site for every runtime-ref resolution disclosure. */ +export function buildRefResolution( + ref: string, + node: SnapshotNode, + kind: 'exact' | 'label-fallback', +): ResolvedRefNode { + return { + ref, + node, + resolution: kind === 'exact' ? EXACT_REF_RESOLUTION : LABEL_FALLBACK_REF_RESOLUTION, + }; +} + +const UNIQUE_RUNTIME_RESOLUTION: ResolutionDisclosure = { + source: 'runtime', + phase: 'pre-action', + kind: 'unique', +}; + +// Disclosure only: the winner stays resolveSelectorChain's pick (ADR 0012). +export function buildSelectorResolutionDisclosure( + resolved: SelectorResolution, + nodes: SnapshotState['nodes'], +): ResolutionDisclosure { + if (!resolved.disambiguation) return UNIQUE_RUNTIME_RESOLUTION; + return { + source: 'runtime', + phase: 'pre-action', + kind: 'disambiguated', + matchCount: resolved.disambiguation.matchCount, + winnerDiagnostic: buildResolutionDiagnosticEntry(resolved.node, nodes), + tiebreak: resolved.disambiguation.tiebreak, + alternatives: resolved.disambiguation.alternatives + .slice(0, MAX_RESOLUTION_ALTERNATIVES) + .map((node) => buildResolutionDiagnosticEntry(node, nodes)), + }; +} + +function buildResolutionDiagnosticEntry( + node: SnapshotNode, + nodes: SnapshotState['nodes'], +): ResolutionDiagnosticEntry { + const role = normalizeType(node.type ?? ''); + const label = resolveRefLabel(node, nodes); + return { + diagnosticRef: `diag-${node.ref}`, + ...(role ? { role: truncateUtf8(role, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } : {}), + ...(label !== undefined + ? { label: truncateUtf8(label, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } + : {}), + }; +} + +// Shared tail of a resolved ref/selector interaction target: the node itself +// plus everything derived from it for the response. Every response field +// describes the DISPATCHED node — the #1280 retarget rides only on the +// `recordingTarget` side channel below. `tree` is the capture the node was +// resolved from, and becomes the pre-action baseline this publishes. +export function describeResolvedInteractionNode( + runtime: AgentDeviceRuntime, + node: SnapshotNode, + tree: SurfaceScopedNodes, + action: InteractionAction, + resolution: ResolutionDisclosure, +): { + node: SnapshotNode; + selectorChain: string[]; + refLabel: string | undefined; + targetHittable?: boolean; + hint?: string; + preAction: SurfaceScopedNodes; + resolution: ResolutionDisclosure; + recordingTarget?: RecordingTargetOverride; +} { + const nodes = tree.nodes; + return { + node, + selectorChain: buildSelectorChainForNode(node, runtime.backend.platform, { + action: action === 'fill' ? 'fill' : 'click', + nodes, + }), + refLabel: resolveRefLabel(node, nodes), + ...describeNonHittableTarget(node, action), + preAction: tree, + resolution, + ...pressRecordingTargetOverride(runtime, node, nodes, action), + }; +} + +/** + * #1280 (ADR 0012 decision 3 amendment): the recording-only side channel. + * When a click/press resolves to an identity-empty container, the RECORDED + * step retargets to its first labeled descendant — node, chain, and + * ref-label computed together here so the recorded action entry and its + * `target-v1` evidence can never half-retarget. The response payloads never + * consume this (see `interaction-touch-response.ts`). `fill` is deliberately + * excluded: its chain carries `editable=true` constraints a label descendant + * cannot satisfy, which would record an unreplayable selector. + */ +function pressRecordingTargetOverride( + runtime: AgentDeviceRuntime, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + action: InteractionAction, +): { recordingTarget?: RecordingTargetOverride } { + if (action !== 'click' && action !== 'press') return {}; + const recordingNode = resolvePressRecordingTarget(node, nodes); + if (recordingNode === node) return {}; + return { + recordingTarget: { + node: recordingNode, + selectorChain: buildSelectorChainForNode(recordingNode, runtime.backend.platform, { + action: 'click', + nodes, + }), + refLabel: resolveRefLabel(recordingNode, nodes), + }, + }; +} + +/** + * iOS AX `hittable` flags are unreliable on deep React Native trees (see #1037: + * a map-pin annotation exact-matched a longer recents row label and reported tap + * success while doing nothing visible). We deliberately do NOT fail or filter on + * this signal — that would break selectors that only ever resolve to nodes the + * platform marks non-hittable. Instead, surface it so the caller can notice a + * likely no-op tap and re-target with a ref or a more specific selector/longer text. + */ +export function describeNonHittableTarget( + node: SnapshotNode, + action: InteractionAction, +): { targetHittable?: boolean; hint?: string } { + if (node.hittable !== false) return {}; + return { + targetHittable: false, + hint: `The resolved element reports hittable: false, so this ${action} may have had no visible effect. Verify with a snapshot, or prefer a @ref or a longer/more specific selector to target the intended element.`, + }; +} diff --git a/src/commands/interaction/runtime/resolution-touch-point.ts b/src/commands/interaction/runtime/resolution-touch-point.ts new file mode 100644 index 0000000000..6e4ab7bd6f --- /dev/null +++ b/src/commands/interaction/runtime/resolution-touch-point.ts @@ -0,0 +1,48 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { Point, SnapshotNode, SnapshotState } from '@agent-device/kernel/snapshot'; +import { normalizeRef } from '@agent-device/kernel/snapshot'; +import { createSnapshotVisibility } from '@agent-device/contracts/snapshot'; +import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { resolveInteractionTouchPoint } from '@agent-device/selectors/interaction-touch-point'; + +export function resolveNodeTouchPoint( + node: SnapshotNode, + nodes: SnapshotState['nodes'], + failure: { + invalidMessage: string; + blockedTargetLabel: string; + blockedTargetDetails: { ref: string } | { selector: string }; + }, +): Point { + const visibility = createSnapshotVisibility(nodes); + const effectiveViewport = visibility.resolveEffectiveViewport(node); + const rootViewport = node.rect ? visibility.resolveViewport(node.rect) : null; + const resolution = resolveInteractionTouchPoint(nodes, node, { + bounds: [effectiveViewport, rootViewport].filter((rect) => rect !== null), + }); + if (resolution.kind === 'resolved') return resolution.point; + if (resolution.kind === 'invalid') { + throw new AppError('COMMAND_FAILED', failure.invalidMessage, { + reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, + ...bareTargetDetails(failure.blockedTargetDetails), + }); + } + throw new AppError( + 'COMMAND_FAILED', + `${failure.blockedTargetLabel} has no parent-owned touch point outside its interactive descendants`, + { + reason: 'covered_by_interactive_descendants', + ...failure.blockedTargetDetails, + competitorRefs: resolution.competitorRefs.slice(0, 5).map((ref) => `@${ref}`), + competitorCount: resolution.competitorRefs.length, + hint: 'Tap the specific interactive child you intend, or use a more specific selector. Every safely tappable region of the parent belongs to one of its child controls.', + }, + ); +} + +/** `details.ref` is the bare ref body on every reason; the blocked-target label keeps its `@`. */ +function bareTargetDetails( + details: { ref: string } | { selector: string }, +): { ref: string } | { selector: string } { + return 'ref' in details ? { ref: normalizeRef(details.ref) ?? details.ref } : details; +} diff --git a/src/commands/interaction/runtime/resolution.test.ts b/src/commands/interaction/runtime/resolution.test.ts index 70d30fe144..166ac26f96 100644 --- a/src/commands/interaction/runtime/resolution.test.ts +++ b/src/commands/interaction/runtime/resolution.test.ts @@ -2,20 +2,13 @@ import assert from 'node:assert/strict'; import { test } from 'vitest'; import type { BackendSnapshotOptions } from '../../../backend.ts'; import { ref, selector } from './selector-read-utils.ts'; -import { - buildRefResolution, - throwIfOffscreenInteractionTarget, - tryResolveRefNode, -} from './resolution.ts'; -import { resolveRecordedTarget, STALE_REF_HINT } from '@agent-device/selectors'; +import { resolveRecordedTarget } from '@agent-device/selectors'; import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; import type { Point } from '@agent-device/kernel/snapshot'; import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; import { clickRefE2, - coveredByTabBarSnapshot, createInteractionDevice, - duplicateCoveredLabelSnapshot, fillableSnapshot, iosTabBarSnapshot, mapPinAnnotationSnapshot, @@ -103,100 +96,6 @@ test('runtime selector misses carry a structured reason for retrying adapters', ); }); -test('runtime press refuses a selector that resolves to an off-screen element', async () => { - // Closed-drawer shape: the only match sits fully left of the viewport. The - // @ref path already refuses this; the selector path must not silently tap - // out-of-viewport coordinates. - const offscreenSnapshot = makeSnapshotState([ - { - index: 0, - depth: 0, - type: 'Application', - rect: { x: 0, y: 0, width: 400, height: 800 }, - hittable: true, - }, - { - index: 1, - depth: 2, - parentIndex: 0, - type: 'Button', - label: 'Explore', - rect: { x: -320, y: 240, width: 300, height: 50 }, - hittable: true, - }, - ]); - const taps: unknown[] = []; - const device = createInteractionDevice(offscreenSnapshot, { - tap: async (_context, point) => { - taps.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.press(selector('label=Explore'), { session: 'default' }), - (error: unknown) => { - assert.ok(error instanceof Error); - assert.match(error.message, /off-screen element and is not safe to press/); - const details = (error as { details?: Record }).details; - assert.equal(details?.reason, 'offscreen_selector'); - // #1366: the closed-drawer shape sits fully left of the viewport, so the - // hint names `scroll left` and steers back through the same selector. - assert.equal(details?.scrollDirection, 'left'); - assert.match(String(details?.hint), /scroll left/i); - assert.match(String(details?.hint), /selector/i); - return true; - }, - ); - assert.equal(taps.length, 0); -}); - -test('runtime press names a direction for a partial clip whose center is off-screen', async () => { - // #1366 regression: the row still OVERLAPS the viewport (top edge inside), so - // the rect-vs-viewport form yields no direction — but its tap-point center is - // below the bottom edge, which is what the visibility guard rejects. The hint - // must still name `scroll down` rather than falling back to the generic phrasing. - const partialClipSnapshot = makeSnapshotState([ - { - index: 0, - depth: 0, - type: 'Application', - rect: { x: 0, y: 0, width: 400, height: 800 }, - hittable: true, - }, - { - index: 1, - depth: 2, - parentIndex: 0, - type: 'Button', - label: 'Cash', - rect: { x: 20, y: 790, width: 200, height: 44 }, - hittable: true, - }, - ]); - const taps: unknown[] = []; - const device = createInteractionDevice(partialClipSnapshot, { - tap: async (_context, point) => { - taps.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.press(selector('label=Cash'), { session: 'default' }), - (error: unknown) => { - const details = (error as { details?: Record }).details; - assert.equal(details?.reason, 'offscreen_selector'); - assert.equal(details?.scrollDirection, 'down'); - assert.match(String(details?.hint), /scroll down/i); - // #1366 recovery must be bounded. `--until` is what bounds it now: it checks the same - // selector between passes, so the hint names one command rather than a manual step loop. - assert.match(String(details?.hint), /scroll down --until 'label=Cash'/); - assert.match(String(details?.hint), /stops on the target/i); - return true; - }, - ); - assert.equal(taps.length, 0); -}); - test('runtime click keeps distinct tab button centers when iOS reports the tab bar as hittable', async () => { const calls: Point[] = []; const device = createInteractionDevice(iosTabBarSnapshot(), { @@ -222,38 +121,6 @@ test('runtime click keeps distinct tab button centers when iOS reports the tab b assert.equal(selectorResult.node?.label, 'Settings'); }); -test('runtime click rejects refs covered by floating overlays', async () => { - const calls: Point[] = []; - const device = createInteractionDevice(coveredByTabBarSnapshot(), { - tap: async (_context, point) => { - calls.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.click(ref('@e2'), { session: 'default' }), - /Ref @e2 is covered by another visible element/, - ); - assert.deepEqual(calls, []); -}); - -test('runtime selector interactions skip covered matches when an uncovered duplicate exists', async () => { - const calls: Point[] = []; - const device = createInteractionDevice(duplicateCoveredLabelSnapshot(), { - tap: async (_context, point) => { - calls.push(point); - }, - }); - - const result = await device.interactions.click(selector('label="Save draft"'), { - session: 'default', - }); - - assert.equal(result.kind, 'selector'); - assert.equal(result.node?.ref, 'e2'); - assert.deepEqual(calls, [{ x: 86, y: 142 }]); -}); - test('runtime click keeps non-button semantic targets at their own center', async () => { const calls: Point[] = []; const device = createInteractionDevice(nonHittableCellSnapshot(), { @@ -481,226 +348,3 @@ test('runtime interactions reject unsupported macOS desktop and menubar surfaces assert.equal(pressed, true); }); - -test('runtime ref interactions fail closed when the authorized ref has no usable bounds (ADR 0014)', async () => { - const staleSnapshot = makeSnapshotState([ - { - index: 0, - depth: 0, - type: 'Button', - label: 'Continue', - hittable: true, - }, - ]); - const calls: Point[] = []; - let captures = 0; - const device = createInteractionDevice(staleSnapshot, { - captureSnapshot: async () => { - captures += 1; - return { snapshot: selectorSnapshot() }; - }, - tap: async (_context, point) => { - calls.push(point); - }, - }); - - // ADR 0014: the authorized frame's @e1 has no usable rect, so it FAILS rather - // than recapturing and accepting the same index from a newer tree by - // positional coincidence. - await assert.rejects( - () => device.interactions.click(ref('@e1'), { session: 'default' }), - (error: unknown) => { - assert.match((error as Error).message, /Ref @e1 has no usable bounds/); - assert.deepEqual( - (error as { details?: Record }).details, - { reason: 'target_bounds_invalid', ref: 'e1', hint: STALE_REF_HINT }, - 'the frame lists @e1, so the refusal names the bounds, not a missing ref', - ); - return true; - }, - ); - assert.equal(captures, 0); - assert.deepEqual(calls, []); -}); - -test('runtime ref interactions refuse a ref the authorized frame does not list with ref_not_found', async () => { - const calls: Point[] = []; - let captures = 0; - const device = createInteractionDevice(selectorSnapshot(), { - captureSnapshot: async () => { - captures += 1; - return { snapshot: selectorSnapshot() }; - }, - tap: async (_context, point) => { - calls.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.click(ref('@e9'), { session: 'default' }), - (error: unknown) => { - assert.match((error as Error).message, /Ref @e9 not found/); - assert.deepEqual((error as { details?: Record }).details, { - reason: 'ref_not_found', - ref: 'e9', - hint: STALE_REF_HINT, - }); - return true; - }, - ); - assert.equal(captures, 0); - assert.deepEqual(calls, []); -}); - -test('tryResolveRefNode discloses exact for a resolved ref and label-fallback for label recovery', () => { - const nodes = selectorSnapshot().nodes; - - const exact = tryResolveRefNode(nodes, '@e1', { fallbackLabel: '' }); - assert.equal(exact.kind, 'resolved'); - if (exact.kind !== 'resolved') throw new Error('unreachable'); - assert.equal(exact.resolved.node.label, 'Continue'); - assert.deepEqual(exact.resolved.resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'exact', - }); - - const recovered = tryResolveRefNode(nodes, '@e9', { fallbackLabel: 'Continue' }); - assert.equal(recovered.kind, 'resolved'); - if (recovered.kind !== 'resolved') throw new Error('unreachable'); - assert.equal(recovered.resolved.node.label, 'Continue'); - assert.deepEqual(recovered.resolved.resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'label-fallback', - }); - - assert.deepEqual(tryResolveRefNode(nodes, '@e9', { fallbackLabel: '' }), { kind: 'missing' }); -}); - -test('tryResolveRefNode tells a listed node without a usable centre from a missing one', () => { - const unusable = makeSnapshotState([ - { index: 0, depth: 0, type: 'Button', label: 'Continue', hittable: true }, - ]).nodes; - - const byRef = tryResolveRefNode(unusable, '@e1', { fallbackLabel: '' }); - assert.equal(byRef.kind, 'unusable'); - if (byRef.kind !== 'unusable') throw new Error('unreachable'); - assert.equal(byRef.node.label, 'Continue'); - - const byLabel = tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Continue' }); - assert.equal( - byLabel.kind, - 'unusable', - 'the trailing-label recovery found the node, so it is not missing', - ); - - assert.deepEqual(tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Elsewhere' }), { - kind: 'missing', - }); -}); - -test('buildRefResolution is the shared exact and label-fallback disclosure constructor', () => { - const node = selectorSnapshot().nodes[0]!; - - assert.deepEqual(buildRefResolution('e1', node, 'exact').resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'exact', - }); - assert.deepEqual(buildRefResolution('e1', node, 'label-fallback').resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'label-fallback', - }); -}); - -// #1542: throwIfOffscreenInteractionTarget is exported for ADR 0011 registry -// honesty (interaction-guarantees.ts's `offscreen` cells point their `via` -// here); this direct-import test is its real consumer, mirroring -// tryResolveRefNode above. End-to-end rescue/refuse coverage through the -// public click/press surface lives in offscreen-double-check.test.ts. -function fakeOffscreenFailure() { - return { - message: 'off-screen', - details: { reason: 'test' }, - hint: () => 'scroll toward it', - }; -} - -test('throwIfOffscreenInteractionTarget: an on-screen node passes through unchanged', async () => { - const device = createInteractionDevice(makeSnapshotState([])); - const nodes = makeSnapshotState([ - { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, - { - index: 1, - depth: 1, - parentIndex: 0, - type: 'Button', - rect: { x: 20, y: 20, width: 40, height: 40 }, - }, - ]).nodes; - - const result = await throwIfOffscreenInteractionTarget( - device, - { session: 'default' }, - nodes[1]!, - nodes, - fakeOffscreenFailure(), - ); - - assert.equal(result, nodes[1]); -}); - -test('throwIfOffscreenInteractionTarget: off-screen + backend confirms -> returns the node patched with the LIVE rect', async () => { - const device = createInteractionDevice(makeSnapshotState([]), { - confirmOffscreenTargetVisible: async () => ({ x: 30, y: 30, width: 40, height: 40 }), - }); - const nodes = makeSnapshotState([ - { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, - { - index: 1, - depth: 1, - parentIndex: 0, - type: 'Button', - rect: { x: 20, y: 2000, width: 40, height: 40 }, - }, - ]).nodes; - - const result = await throwIfOffscreenInteractionTarget( - device, - { session: 'default' }, - nodes[1]!, - nodes, - fakeOffscreenFailure(), - ); - - assert.deepEqual(result.rect, { x: 30, y: 30, width: 40, height: 40 }); - assert.equal(result.index, nodes[1]!.index); -}); - -test('throwIfOffscreenInteractionTarget: off-screen + no rescue -> throws with the supplied failure shape', async () => { - const device = createInteractionDevice(makeSnapshotState([])); - const nodes = makeSnapshotState([ - { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, - { - index: 1, - depth: 1, - parentIndex: 0, - type: 'Button', - rect: { x: 20, y: 2000, width: 40, height: 40 }, - }, - ]).nodes; - - await assert.rejects( - () => - throwIfOffscreenInteractionTarget( - device, - { session: 'default' }, - nodes[1]!, - nodes, - fakeOffscreenFailure(), - ), - /off-screen/, - ); -}); diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index 48bf82e61b..a141a486ee 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -1,193 +1,44 @@ import { AppError } from '@agent-device/kernel/errors'; -import type { - Point, - SnapshotKeyboardBandFact, - SnapshotNode, - SnapshotState, -} from '@agent-device/kernel/snapshot'; -import { - findNodeByRef, - inheritPostGestureOutcome, - normalizeRef, -} from '@agent-device/kernel/snapshot'; -import { resolveRectCenter } from '@agent-device/kernel/rect-center'; -import type { - AgentDeviceRuntime, - CommandContext, - CommandSessionRecord, -} from '../../../runtime-contract.ts'; -import { - formatSelectorFailure, - selectorFailureHint, - STALE_REF_HINT, - type SelectorResolution, - buildSelectorChainForNode, -} from '@agent-device/selectors'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import type { SelectorResolution } from '@agent-device/selectors'; +import { readinessScheduleFor } from '@agent-device/selectors/selector-pipeline-policy'; import { - resolveSelectorPipeline, - runNodePipelineStages, - type SelectorPipelineHooks, -} from '@agent-device/selectors/selector-pipeline'; + attemptSelectorResolution, + pollForSelectorReadiness, + selectorInteractionFailure, +} from './selector-readiness.ts'; import { - SELECTOR_PIPELINE_POLICIES, - type ActingPipelinePolicy, - type SelectorPipelinePolicy, -} from '@agent-device/selectors/selector-pipeline-policy'; -import { resolvePressRecordingTarget } from '@agent-device/selectors/press-retarget'; -import { requireSnapshotSession } from './selector-read-shared.ts'; -import { findNodeByLabel, resolveRefLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; + captureInteractionSnapshot, + type InteractionSnapshot, +} from './interaction-snapshot-capture.ts'; import { containsPoint } from '@agent-device/kernel/rect'; -import { createSnapshotVisibility, normalizeType } from '@agent-device/contracts/snapshot'; -import { - classifyOffscreenScrollDirection, - type OffscreenScrollDirection, -} from '@agent-device/capture-kit/mobile-snapshot-semantics'; -import { truncateUtf8 } from './truncate-utf8.ts'; +import { createSnapshotVisibility } from '@agent-device/contracts/snapshot'; import { surfaceScopedNodes } from './post-action-surface.ts'; import type { InteractionTarget, PointTarget, - PreresolvedInteractionTarget, - RecordingTargetOverride, - ResolutionDiagnosticEntry, - ResolutionDisclosure, ResolvedInteractionTarget, SurfaceScopedNodes, } from '@agent-device/contracts/interaction'; -import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { describeKeyboardOccludedPointWarning } from './keyboard-occlusion.ts'; +import { assertReplayTargetResolution } from './replay-target-guard.ts'; import type { - BackendActionResult, - BackendCommandContext, - BackendRefTarget, -} from '../../../backend.ts'; -import { now, toBackendContext } from '../../runtime-common.ts'; -import { toBackendResult } from '../../runtime-types.ts'; -import { resolveInteractionTouchPoint } from '@agent-device/selectors/interaction-touch-point'; -import { - localIdentitiesEqual, - readNodeLocalIdentity, - readNodeStructuralDenotation, - structuralDenotationsEqual, -} from '@agent-device/ad-script'; + InteractionAction, + ResolveInteractionTargetParams, +} from './interaction-resolution-request.ts'; +import { resolveRefInteractionTarget } from './ref-target-resolution.ts'; import { - REPLAY_TARGET_GUARD_MISMATCH_REASON, - type ReplayTargetGuardDenotation, -} from '@agent-device/contracts/replay'; -import { resolveActionSelector } from './selector-action-resolution.ts'; + buildSelectorResolutionDisclosure, + describeResolvedInteractionNode, +} from './resolution-disclosure.ts'; import { - assertTapTargetClearOfVisibleKeyboard, - describeKeyboardOccludedPointWarning, -} from './keyboard-occlusion.ts'; -import { interactionVerb } from './interaction-verb.ts'; + assertVisibleSelectorTarget, + runInteractionPipelineStages, +} from './target-visibility-stages.ts'; +import { resolveNodeTouchPoint } from './resolution-touch-point.ts'; export type { InteractionTarget, ResolvedInteractionTarget }; -/** - * ADR 0012 migration step 4, post-resolution guard: the LOCAL identity AND - * the STRUCTURAL denotation (pre-order document index + same-parent sibling - * ordinal) of the element replay's pre-action verification isolated. Set ONLY - * by the replay step loop (via `DaemonRequest.internal.replayTargetGuard`) for - * annotated verified actions — never on live interactive commands. - * - * Local identity alone is insufficient: ADR path 6 isolates ONE member among - * several nodes that share the same `{id, role, label}` using sibling / - * region-scoped viewportOrder. If verification isolates duplicate A but - * dispatch's occlusion/visibility filtering selects duplicate B with the same - * local identity, a local-identity-only guard would pass and tap the wrong - * element. The structural denotation is the discriminator that catches that - * split BEFORE the device action. - */ -export type ExpectedResolvedTarget = ReplayTargetGuardDenotation; - -/** - * Compares the resolution winner (pre-promotion: hittable-ancestor promotion - * deliberately retargets to the same LEAF's actionable container and must not - * trip the guard — duplicates are distinct leaves with distinct structural - * denotations, so comparing the leaf is exactly right) against the verified - * member's local identity AND structural denotation; throws pre-action when - * EITHER differs. - */ -export function assertExpectedResolvedTarget( - node: SnapshotNode, - nodes: SnapshotState['nodes'], - expected: ExpectedResolvedTarget | undefined, - action: string, - targetRole?: 'source' | 'destination', -): void { - if (!expected) return; - const observedIdentity = readNodeLocalIdentity(node); - const observedStructural = readNodeStructuralDenotation(node, nodes); - if ( - localIdentitiesEqual(observedIdentity, expected.identity) && - structuralDenotationsEqual(observedStructural, expected.structural) - ) { - return; - } - throw new AppError( - 'COMMAND_FAILED', - `${action} resolved to a different element than replay verification isolated; the action was not sent`, - { - reason: REPLAY_TARGET_GUARD_MISMATCH_REASON, - observed: observedIdentity, - observedStructural, - expected: expected.identity, - expectedStructural: expected.structural, - ...(targetRole ? { targetRole } : {}), - }, - ); -} - -export type InteractionAction = - | 'click' - | 'press' - | 'fill' - | 'focus' - | 'longPress' - | 'hover' - | 'scroll' - | 'swipe' - | 'pinch' - | 'pan' - | 'drag' - | 'fling' - | 'rotate' - | 'transform'; - -export type InteractionSnapshot = { - snapshot: SnapshotState; -}; - -type ResolveInteractionTargetParams = { - action: InteractionAction; - requireInteractive: boolean; - /** - * The structural pipeline this action runs (#1656): occlusion, off-screen, - * and hittable-ancestor promotion are the row's decisions. `promotedTarget` - * for tap-shaped actions, `resolvedTarget` for the actions that must keep - * the element they resolved. - */ - pipeline: ActingPipelinePolicy; - /** - * `--verify` (#1047): also capture the pre-action node set for a `point` target - * so `changedFromBefore` evidence has a baseline. Ref/selector targets already - * capture a snapshot to resolve the target, so this is a no-op cost for them — - * their nodes are attached below regardless of this flag. For point targets, - * which normally skip capture entirely, this opts into one extra capture, only - * when the caller explicitly asked for verify evidence. Defaults to false. - */ - captureEvidenceBaseline?: boolean; - /** ADR 0012 step 4 post-resolution guard; see `ExpectedResolvedTarget`. */ - expectedResolvedTarget?: ExpectedResolvedTarget; - /** Identifies one endpoint when a multi-target replay guard refuses. */ - replayTargetRole?: 'source' | 'destination'; - /** - * #1654: the caller already resolved this `@ref` against its own capture, so - * the ref branch adopts that node instead of looking the ref up again. Ref - * targets only — a selector target has nothing pre-resolved to adopt. - */ - preresolvedTarget?: PreresolvedInteractionTarget; -}; - export async function resolveInteractionTarget( runtime: AgentDeviceRuntime, options: CommandContext & { target: InteractionTarget }, @@ -278,107 +129,6 @@ async function tryCaptureEvidenceBaseline( } } -/** The node a ref target acts on, plus the tree the shared guards read it against. */ -type RefResolution = { - tree: SurfaceScopedNodes; - resolved: ResolvedRefNode; - /** The keyboard band the capture of `tree.nodes` measured, when it measured one (#2660). */ - keyboard?: SnapshotKeyboardBandFact; -}; - -/** - * #1654: adopt the node the caller already resolved instead of resolving the - * same `@ref` a second time. This replaces the LOOKUP only — every guard below - * still runs, against the caller's tree, at the symbols the ADR 0011 - * `runtime-ref` cells name. - * - * `exact` is truthful only when all three pieces of carried provenance agree: - * the positional ref, the payload ref, and the node's own ref. Fail closed if - * future internal plumbing lets them drift. - */ -function adoptPreresolvedRefTarget( - target: Extract, - preresolved: PreresolvedInteractionTarget, -): RefResolution { - const ref = normalizeRef(target.ref); - if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${target.ref}`); - const carriedRef = normalizeRef(preresolved.ref); - const nodeRef = preresolved.node.ref ? normalizeRef(preresolved.node.ref) : null; - if (carriedRef !== ref || nodeRef !== ref || !preresolved.nodes.includes(preresolved.node)) { - throw new AppError( - 'COMMAND_FAILED', - 'Internal find target provenance does not match the interaction ref', - ); - } - return { - tree: { - nodes: preresolved.nodes, - ...(preresolved.iosSystemSurfaceBundleId - ? { surfaceBundleId: preresolved.iosSystemSurfaceBundleId } - : {}), - }, - ...(preresolved.keyboard ? { keyboard: preresolved.keyboard } : {}), - resolved: buildRefResolution(ref, preresolved.node, 'exact'), - }; -} - -async function readRefResolution( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, -): Promise { - const capture = await resolveSnapshotForRef(runtime, options, target); - return { - tree: surfaceScopedNodes(capture.snapshot), - ...(capture.snapshot.keyboard ? { keyboard: capture.snapshot.keyboard } : {}), - resolved: capture.resolved, - }; -} - -async function resolveRefInteractionTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, - params: ResolveInteractionTargetParams, -): Promise { - const { tree, keyboard, resolved } = params.preresolvedTarget - ? adoptPreresolvedRefTarget(target, params.preresolvedTarget) - : await readRefResolution(runtime, options, target); - const nodes = tree.nodes; - // #1542: point/response read from the returned (possibly rescue-patched) node. - const { node: visibleNode, tapPoint: point } = await runInteractionPipelineStages({ - policy: params.pipeline, - nodes, - ...(keyboard ? { keyboard } : {}), - node: resolved.node, - action: params.action, - label: `Ref ${target.ref}`, - hooks: { - onResolved: (node, tree) => assertReplayTargetResolution(node, tree, params), - offscreen: async (node, tree) => - await assertVisibleRefTarget(runtime, options, node, tree, target.ref, params), - }, - resolveTapPoint: (node) => - resolveNodeTouchPoint(node, nodes, { - invalidMessage: `Ref ${target.ref} has no usable bounds`, - blockedTargetLabel: `Ref ${target.ref}`, - blockedTargetDetails: { ref: `@${normalizeRef(target.ref) ?? node.ref}` }, - }), - }); - return { - kind: 'ref', - point, - target: { kind: 'ref', ref: `@${resolved.ref}` }, - ...describeResolvedInteractionNode( - runtime, - visibleNode, - tree, - params.action, - resolved.resolution, - ), - }; -} - async function resolveSelectorInteractionTarget( runtime: AgentDeviceRuntime, options: CommandContext, @@ -386,34 +136,34 @@ async function resolveSelectorInteractionTarget( params: ResolveInteractionTargetParams, ): Promise { const selectorExpression = target.selector; - let capture = await captureInteractionSnapshot(runtime, options, params.requireInteractive); - let resolved = resolveActionSelector( - capture.snapshot.nodes, - selectorExpression, - runtime.backend.platform, - params.pipeline, - ); - if ((!resolved || !resolved.node.rect) && params.requireInteractive) { - const interactive = capture.snapshot; - capture = await captureInteractionSnapshot(runtime, options, false); - inheritPostGestureOutcome(interactive, capture.snapshot); - resolved = resolveActionSelector( - capture.snapshot.nodes, - selectorExpression, - runtime.backend.platform, - params.pipeline, - ); - } - if (!resolved || !resolved.node.rect) { - throw await selectorInteractionFailure({ + let capture: InteractionSnapshot; + let resolved: SelectorResolution | null; + const readinessSchedule = readinessScheduleFor(params.pipeline.poll, params.readinessTimeoutMs); + if (!readinessSchedule) { + const attempt = await attemptSelectorResolution(runtime, options, selectorExpression, params); + capture = attempt.capture; + resolved = attempt.resolved; + if (!resolved || !resolved.node.rect) { + throw await selectorInteractionFailure({ + runtime, + nodes: capture.snapshot.nodes, + selectorExpression, + action: params.action, + resolved, + }); + } + } else { + const ready = await pollForSelectorReadiness( runtime, - nodes: capture.snapshot.nodes, + options, selectorExpression, - action: params.action, - resolved, - }); + params, + readinessSchedule, + ); + capture = ready.capture; + resolved = ready.resolved; } - // #1542: see the ref-target twin above. + // #1542: see the ref-target twin in ref-target-resolution.ts. const selected = resolved; const { node: visibleNode, tapPoint: point } = await runInteractionPipelineStages({ policy: params.pipeline, @@ -448,317 +198,6 @@ async function resolveSelectorInteractionTarget( }; } -/** - * No usable acting target. Before reporting "did not match", re-probe the same - * tree through the diagnosis row: a selector that DOES match but landed on a - * covered node is a different failure with a different recovery, and the - * acting row — rect-required, candidates rejected — cannot tell the caller - * that. Both probes name a policy row, so the two contracts stay visible side - * by side instead of as two sets of engine knobs (#1630). - */ -async function selectorInteractionFailure(params: { - runtime: AgentDeviceRuntime; - nodes: SnapshotState['nodes']; - selectorExpression: string; - action: InteractionAction; - resolved: SelectorResolution | null; -}): Promise { - const { runtime, nodes, selectorExpression, action, resolved } = params; - // The diagnosis row keeps covered nodes as candidates precisely so its - // occlusion stage can report them: "matched but covered" is a different - // failure with a different recovery than "did not match". - const covered = await resolveSelectorPipeline( - SELECTOR_PIPELINE_POLICIES.coveredDiagnosis, - nodes, - selectorExpression, - { platform: runtime.backend.platform }, - ); - if (covered.kind === 'occluded') { - return buildCoveredInteractionError({ - label: `Selector ${covered.selector}`, - node: covered.node, - action, - selector: covered.selector, - }); - } - const diagnostics = resolved?.diagnostics ?? []; - return new AppError( - 'COMMAND_FAILED', - formatSelectorFailure(selectorExpression, diagnostics, { unique: true }), - { - reason: INTERACTION_ERROR_REASONS.selectorNotFound, - hint: selectorFailureHint(diagnostics), - }, - ); -} - -function assertReplayTargetResolution( - node: SnapshotNode, - nodes: SnapshotState['nodes'], - params: ResolveInteractionTargetParams, -): void { - assertExpectedResolvedTarget( - node, - nodes, - params.expectedResolvedTarget, - params.action, - params.replayTargetRole, - ); -} - -// ADR 0012 decision 2 bounds: diagnostic strings and losing alternatives. -const RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP = 256; -const MAX_RESOLUTION_ALTERNATIVES = 5; - -/** - * A successful `@ref` lookup names exactly one node; label recovery discloses label-fallback instead. - * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites it as a `via` - * symbol and the gate test imports it dynamically, which fallow cannot trace statically. - */ -// fallow-ignore-next-line unused-export -export const EXACT_REF_RESOLUTION: ResolutionDisclosure = { - source: 'ref', - phase: 'pre-action', - kind: 'exact', -}; - -const LABEL_FALLBACK_REF_RESOLUTION: ResolutionDisclosure = { - source: 'ref', - phase: 'pre-action', - kind: 'label-fallback', -}; - -/** Shared construction site for every runtime-ref resolution disclosure. */ -export function buildRefResolution( - ref: string, - node: SnapshotNode, - kind: 'exact' | 'label-fallback', -): ResolvedRefNode { - return { - ref, - node, - resolution: kind === 'exact' ? EXACT_REF_RESOLUTION : LABEL_FALLBACK_REF_RESOLUTION, - }; -} - -const UNIQUE_RUNTIME_RESOLUTION: ResolutionDisclosure = { - source: 'runtime', - phase: 'pre-action', - kind: 'unique', -}; - -// Disclosure only: the winner stays resolveSelectorChain's pick (ADR 0012). -function buildSelectorResolutionDisclosure( - resolved: SelectorResolution, - nodes: SnapshotState['nodes'], -): ResolutionDisclosure { - if (!resolved.disambiguation) return UNIQUE_RUNTIME_RESOLUTION; - return { - source: 'runtime', - phase: 'pre-action', - kind: 'disambiguated', - matchCount: resolved.disambiguation.matchCount, - winnerDiagnostic: buildResolutionDiagnosticEntry(resolved.node, nodes), - tiebreak: resolved.disambiguation.tiebreak, - alternatives: resolved.disambiguation.alternatives - .slice(0, MAX_RESOLUTION_ALTERNATIVES) - .map((node) => buildResolutionDiagnosticEntry(node, nodes)), - }; -} - -function buildResolutionDiagnosticEntry( - node: SnapshotNode, - nodes: SnapshotState['nodes'], -): ResolutionDiagnosticEntry { - const role = normalizeType(node.type ?? ''); - const label = resolveRefLabel(node, nodes); - return { - diagnosticRef: `diag-${node.ref}`, - ...(role ? { role: truncateUtf8(role, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } : {}), - ...(label !== undefined - ? { label: truncateUtf8(label, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } - : {}), - }; -} - -// Shared tail of a resolved ref/selector interaction target: the node itself -// plus everything derived from it for the response. Every response field -// describes the DISPATCHED node — the #1280 retarget rides only on the -// `recordingTarget` side channel below. `tree` is the capture the node was -// resolved from, and becomes the pre-action baseline this publishes. -function describeResolvedInteractionNode( - runtime: AgentDeviceRuntime, - node: SnapshotNode, - tree: SurfaceScopedNodes, - action: InteractionAction, - resolution: ResolutionDisclosure, -): { - node: SnapshotNode; - selectorChain: string[]; - refLabel: string | undefined; - targetHittable?: boolean; - hint?: string; - preAction: SurfaceScopedNodes; - resolution: ResolutionDisclosure; - recordingTarget?: RecordingTargetOverride; -} { - const nodes = tree.nodes; - return { - node, - selectorChain: buildSelectorChainForNode(node, runtime.backend.platform, { - action: action === 'fill' ? 'fill' : 'click', - nodes, - }), - refLabel: resolveRefLabel(node, nodes), - ...describeNonHittableTarget(node, action), - preAction: tree, - resolution, - ...pressRecordingTargetOverride(runtime, node, nodes, action), - }; -} - -/** - * #1280 (ADR 0012 decision 3 amendment): the recording-only side channel. - * When a click/press resolves to an identity-empty container, the RECORDED - * step retargets to its first labeled descendant — node, chain, and - * ref-label computed together here so the recorded action entry and its - * `target-v1` evidence can never half-retarget. The response payloads never - * consume this (see `interaction-touch-response.ts`). `fill` is deliberately - * excluded: its chain carries `editable=true` constraints a label descendant - * cannot satisfy, which would record an unreplayable selector. - */ -function pressRecordingTargetOverride( - runtime: AgentDeviceRuntime, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - action: InteractionAction, -): { recordingTarget?: RecordingTargetOverride } { - if (action !== 'click' && action !== 'press') return {}; - const recordingNode = resolvePressRecordingTarget(node, nodes); - if (recordingNode === node) return {}; - return { - recordingTarget: { - node: recordingNode, - selectorChain: buildSelectorChainForNode(recordingNode, runtime.backend.platform, { - action: 'click', - nodes, - }), - refLabel: resolveRefLabel(recordingNode, nodes), - }, - }; -} - -/** - * iOS AX `hittable` flags are unreliable on deep React Native trees (see #1037: - * a map-pin annotation exact-matched a longer recents row label and reported tap - * success while doing nothing visible). We deliberately do NOT fail or filter on - * this signal — that would break selectors that only ever resolve to nodes the - * platform marks non-hittable. Instead, surface it so the caller can notice a - * likely no-op tap and re-target with a ref or a more specific selector/longer text. - */ -function describeNonHittableTarget( - node: SnapshotNode, - action: InteractionAction, -): { targetHittable?: boolean; hint?: string } { - if (node.hittable !== false) return {}; - return { - targetHittable: false, - hint: `The resolved element reports hittable: false, so this ${action} may have had no visible effect. Verify with a snapshot, or prefer a @ref or a longer/more specific selector to target the intended element.`, - }; -} - -/** - * Every node stage this action's row declares, plus the covered and keyboard refusals the - * interaction runtime owns. Which stages run is the row's decision; every acting path — selector, - * ref, and the native-ref preflight — enters them here, which is what keeps the native-ref fast - * path from succeeding on a target the shared rules would refuse. - * - * Each path hands in the resolver that produces the point it taps with, and taps the point that - * comes back, so the keyboard guard measures the coordinate the interaction is actually made of and no - * path derives a second one. A path whose point can fail to exist — the native-ref fast path taps by - * ref, reading the rect center the platform aims at — says so in its resolver's return type. - */ -async function runInteractionPipelineStages(params: { - policy: SelectorPipelinePolicy; - nodes: SnapshotState['nodes']; - /** The keyboard band `nodes`' capture measured, when it measured one (#2660). */ - keyboard?: SnapshotKeyboardBandFact; - node: SnapshotNode; - action: InteractionAction; - label: string; - hooks: SelectorPipelineHooks; - resolveTapPoint: (node: SnapshotNode) => TPoint; -}): Promise<{ node: SnapshotNode; tapPoint: TPoint }> { - const target = await runNodePipelineStages( - params.policy, - params.nodes, - params.node, - params.hooks, - ); - if (target.kind === 'occluded') { - throw buildCoveredInteractionError({ - label: params.label, - node: target.node, - action: params.action, - }); - } - const tapPoint = params.resolveTapPoint(target.node); - assertTapTargetClearOfVisibleKeyboard({ - nodes: params.nodes, - node: target.node, - action: params.action, - label: params.label, - ...(params.keyboard ? { keyboard: params.keyboard } : {}), - tapPoint, - }); - return { node: target.node, tapPoint }; -} - -function buildCoveredInteractionError(params: { - label: string; - node: SnapshotNode; - action: InteractionAction; - selector?: string; -}): AppError { - return new AppError( - 'COMMAND_FAILED', - `${params.label} is covered by another visible element and cannot ${interactionVerb(params.action)} safely`, - { - hint: 'Use a different visible target, scroll it clear of the overlay, or inspect with snapshot/screenshot before retrying.', - ...(params.selector ? { selector: params.selector } : {}), - ref: `@${params.node.ref}`, - interactionBlocked: params.node.interactionBlocked, - }, - ); -} - -export async function captureInteractionSnapshot( - runtime: AgentDeviceRuntime, - options: CommandContext, - interactiveOnly: boolean, -): Promise { - if (!runtime.backend.captureSnapshot) { - throw new AppError('UNSUPPORTED_OPERATION', 'snapshot is not supported by this backend'); - } - const sessionName = options.session ?? 'default'; - const session = await runtime.sessions.get(sessionName); - if (!session) throw new AppError('SESSION_NOT_FOUND', 'No active session. Run open first.'); - const result = await runtime.backend.captureSnapshot(toBackendContext(runtime, options), { - interactiveOnly, - includeRects: true, - }); - const snapshot = - result.snapshot ?? - ({ - nodes: result.nodes ?? [], - truncated: result.truncated, - backend: result.backend as SnapshotState['backend'], - createdAt: now(runtime), - } satisfies SnapshotState); - await runtime.sessions.set({ ...session, snapshot }); - return { snapshot }; -} - export async function assertSupportedInteractionSurface( runtime: AgentDeviceRuntime, options: CommandContext, @@ -782,410 +221,3 @@ async function resolveInteractionSurface( const session = await runtime.sessions.get(options.session ?? 'default'); return session?.metadata?.surface; } - -async function resolveSnapshotForRef( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, -): Promise { - const { session, snapshot: frameTree } = await requireSnapshotSession(runtime, options.session); - - const fallbackLabel = target.fallbackLabel ?? ''; - const outcome = tryResolveRefNode(frameTree.nodes, target.ref, { - fallbackLabel, - }); - // ADR 0014: missing authorized-frame evidence FAILS. It must not fall through - // to a fresh capture and accept the same ref body from a newer tree — that is - // exactly the positional-coincidence retarget the frame model forbids. A stale - // read is observable and recoverable; a stale mutation can act on the wrong - // element. The caller re-observes (snapshot) or uses a selector. - if (outcome.kind !== 'resolved') throw refMissRefusal(outcome, target.ref); - return reconcileFreshObservation({ - session, - frameTree, - target, - fallbackLabel, - authorized: outcome.resolved, - }); -} - -/** - * ADR 0014 step 5: decouple Android freshness from ref authorization. The frame - * tree names WHICH node `@eN` authorizes. When a freshness (or other read-only) - * capture has advanced the operational observation past the frame, adopt the - * observation's node — its fresh on-screen coordinates — ONLY when its local - * identity still matches the authorized node. That covers the legitimate case of - * an element that merely moved. If the identity differs (a different element now - * sits at that index) or the ref is absent from the observation, keep the - * authorized frame node so a positional coincidence cannot retarget the action. - */ -function reconcileFreshObservation(params: { - session: CommandSessionRecord; - frameTree: SnapshotState; - target: Extract; - fallbackLabel: string; - authorized: ResolvedRefNode; -}): InteractionSnapshot & { resolved: ResolvedRefNode } { - const { session, frameTree, target, fallbackLabel, authorized } = params; - const observation = session.snapshot; - if (!observation || observation === frameTree) { - return { snapshot: frameTree, resolved: authorized }; - } - const observed = tryResolveRefNode(observation.nodes, target.ref, { fallbackLabel }); - if ( - observed.kind === 'resolved' && - localIdentitiesEqual( - readNodeLocalIdentity(authorized.node), - readNodeLocalIdentity(observed.resolved.node), - ) - ) { - return { snapshot: observation, resolved: observed.resolved }; - } - return { snapshot: frameTree, resolved: authorized }; -} - -/** The runtime-ref resolver: `exact` for a resolved `@ref`, `label-fallback` for trailing-label recovery. */ -/** - * What one tree makes of a ref: the node it authorizes (exact, or the trailing-label recovery), a - * node it lists (by ref or by that label) that has no usable centre, or no node at all. The two - * misses are distinct outcomes so a caller can name a stale ref and an unactionable target apart. - */ -export type RefResolutionOutcome = - | { kind: 'resolved'; resolved: ResolvedRefNode } - | { kind: 'unusable'; node: SnapshotNode } - | { kind: 'missing' }; - -export function tryResolveRefNode( - nodes: SnapshotState['nodes'], - refInput: string, - options: { - fallbackLabel: string; - }, -): RefResolutionOutcome { - const ref = normalizeRef(refInput); - if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${refInput}`); - const refNode = findNodeByRef(nodes, ref); - if (isUsableResolvedNode(refNode)) { - return { kind: 'resolved', resolved: buildRefResolution(ref, refNode, 'exact') }; - } - const fallbackNode = - options.fallbackLabel.length > 0 ? findNodeByLabel(nodes, options.fallbackLabel) : null; - if (isUsableResolvedNode(fallbackNode)) { - return { kind: 'resolved', resolved: buildRefResolution(ref, fallbackNode, 'label-fallback') }; - } - const found = refNode ?? fallbackNode; - return found ? { kind: 'unusable', node: found } : { kind: 'missing' }; -} - -type ResolvedRefNode = { - ref: string; - node: SnapshotNode; - resolution: ResolutionDisclosure; -}; - -/** - * The refusal for a ref the frame could not authorize: a ref no node carries is stale or was never - * issued (`ref_not_found`); a ref whose node is listed but has no usable centre is present and - * unactionable (`target_bounds_invalid`). Both recover the same way, a fresh observation, so both - * carry the stale-ref hint; `details.ref` is the bare ref body either way. - */ -function refMissRefusal( - miss: Exclude, - refInput: string, -): AppError { - const ref = normalizeRef(refInput) ?? refInput; - return miss.kind === 'unusable' - ? new AppError('COMMAND_FAILED', `Ref ${refInput} has no usable bounds`, { - reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, - ref, - hint: STALE_REF_HINT, - }) - : new AppError('COMMAND_FAILED', `Ref ${refInput} not found`, { - reason: INTERACTION_ERROR_REASONS.refNotFound, - ref, - hint: STALE_REF_HINT, - }); -} - -function resolveNodeTouchPoint( - node: SnapshotNode, - nodes: SnapshotState['nodes'], - failure: { - invalidMessage: string; - blockedTargetLabel: string; - blockedTargetDetails: { ref: string } | { selector: string }; - }, -): Point { - const visibility = createSnapshotVisibility(nodes); - const effectiveViewport = visibility.resolveEffectiveViewport(node); - const rootViewport = node.rect ? visibility.resolveViewport(node.rect) : null; - const resolution = resolveInteractionTouchPoint(nodes, node, { - bounds: [effectiveViewport, rootViewport].filter((rect) => rect !== null), - }); - if (resolution.kind === 'resolved') return resolution.point; - if (resolution.kind === 'invalid') { - throw new AppError('COMMAND_FAILED', failure.invalidMessage, { - reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, - ...bareTargetDetails(failure.blockedTargetDetails), - }); - } - throw new AppError( - 'COMMAND_FAILED', - `${failure.blockedTargetLabel} has no parent-owned touch point outside its interactive descendants`, - { - reason: 'covered_by_interactive_descendants', - ...failure.blockedTargetDetails, - competitorRefs: resolution.competitorRefs.slice(0, 5).map((ref) => `@${ref}`), - competitorCount: resolution.competitorRefs.length, - hint: 'Tap the specific interactive child you intend, or use a more specific selector. Every safely tappable region of the parent belongs to one of its child controls.', - }, - ); -} - -/** `details.ref` is the bare ref body on every reason; the blocked-target label keeps its `@`. */ -function bareTargetDetails( - details: { ref: string } | { selector: string }, -): { ref: string } | { selector: string } { - return 'ref' in details ? { ref: normalizeRef(details.ref) ?? details.ref } : details; -} - -function isUsableResolvedNode(node: SnapshotNode | null | undefined): node is SnapshotNode { - if (!node) return false; - return resolveRectCenter(node.rect) !== null; -} - -/** - * The off-screen stage's refusal shape. Reached only through the pipeline - * owner, and only for rows whose off-screen stage refuses — the row's decision - * is made there, so this builds the message and never re-decides. - */ -type OffscreenStageParams = { action: InteractionAction; pipeline: SelectorPipelinePolicy }; - -// Selector parity for the @ref off-screen guard: without it, a selector -// resolving to a closed drawer/carousel item "succeeds" by tapping coordinates -// outside the viewport (observed as `Tapped (-161, 265)` against Bluesky's -// closed drawer) while the same node via @ref is refused. -async function assertVisibleSelectorTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - selector: string, - { action }: OffscreenStageParams, -): Promise { - return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { - message: `Selector ${selector} resolved to an off-screen element and is not safe to ${action}`, - details: { reason: 'offscreen_selector', selector }, - // A selector re-resolves against a fresh snapshot on every attempt, so the - // recovery is: move the named direction, then retry THIS selector — no - // separate snapshot step, and no @ref (a scroll expires the ref frame, - // #1366). `--until` is that whole loop as one command: it checks the same - // selector between passes, which is also what keeps a large step from - // overshooting, so the hint no longer has to trade distance for accuracy. - hint: (direction) => - `${scrollRevealClause(direction, selector)} then retry ${action} with the same selector. --until checks the selector between passes, so it stops on the target rather than sailing past it. If it is inside a closed drawer or another tab, open that container first.`, - }); -} - -async function assertVisibleRefTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - refInput: string, - { action }: OffscreenStageParams, -): Promise { - return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { - message: `Ref ${refInput} is off-screen and not safe to ${action}`, - details: { reason: 'offscreen_ref', ref: normalizeRef(refInput) }, - // The scroll that reveals the target expires the ref frame (#1366, ADR - // 0014), so retrying this @ref would be rejected next. Steer to a selector, - // which re-resolves against a fresh snapshot and bypasses the ref-frame guard - // — and which `--until` can then check between passes. - hint: (direction) => - `${scrollRevealClause(direction, null)} then retry ${action} with a selector (e.g. text=/id=) rather than this @ref — the scroll expires the ref frame, so re-run snapshot -i before reusing any @ref.`, - }); -} - -/** - * Shared lead-in for both off-screen hints: the one command that reveals the target. - * - * When the geometry names a direction AND the caller has a selector to check, this is a complete - * `scroll --until ` — one request that stops on the target instead of the - * scroll-then-look-again loop the hint used to prescribe. Without a selector to check (an @ref - * refusal) or without a single reveal direction (off more than one edge), it degrades to naming - * the move and leaves the stop condition to the caller's own next step. - */ -function scrollRevealClause( - direction: OffscreenScrollDirection | null, - selector: string | null, -): string { - if (!direction) return 'Scroll toward it,'; - if (!selector) return `Scroll ${direction} toward it,`; - return `Run scroll ${direction} --until '${selector}' to bring it on screen,`; -} - -/** - * ADR 0011 native-ref preflight: `click @ref` / `fill @ref` fast paths - * dispatch straight to `backend.tapTarget`/`fillTarget`, and a backend fast - * path can silently "succeed" — delegation-on-error never triggers there. The - * ref came from the stored session snapshot, so the node is already in hand: - * run the SAME shared guards the runtime path uses against it before the - * backend call — occlusion (`isSnapshotNodeInteractionBlocked` via - * `assertInteractionNotBlocked`) and offscreen (the snapshot visibility resolver via - * `assertVisibleRefTarget`) ERROR with the runtime path's exact shapes, and - * the non-hittable annotation is returned for the fast-path result. - * - * Zero extra round trips by construction on the accept path: no session, no - * stored snapshot, an unresolvable/invalid ref, or a node without a usable - * rect all make the preflight a no-op and the fast path proceeds exactly as - * before. Promotion to a hittable ancestor stays a runtime-path behavior — - * the preflight never changes which element the backend acts on. Exception: - * a would-be off-screen refusal may spend one extra iOS runner round trip - * (#1542's double-check) before erroring — cost only on the path that was - * about to fail anyway. - * - * Exported as an ADR 0011 registry anchor (interaction-guarantees.ts `via` - * symbol, imported dynamically by the gate test); production callers reach - * it through `dispatchNativeRefInteraction`. - */ -// fallow-ignore-next-line unused-export -export async function preflightNativeRefInteraction( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, - action: InteractionAction, -): Promise<{ - targetHittable?: boolean; - hint?: string; - node?: SnapshotNode; - preAction?: SurfaceScopedNodes; -}> { - const session = await runtime.sessions.get(options.session ?? 'default'); - const storedSnapshot = session?.snapshot; - const nodes = storedSnapshot?.nodes; - if (!storedSnapshot || !nodes || normalizeRef(target.ref) === null) return {}; - const outcome = tryResolveRefNode(nodes, target.ref, { - fallbackLabel: target.fallbackLabel ?? '', - }); - if (outcome.kind !== 'resolved') return {}; - const { resolved } = outcome; - // `resolvedTarget` whatever the command: its `none` promotion is what holds - // ADR 0011's "the preflight never changes which element the backend acts on". - const pipeline = SELECTOR_PIPELINE_POLICIES.resolvedTarget; - // #1542: dispatches by REF, not coordinate, so no point is re-derived for the dispatch — but - // evidence/annotation below still describes the returned (visible) node. - const { node: visibleNode } = await runInteractionPipelineStages({ - policy: pipeline, - nodes, - node: resolved.node, - action, - label: `Ref ${target.ref}`, - hooks: { - offscreen: async (node, tree) => - await assertVisibleRefTarget(runtime, options, node, tree, target.ref, { - action, - pipeline, - }), - }, - resolveTapPoint: (node) => resolveRectCenter(node.rect), - }); - return { - ...describeNonHittableTarget(visibleNode, action), - // ADR 0012 decision 3: the guard lookup above doubles as the record-time - // evidence source for the fast path, at zero extra capture cost. - node: visibleNode, - preAction: surfaceScopedNodes(storedSnapshot), - }; -} - -/** - * ADR 0011 native-ref dispatch, shared by click/fill/hover @ref: run the - * preflight guards against the stored node, hand the ref to the backend as - * its own element handle, and return the exact-ref result envelope. Callers - * decide WHEN the path applies (backend capability, no non-default options, - * no replay guard, no settle baseline); this owns only the dispatch itself so - * the three commands cannot drift on preflight or disclosure. - */ -export async function dispatchNativeRefInteraction( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, - action: InteractionAction, - dispatch: ( - context: BackendCommandContext, - refTarget: BackendRefTarget, - ) => Promise, -): Promise< - Extract & { backendResult?: Record } -> { - const preflight = await preflightNativeRefInteraction(runtime, options, target, action); - const backendResult = await dispatch(toBackendContext(runtime, options), { - kind: 'ref', - ref: target.ref, - ...(target.fallbackLabel ? { fallbackLabel: target.fallbackLabel } : {}), - }); - const formattedBackendResult = toBackendResult(backendResult); - return { - kind: 'ref', - target: { kind: 'ref', ref: target.ref }, - resolution: EXACT_REF_RESOLUTION, - ...preflight, - ...(formattedBackendResult ? { backendResult: formattedBackendResult } : {}), - }; -} - -// Full on-screen visibility (not only the effective-viewport form): items inside an -// off-screen scrollable container (closed drawer) must also count as -// off-screen, not just items scrolled out of an on-screen container. -// -// #1542: once the bulk tree says off-screen, the guard gives iOS one chance -// to rescue a FALSE refusal via the optional backend.confirmOffscreenTargetVisible -// hook — a stale/corrupted bulk tree can say off-screen while the app is -// visually fine (zero cost on the accept path; runs only here). A confirmed -// rescue returns the node PATCHED WITH THE LIVE RECT: the caller must act on -// that returned node, never the original, because in the frozen-bulk-tree -// manifestation the original rect can be stale even when the rescue verdict -// is correct — tapping it would silently land at the wrong coordinate. The -// hook fails closed (null) on anything short of a positive confirmation, so -// a genuine refusal, or any backend without the hook, is unchanged. -// -// Exported (not just for callers here) for ADR 0011 registry honesty: -// interaction-guarantees.ts's `offscreen` cells point their `via` at this -// function, not at the contracts predicate alone, since this is the actual -// end-to-end enforcement point. -export async function throwIfOffscreenInteractionTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - failure: { - message: string; - details: Record; - hint: (direction: OffscreenScrollDirection | null) => string; - }, -): Promise { - const visibility = createSnapshotVisibility(nodes); - const viewport = node.rect ? visibility.resolveEffectiveViewport(node) : null; - if (!node.rect || !viewport || visibility.isVisibleOnScreen(node)) return node; - const rootViewport = visibility.resolveViewport(node.rect); - const liveRect = await runtime.backend.confirmOffscreenTargetVisible?.( - toBackendContext(runtime, options), - node, - rootViewport, - ); - if (liveRect) return { ...node, rect: liveRect }; - // The direction that scrolls this off-screen target into view. Named in the - // hint (and surfaced as a machine-readable detail) so the recovery is a single - // deterministic move instead of a guess (#1366). Derived from the same - // boundary the rejection above used, so partial clips and off-screen - // containers get a direction too, not just fully-scrolled-out items. - const scrollDirection = classifyOffscreenScrollDirection(node, visibility); - throw new AppError('COMMAND_FAILED', failure.message, { - ...failure.details, - rect: node.rect, - viewport, - ...(scrollDirection ? { scrollDirection } : {}), - hint: failure.hint(scrollDirection), - }); -} diff --git a/src/commands/interaction/runtime/selector-is.ts b/src/commands/interaction/runtime/selector-is.ts index 498e83acbb..03fb1eaece 100644 --- a/src/commands/interaction/runtime/selector-is.ts +++ b/src/commands/interaction/runtime/selector-is.ts @@ -16,7 +16,8 @@ import { AppError, isRequestCanceledError } from '@agent-device/kernel/errors'; import type { SelectorTarget } from '@agent-device/contracts/interaction'; import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; import type { RuntimeCommand } from '../../runtime-types.ts'; -import { assertExpectedResolvedTarget, type ExpectedResolvedTarget } from './resolution.ts'; +import { assertExpectedResolvedTarget } from './replay-target-guard.ts'; +import type { ExpectedResolvedTarget } from './interaction-resolution-request.ts'; import { type CapturedSnapshot, type SelectorSnapshotOptions, diff --git a/src/commands/interaction/runtime/selector-read.ts b/src/commands/interaction/runtime/selector-read.ts index f69d8c956c..65d5df9579 100644 --- a/src/commands/interaction/runtime/selector-read.ts +++ b/src/commands/interaction/runtime/selector-read.ts @@ -24,7 +24,8 @@ import type { ResolvedTarget, } from '@agent-device/contracts/interaction'; import type { RuntimeCommand } from '../../runtime-types.ts'; -import { assertExpectedResolvedTarget, type ExpectedResolvedTarget } from './resolution.ts'; +import { assertExpectedResolvedTarget } from './replay-target-guard.ts'; +import type { ExpectedResolvedTarget } from './interaction-resolution-request.ts'; import { type CapturedSnapshot, type SelectorSnapshotOptions, diff --git a/src/commands/interaction/runtime/selector-readiness.test.ts b/src/commands/interaction/runtime/selector-readiness.test.ts new file mode 100644 index 0000000000..a1493b2247 --- /dev/null +++ b/src/commands/interaction/runtime/selector-readiness.test.ts @@ -0,0 +1,123 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { AppError } from '@agent-device/kernel/errors'; +import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; +import { selector } from './selector-read-utils.ts'; +import { createFakeClock, createInteractionDevice } from './__tests__/test-utils/index.ts'; + +// promotedTarget's readiness poll runs only when the caller supplies `readinessTimeoutMs`, capped +// at the row's `maxTimeoutMs` (2_000). These pin the gating/capping decision at the runtime layer; the end-to-end poll mechanics (interactive/fresh +// capture sequencing, covered-target diagnosis, ridden-out capture errors) are unit-tested at the +// daemon level in test/integration/provider-scenarios/press-target-readiness.test.ts. + +const CONTINUE_BUTTON = { + index: 0, + depth: 0, + type: 'Button', + label: 'Continue', + rect: { x: 10, y: 20, width: 100, height: 40 }, + hittable: true, +}; + +test('runtime press without readinessTimeoutMs takes the one-attempt path and reports no readiness evidence', async () => { + let captures = 0; + const device = createInteractionDevice(makeSnapshotState([]), { + captureSnapshot: async () => { + captures += 1; + return { snapshot: makeSnapshotState([]) }; + }, + }); + + await assert.rejects( + () => device.interactions.press(selector('label=Continue'), { session: 'default' }), + (error: unknown) => { + assert.ok(error instanceof AppError); + assert.equal(error.details?.readiness, undefined); + return true; + }, + ); + // The interactive-capture-then-full-capture-fallback pair, not the readiness loop's repeated + // polling. + assert.equal(captures, 2); +}); + +test('runtime press with readinessTimeoutMs polls until the target appears', async () => { + let captures = 0; + const device = createInteractionDevice(makeSnapshotState([]), { + clock: createFakeClock(), + captureSnapshot: async () => { + captures += 1; + // Every capture before the fourth misses; the fourth (and every one after) has the button. + return { snapshot: makeSnapshotState(captures >= 4 ? [CONTINUE_BUTTON] : []) }; + }, + tap: async () => ({ ok: true }), + }); + + const result = await device.interactions.press(selector('label=Continue'), { + session: 'default', + readinessTimeoutMs: 2_000, + }); + + assert.equal(result.kind, 'selector'); + assert.equal(result.node?.label, 'Continue'); + assert.ok( + captures >= 4, + `expected at least 4 captures before the target appeared, got ${captures}`, + ); +}); + +test('runtime press caps a readinessTimeoutMs larger than the row maxTimeoutMs at 2_000ms', async () => { + const device = createInteractionDevice(makeSnapshotState([]), { + clock: createFakeClock(), + captureSnapshot: async () => ({ snapshot: makeSnapshotState([]) }), + }); + + await assert.rejects( + () => + device.interactions.press(selector('label=Continue'), { + session: 'default', + // Far past the promotedTarget row's own maxTimeoutMs (2_000): the row's cap governs, not + // this caller-supplied value. + readinessTimeoutMs: 999_000, + }), + (error: unknown) => { + assert.ok(error instanceof AppError); + const readiness = error.details?.readiness as { waitedMs: number; end: string } | undefined; + assert.ok(readiness, 'expected readiness evidence on an exhausted poll'); + assert.equal(readiness.end, 'expired'); + assert.ok( + readiness.waitedMs >= 2_000 && readiness.waitedMs <= 2_200, + `expected waitedMs capped near 2_000ms, got ${readiness.waitedMs}`, + ); + return true; + }, + ); +}); + +test('runtime press whose every capture is unreadable fails with the unreadable-content error, not a selector miss', async () => { + let captures = 0; + const device = createInteractionDevice(makeSnapshotState([]), { + clock: createFakeClock(), + captureSnapshot: async () => { + captures += 1; + throw new AppError('COMMAND_FAILED', 'Android snapshot has no readable app content', { + androidSnapshotHelperFailureReason: 'system-window-only', + }); + }, + }); + + await assert.rejects( + () => + device.interactions.press(selector('label=Continue'), { + session: 'default', + readinessTimeoutMs: 2_000, + }), + (error: unknown) => { + assert.ok(error instanceof AppError); + assert.equal(error.details?.androidSnapshotHelperFailureReason, 'system-window-only'); + assert.notEqual(error.details?.reason, 'selector_not_found'); + return true; + }, + ); + assert.ok(captures >= 2, `expected the unreadable captures to be ridden out, got ${captures}`); +}); diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts new file mode 100644 index 0000000000..cddb669138 --- /dev/null +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -0,0 +1,297 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotState } from '@agent-device/kernel/snapshot'; +import { inheritPostGestureOutcome } from '@agent-device/kernel/snapshot'; +import { + formatSelectorFailure, + selectorFailureHint, + type SelectorResolution, +} from '@agent-device/selectors'; +import { resolveSelectorPipeline } from '@agent-device/selectors/selector-pipeline'; +import { + SELECTOR_PIPELINE_POLICIES, + type ReadinessSchedule, +} from '@agent-device/selectors/selector-pipeline-policy'; +import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { observeUntil } from '@agent-device/capture-kit/observe-until'; +import { isSparseSnapshotQualityVerdict } from '@agent-device/capture-kit/snapshot-quality-verdict'; +import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { + captureInteractionSnapshot, + type InteractionSnapshot, +} from './interaction-snapshot-capture.ts'; +import { buildCoveredInteractionError } from './target-visibility-stages.ts'; +import { resolveActionSelector } from './selector-action-resolution.ts'; +import type { + InteractionAction, + ResolveInteractionTargetParams, +} from './interaction-resolution-request.ts'; + +/** + * `promotedTarget`'s readiness poll: the one-attempt capture-and-resolve, the covered-target + * diagnosis it shares with the exhausted-budget failure, and the `observeUntil`-driven loop itself. + * `resolution.ts#resolveSelectorInteractionTarget` is this module's one caller, deciding whether a + * call polls at all and under what capped budget. + */ + +/** One poll's outcome: the capture it read the tree from, and what it resolved (if anything). */ +export type SelectorResolutionAttempt = { + capture: InteractionSnapshot; + resolved: SelectorResolution | null; +}; + +/** The narrowed attempt a readiness poll accepts: a resolution with a usable rect. */ +export type ResolvedSelectorAttempt = { + capture: InteractionSnapshot; + resolved: SelectorResolution; +}; + +/** The readiness poll's evidence, attached to a target-not-found failure only (never to a refusal). */ +export type SelectorReadinessDetails = { + polls: number; + waitedMs: number; + end: 'expired' | 'stalled' | 'sparse'; +}; + +/** + * One capture-and-resolve attempt: interactive capture, resolve; on a miss that requires + * interactivity, fall back to a full (non-interactive) capture and resolve again. Identical body + * whether run once (`promotedTarget.poll`'s one-attempt twin, `resolvedTarget`) or repeated under + * `observeUntil` for a call that polls — this function draws no distinction, and does not decide + * whether the miss is a refusal (ambiguity, occlusion, off-screen) or a plain no-match: those stay + * the caller's decisions, made from `resolved`/thrown errors exactly as before. + */ +export async function attemptSelectorResolution( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, +): Promise { + let capture = await captureInteractionSnapshot(runtime, options, params.requireInteractive); + let resolved = resolveActionSelector( + capture.snapshot.nodes, + selectorExpression, + runtime.backend.platform, + params.pipeline, + ); + if ((!resolved || !resolved.node.rect) && params.requireInteractive) { + const interactive = capture.snapshot; + capture = await captureInteractionSnapshot(runtime, options, false); + inheritPostGestureOutcome(interactive, capture.snapshot); + resolved = resolveActionSelector( + capture.snapshot.nodes, + selectorExpression, + runtime.backend.platform, + params.pipeline, + ); + } + return { capture, resolved }; +} + +/** + * No usable acting target. Before reporting "did not match", re-probe the same + * tree through the diagnosis row: a selector that DOES match but landed on a + * covered node is a different failure with a different recovery, and the + * acting row — rect-required, candidates rejected — cannot tell the caller + * that. Both probes name a policy row, so the two contracts stay visible side + * by side instead of as two sets of engine knobs. + * + * The diagnosis row keeps covered nodes as candidates precisely so its occlusion stage can report + * them: "matched but covered" is a different failure than "did not match", produced by re-probing + * the same tree under `coveredDiagnosis` rather than by the acting row's own (candidate-excluding) + * resolution. Shared by the single-attempt failure path and the readiness poll: a covered target is + * a refusal, not an absence, and must end a poll loop rather than spend its budget (a `promotedTarget` + * poll cannot otherwise tell "excluded because covered" apart from "does not exist yet"). + */ +async function detectCoveredSelectorTarget(params: { + runtime: AgentDeviceRuntime; + nodes: SnapshotState['nodes']; + selectorExpression: string; + action: InteractionAction; +}): Promise { + const { runtime, nodes, selectorExpression, action } = params; + const covered = await resolveSelectorPipeline( + SELECTOR_PIPELINE_POLICIES.coveredDiagnosis, + nodes, + selectorExpression, + { platform: runtime.backend.platform }, + ); + if (covered.kind !== 'occluded') return undefined; + return buildCoveredInteractionError({ + label: `Selector ${covered.selector}`, + node: covered.node, + action, + selector: covered.selector, + }); +} + +/** + * Shared by the one-attempt path (`resolution.ts#resolveSelectorInteractionTarget`) and this + * module's own exhausted-budget path: the same "selector not found" shape either way, covered + * targets disclosed through `detectCoveredSelectorTarget` first. + */ +export async function selectorInteractionFailure(params: { + runtime: AgentDeviceRuntime; + nodes: SnapshotState['nodes']; + selectorExpression: string; + action: InteractionAction; + resolved: SelectorResolution | null; +}): Promise { + const { runtime, nodes, selectorExpression, action, resolved } = params; + const covered = await detectCoveredSelectorTarget({ runtime, nodes, selectorExpression, action }); + if (covered) return covered; + const diagnostics = resolved?.diagnostics ?? []; + return new AppError( + 'COMMAND_FAILED', + formatSelectorFailure(selectorExpression, diagnostics, { unique: true }), + { + reason: INTERACTION_ERROR_REASONS.selectorNotFound, + hint: selectorFailureHint(diagnostics), + }, + ); +} + +/** + * One readiness poll: a capture-and-resolve attempt, the previous poll's post-gesture outcome + * carried forward, and the covered-target probe on a miss. A covered candidate is excluded by + * promotedTarget's own occlusion stage before it reaches `resolved`, so it looks identical to + * "not found yet"; detecting it here ends the loop on this poll instead of spending the budget on a + * target that will never stop being covered. + */ +async function pollSelectorReadinessOnce( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + previousPoll: InteractionSnapshot | undefined, +): Promise { + const attempt = await attemptSelectorResolution(runtime, options, selectorExpression, params); + if (previousPoll) inheritPostGestureOutcome(previousPoll.snapshot, attempt.capture.snapshot); + if (attempt.resolved?.node.rect) return attempt; + const quality = attempt.capture.snapshot.snapshotQuality; + if (isSparseSnapshotQualityVerdict(quality)) { + throw new AppError( + 'COMMAND_FAILED', + `Selector ${selectorExpression} was not found in a sparse capture; the tree cannot prove it absent`, + { + reason: INTERACTION_ERROR_REASONS.captureSparse, + snapshotQuality: quality, + hint: 'Re-run after the screen settles, or capture a snapshot to inspect the tree.', + }, + ); + } + const covered = await detectCoveredSelectorTarget({ + runtime, + nodes: attempt.capture.snapshot.nodes, + selectorExpression, + action: params.action, + }); + if (covered) throw covered; + return attempt; +} + +/** + * The budget is spent or a capture stalled at the deadline. When no poll ever observed the tree, the + * last ridden-out capture error is the cause and is raised as such; otherwise the ordinary + * selector failure carries the readiness evidence. + */ +async function readinessExhaustedFailure( + runtime: AgentDeviceRuntime, + selectorExpression: string, + params: ResolveInteractionTargetParams, + observed: Extract< + Awaited>>, + { kind: 'expired' | 'stalled' } + >, +): Promise { + if (observed.last === undefined && observed.lastError !== undefined) return observed.lastError; + const readiness: SelectorReadinessDetails = { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: observed.kind, + }; + const failure = await selectorInteractionFailure({ + runtime, + nodes: observed.last?.capture.snapshot.nodes ?? [], + selectorExpression, + action: params.action, + resolved: observed.last?.resolved ?? null, + }); + failure.details = { ...failure.details, readiness }; + return failure; +} + +/** + * `promotedTarget`'s readiness budget: the target may not exist yet, so a plain no-match (no + * resolution, or a resolution with no usable rect) keeps polling under `schedule` instead of + * refusing on the first capture. Every other outcome stays terminal exactly as a single attempt + * would produce it — an ambiguity throw from `resolveActionSelector`, or a capture error the loop + * does not ride out, ends the loop on this same poll, and is rethrown unchanged. Occlusion, + * off-screen, non-hittable, and keyboard refusals are not judged here: they run once, after this + * loop returns a rect-bearing resolution, as `runInteractionPipelineStages` does. + * + * The first poll is unbounded, so a caller whose first capture already matches pays the + * one-or-two-capture cost of a single attempt. + * + * ADR 0011 registry anchor: interaction-guarantees.ts cites this as the runtime-selector + * `targetReadiness` `via` symbol. + */ +export async function pollForSelectorReadiness( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + schedule: ReadinessSchedule, +): Promise { + const signal = options.signal ?? runtime.signal; + let previousPoll: InteractionSnapshot | undefined; + const observed = await observeUntil({ + capture: async (pollSignal) => { + const attempt = await pollSelectorReadinessOnce( + runtime, + { ...options, signal: pollSignal }, + selectorExpression, + params, + previousPoll, + ); + previousPoll = attempt.capture; + return attempt; + }, + verdict: (latest) => + latest.resolved && latest.resolved.node.rect + ? { kind: 'done', result: { capture: latest.capture, resolved: latest.resolved } } + : { kind: 'continue' }, + schedule: { + ...schedule, + // The poll signal reaches the platform as CaptureSnapshotInput.signal, which the snapshot + // binding joins (captureSnapshotSignal): the same per-capture cancellation `wait` relies on. + captureDeadline: 'cancel', + }, + rideOut: isUnreadableCaptureContentError, + ...(signal ? { signal } : {}), + ...(runtime.clock ? { clock: runtime.clock } : {}), + phase: 'interaction_target_readiness', + }); + if (observed.kind === 'done') return observed.result; + if (observed.kind === 'failed') throw withSparseReadiness(observed.error, observed); + throw await readinessExhaustedFailure(runtime, selectorExpression, params, observed); +} + +function withSparseReadiness( + error: unknown, + observed: { polls: readonly unknown[]; waitedMs: number }, +): unknown { + if ( + !(error instanceof AppError) || + error.details?.reason !== INTERACTION_ERROR_REASONS.captureSparse + ) { + return error; + } + const readiness: SelectorReadinessDetails = { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: 'sparse', + }; + error.details = { ...error.details, readiness }; + return error; +} diff --git a/src/commands/interaction/runtime/target-visibility-stages.test.ts b/src/commands/interaction/runtime/target-visibility-stages.test.ts new file mode 100644 index 0000000000..7b689f0334 --- /dev/null +++ b/src/commands/interaction/runtime/target-visibility-stages.test.ts @@ -0,0 +1,227 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { ref, selector } from './selector-read-utils.ts'; +import { throwIfOffscreenInteractionTarget } from './target-visibility-stages.ts'; +import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; +import type { Point } from '@agent-device/kernel/snapshot'; +import { + coveredByTabBarSnapshot, + createInteractionDevice, + duplicateCoveredLabelSnapshot, +} from './__tests__/test-utils/index.ts'; + +test('runtime press refuses a selector that resolves to an off-screen element', async () => { + // Closed-drawer shape: the only match sits fully left of the viewport. The + // @ref path already refuses this; the selector path must not silently tap + // out-of-viewport coordinates. + const offscreenSnapshot = makeSnapshotState([ + { + index: 0, + depth: 0, + type: 'Application', + rect: { x: 0, y: 0, width: 400, height: 800 }, + hittable: true, + }, + { + index: 1, + depth: 2, + parentIndex: 0, + type: 'Button', + label: 'Explore', + rect: { x: -320, y: 240, width: 300, height: 50 }, + hittable: true, + }, + ]); + const taps: unknown[] = []; + const device = createInteractionDevice(offscreenSnapshot, { + tap: async (_context, point) => { + taps.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.press(selector('label=Explore'), { session: 'default' }), + (error: unknown) => { + assert.ok(error instanceof Error); + assert.match(error.message, /off-screen element and is not safe to press/); + const details = (error as { details?: Record }).details; + assert.equal(details?.reason, 'offscreen_selector'); + // #1366: the closed-drawer shape sits fully left of the viewport, so the + // hint names `scroll left` and steers back through the same selector. + assert.equal(details?.scrollDirection, 'left'); + assert.match(String(details?.hint), /scroll left/i); + assert.match(String(details?.hint), /selector/i); + return true; + }, + ); + assert.equal(taps.length, 0); +}); + +test('runtime press names a direction for a partial clip whose center is off-screen', async () => { + // #1366 regression: the row still OVERLAPS the viewport (top edge inside), so + // the rect-vs-viewport form yields no direction — but its tap-point center is + // below the bottom edge, which is what the visibility guard rejects. The hint + // must still name `scroll down` rather than falling back to the generic phrasing. + const partialClipSnapshot = makeSnapshotState([ + { + index: 0, + depth: 0, + type: 'Application', + rect: { x: 0, y: 0, width: 400, height: 800 }, + hittable: true, + }, + { + index: 1, + depth: 2, + parentIndex: 0, + type: 'Button', + label: 'Cash', + rect: { x: 20, y: 790, width: 200, height: 44 }, + hittable: true, + }, + ]); + const taps: unknown[] = []; + const device = createInteractionDevice(partialClipSnapshot, { + tap: async (_context, point) => { + taps.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.press(selector('label=Cash'), { session: 'default' }), + (error: unknown) => { + const details = (error as { details?: Record }).details; + assert.equal(details?.reason, 'offscreen_selector'); + assert.equal(details?.scrollDirection, 'down'); + assert.match(String(details?.hint), /scroll down/i); + // #1366 recovery must be bounded. `--until` is what bounds it now: it checks the same + // selector between passes, so the hint names one command rather than a manual step loop. + assert.match(String(details?.hint), /scroll down --until 'label=Cash'/); + assert.match(String(details?.hint), /stops on the target/i); + return true; + }, + ); + assert.equal(taps.length, 0); +}); + +test('runtime click rejects refs covered by floating overlays', async () => { + const calls: Point[] = []; + const device = createInteractionDevice(coveredByTabBarSnapshot(), { + tap: async (_context, point) => { + calls.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.click(ref('@e2'), { session: 'default' }), + /Ref @e2 is covered by another visible element/, + ); + assert.deepEqual(calls, []); +}); + +test('runtime selector interactions skip covered matches when an uncovered duplicate exists', async () => { + const calls: Point[] = []; + const device = createInteractionDevice(duplicateCoveredLabelSnapshot(), { + tap: async (_context, point) => { + calls.push(point); + }, + }); + + const result = await device.interactions.click(selector('label="Save draft"'), { + session: 'default', + }); + + assert.equal(result.kind, 'selector'); + assert.equal(result.node?.ref, 'e2'); + assert.deepEqual(calls, [{ x: 86, y: 142 }]); +}); + +// #1542: throwIfOffscreenInteractionTarget is exported for ADR 0011 registry +// honesty (interaction-guarantees.ts's `offscreen` cells point their `via` +// here); this direct-import test is its real consumer, mirroring +// tryResolveRefNode in ref-target-resolution.test.ts. End-to-end rescue/refuse coverage through the +// public click/press surface lives in offscreen-double-check.test.ts. +function fakeOffscreenFailure() { + return { + message: 'off-screen', + details: { reason: 'test' }, + hint: () => 'scroll toward it', + }; +} + +test('throwIfOffscreenInteractionTarget: an on-screen node passes through unchanged', async () => { + const device = createInteractionDevice(makeSnapshotState([])); + const nodes = makeSnapshotState([ + { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, + { + index: 1, + depth: 1, + parentIndex: 0, + type: 'Button', + rect: { x: 20, y: 20, width: 40, height: 40 }, + }, + ]).nodes; + + const result = await throwIfOffscreenInteractionTarget( + device, + { session: 'default' }, + nodes[1]!, + nodes, + fakeOffscreenFailure(), + ); + + assert.equal(result, nodes[1]); +}); + +test('throwIfOffscreenInteractionTarget: off-screen + backend confirms -> returns the node patched with the LIVE rect', async () => { + const device = createInteractionDevice(makeSnapshotState([]), { + confirmOffscreenTargetVisible: async () => ({ x: 30, y: 30, width: 40, height: 40 }), + }); + const nodes = makeSnapshotState([ + { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, + { + index: 1, + depth: 1, + parentIndex: 0, + type: 'Button', + rect: { x: 20, y: 2000, width: 40, height: 40 }, + }, + ]).nodes; + + const result = await throwIfOffscreenInteractionTarget( + device, + { session: 'default' }, + nodes[1]!, + nodes, + fakeOffscreenFailure(), + ); + + assert.deepEqual(result.rect, { x: 30, y: 30, width: 40, height: 40 }); + assert.equal(result.index, nodes[1]!.index); +}); + +test('throwIfOffscreenInteractionTarget: off-screen + no rescue -> throws with the supplied failure shape', async () => { + const device = createInteractionDevice(makeSnapshotState([])); + const nodes = makeSnapshotState([ + { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, + { + index: 1, + depth: 1, + parentIndex: 0, + type: 'Button', + rect: { x: 20, y: 2000, width: 40, height: 40 }, + }, + ]).nodes; + + await assert.rejects( + () => + throwIfOffscreenInteractionTarget( + device, + { session: 'default' }, + nodes[1]!, + nodes, + fakeOffscreenFailure(), + ), + /off-screen/, + ); +}); diff --git a/src/commands/interaction/runtime/target-visibility-stages.ts b/src/commands/interaction/runtime/target-visibility-stages.ts new file mode 100644 index 0000000000..5963dd1a42 --- /dev/null +++ b/src/commands/interaction/runtime/target-visibility-stages.ts @@ -0,0 +1,219 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { + Point, + SnapshotKeyboardBandFact, + SnapshotNode, + SnapshotState, +} from '@agent-device/kernel/snapshot'; +import { normalizeRef } from '@agent-device/kernel/snapshot'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { + runNodePipelineStages, + type SelectorPipelineHooks, +} from '@agent-device/selectors/selector-pipeline'; +import type { SelectorPipelinePolicy } from '@agent-device/selectors/selector-pipeline-policy'; +import { createSnapshotVisibility } from '@agent-device/contracts/snapshot'; +import { + classifyOffscreenScrollDirection, + type OffscreenScrollDirection, +} from '@agent-device/capture-kit/mobile-snapshot-semantics'; +import { toBackendContext } from '../../runtime-common.ts'; +import { assertTapTargetClearOfVisibleKeyboard } from './keyboard-occlusion.ts'; +import { interactionVerb } from './interaction-verb.ts'; +import type { InteractionAction } from './interaction-resolution-request.ts'; + +/** + * The one construction site for "covered by another visible element" refusals, shared by the + * node-stage runner below (ref, selector, native-ref preflight) and `selector-readiness.ts`'s + * covered-target diagnosis probe. + */ +export function buildCoveredInteractionError(params: { + label: string; + node: SnapshotNode; + action: InteractionAction; + selector?: string; +}): AppError { + return new AppError( + 'COMMAND_FAILED', + `${params.label} is covered by another visible element and cannot ${interactionVerb(params.action)} safely`, + { + hint: 'Use a different visible target, scroll it clear of the overlay, or inspect with snapshot/screenshot before retrying.', + ...(params.selector ? { selector: params.selector } : {}), + ref: `@${params.node.ref}`, + interactionBlocked: params.node.interactionBlocked, + }, + ); +} + +/** + * Every node stage this action's row declares, plus the covered and keyboard refusals the + * interaction runtime owns. Which stages run is the row's decision; every acting path — selector, + * ref, and the native-ref preflight — enters them here, which is what keeps the native-ref fast + * path from succeeding on a target the shared rules would refuse. + * + * Each path hands in the resolver that produces the point it taps with, and taps the point that + * comes back, so the keyboard guard measures the coordinate the interaction is actually made of and no + * path derives a second one. A path whose point can fail to exist — the native-ref fast path taps by + * ref, reading the rect center the platform aims at — says so in its resolver's return type. + */ +export async function runInteractionPipelineStages(params: { + policy: SelectorPipelinePolicy; + nodes: SnapshotState['nodes']; + /** The keyboard band `nodes`' capture measured, when it measured one (#2660). */ + keyboard?: SnapshotKeyboardBandFact; + node: SnapshotNode; + action: InteractionAction; + label: string; + hooks: SelectorPipelineHooks; + resolveTapPoint: (node: SnapshotNode) => TPoint; +}): Promise<{ node: SnapshotNode; tapPoint: TPoint }> { + const target = await runNodePipelineStages( + params.policy, + params.nodes, + params.node, + params.hooks, + ); + if (target.kind === 'occluded') { + throw buildCoveredInteractionError({ + label: params.label, + node: target.node, + action: params.action, + }); + } + const tapPoint = params.resolveTapPoint(target.node); + assertTapTargetClearOfVisibleKeyboard({ + nodes: params.nodes, + node: target.node, + action: params.action, + label: params.label, + ...(params.keyboard ? { keyboard: params.keyboard } : {}), + tapPoint, + }); + return { node: target.node, tapPoint }; +} + +/** + * The off-screen stage's refusal shape. Reached only through the pipeline + * owner, and only for rows whose off-screen stage refuses — the row's decision + * is made there, so this builds the message and never re-decides. + */ +type OffscreenStageParams = { action: InteractionAction; pipeline: SelectorPipelinePolicy }; + +// Selector parity for the @ref off-screen guard: without it, a selector +// resolving to a closed drawer/carousel item "succeeds" by tapping coordinates +// outside the viewport (observed as `Tapped (-161, 265)` against Bluesky's +// closed drawer) while the same node via @ref is refused. +export async function assertVisibleSelectorTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + selector: string, + { action }: OffscreenStageParams, +): Promise { + return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { + message: `Selector ${selector} resolved to an off-screen element and is not safe to ${action}`, + details: { reason: 'offscreen_selector', selector }, + // A selector re-resolves against a fresh snapshot on every attempt, so the + // recovery is: move the named direction, then retry THIS selector — no + // separate snapshot step, and no @ref (a scroll expires the ref frame, + // #1366). `--until` is that whole loop as one command: it checks the same + // selector between passes, which is also what keeps a large step from + // overshooting, so the hint no longer has to trade distance for accuracy. + hint: (direction) => + `${scrollRevealClause(direction, selector)} then retry ${action} with the same selector. --until checks the selector between passes, so it stops on the target rather than sailing past it. If it is inside a closed drawer or another tab, open that container first.`, + }); +} + +export async function assertVisibleRefTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + refInput: string, + { action }: OffscreenStageParams, +): Promise { + return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { + message: `Ref ${refInput} is off-screen and not safe to ${action}`, + details: { reason: 'offscreen_ref', ref: normalizeRef(refInput) }, + // The scroll that reveals the target expires the ref frame (#1366, ADR + // 0014), so retrying this @ref would be rejected next. Steer to a selector, + // which re-resolves against a fresh snapshot and bypasses the ref-frame guard + // — and which `--until` can then check between passes. + hint: (direction) => + `${scrollRevealClause(direction, null)} then retry ${action} with a selector (e.g. text=/id=) rather than this @ref — the scroll expires the ref frame, so re-run snapshot -i before reusing any @ref.`, + }); +} + +/** + * Shared lead-in for both off-screen hints: the one command that reveals the target. + * + * When the geometry names a direction AND the caller has a selector to check, this is a complete + * `scroll --until ` — one request that stops on the target instead of the + * scroll-then-look-again loop the hint used to prescribe. Without a selector to check (an @ref + * refusal) or without a single reveal direction (off more than one edge), it degrades to naming + * the move and leaves the stop condition to the caller's own next step. + */ +function scrollRevealClause( + direction: OffscreenScrollDirection | null, + selector: string | null, +): string { + if (!direction) return 'Scroll toward it,'; + if (!selector) return `Scroll ${direction} toward it,`; + return `Run scroll ${direction} --until '${selector}' to bring it on screen,`; +} + +// Full on-screen visibility (not only the effective-viewport form): items inside an +// off-screen scrollable container (closed drawer) must also count as +// off-screen, not just items scrolled out of an on-screen container. +// +// #1542: once the bulk tree says off-screen, the guard gives iOS one chance +// to rescue a FALSE refusal via the optional backend.confirmOffscreenTargetVisible +// hook — a stale/corrupted bulk tree can say off-screen while the app is +// visually fine (zero cost on the accept path; runs only here). A confirmed +// rescue returns the node PATCHED WITH THE LIVE RECT: the caller must act on +// that returned node, never the original, because in the frozen-bulk-tree +// manifestation the original rect can be stale even when the rescue verdict +// is correct — tapping it would silently land at the wrong coordinate. The +// hook fails closed (null) on anything short of a positive confirmation, so +// a genuine refusal, or any backend without the hook, is unchanged. +// +// Exported (not just for callers here) for ADR 0011 registry honesty: +// interaction-guarantees.ts's `offscreen` cells point their `via` at this +// function, not at the contracts predicate alone, since this is the actual +// end-to-end enforcement point. +export async function throwIfOffscreenInteractionTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + failure: { + message: string; + details: Record; + hint: (direction: OffscreenScrollDirection | null) => string; + }, +): Promise { + const visibility = createSnapshotVisibility(nodes); + const viewport = node.rect ? visibility.resolveEffectiveViewport(node) : null; + if (!node.rect || !viewport || visibility.isVisibleOnScreen(node)) return node; + const rootViewport = visibility.resolveViewport(node.rect); + const liveRect = await runtime.backend.confirmOffscreenTargetVisible?.( + toBackendContext(runtime, options), + node, + rootViewport, + ); + if (liveRect) return { ...node, rect: liveRect }; + // The direction that scrolls this off-screen target into view. Named in the + // hint (and surfaced as a machine-readable detail) so the recovery is a single + // deterministic move instead of a guess (#1366). Derived from the same + // boundary the rejection above used, so partial clips and off-screen + // containers get a direction too, not just fully-scrolled-out items. + const scrollDirection = classifyOffscreenScrollDirection(node, visibility); + throw new AppError('COMMAND_FAILED', failure.message, { + ...failure.details, + rect: node.rect, + viewport, + ...(scrollDirection ? { scrollDirection } : {}), + hint: failure.hint(scrollDirection), + }); +} diff --git a/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts b/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts index 453d3ccd02..1e49f6d0d8 100644 --- a/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts +++ b/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts @@ -11,6 +11,7 @@ import { splitReplayCommandRequest, } from '@agent-device/replay-port/replay-dispatch-envelope'; import type { ReplayCommand } from '@agent-device/replay-port/command-types'; +import type { ObservationClock } from '@agent-device/capture-kit/observe-until'; export type ReplayCommandTestInput = Readonly<{ req: DaemonRequest; @@ -20,6 +21,8 @@ export type ReplayCommandTestInput = Readonly<{ invoke: DaemonInvokeFn; tracePath?: string; onStep?: ReplayTestAttemptStepSink; + /** Paces the target-readiness wait; default: `instantReplayClock`. */ + clock?: ObservationClock; }>; /** @@ -28,12 +31,12 @@ export type ReplayCommandTestInput = Readonly<{ * test's `invoke` sees the same `DaemonRequest` the daemon would. */ export function replayCommandForTest(params: ReplayCommandTestInput): ReplayCommand { - const { req, sessionName, logPath, sessionStore, invoke, tracePath, onStep } = params; + const { req, sessionName, logPath, sessionStore, invoke, tracePath, onStep, clock } = params; return { ...splitReplayCommandRequest(req), session: createReplaySession(sessionName, logPath, sessionStore), invoke: replayInvokeOverDispatch(invoke, req), - dependencies: replayDaemonDependencies, + dependencies: { ...replayDaemonDependencies, clock: clock ?? instantReplayClock() }, ...(tracePath === undefined ? {} : { tracePath }), ...(onStep === undefined ? {} : { onStep }), }; @@ -42,3 +45,14 @@ export function replayCommandForTest(params: ReplayCommandTestInput): ReplayComm export function runReplayForTest(params: ReplayCommandTestInput): Promise { return runReplayCommand(replayCommandForTest(params)); } + +/** Target-readiness waits advance this clock instead of sleeping, so a unit replay spends no wall time. */ +function instantReplayClock(): ObservationClock { + let nowMs = Date.now(); + return { + now: () => nowMs, + sleep: async (ms) => { + nowMs += ms; + }, + }; +} diff --git a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts index 2234c80b44..e50b53df9f 100644 --- a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts +++ b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts @@ -3,9 +3,20 @@ import { expect, test } from 'vitest'; import { makeIosSession } from '../../../__tests__/test-utils/session-factories.ts'; import { recordActionEntry } from '../../session-action-recorder.ts'; import type { DaemonRequest } from '../../daemon-request.ts'; -import { invokeReplayAction } from '@agent-device/replay-port/session-replay-action-runtime'; +import { + invokeReplayAction, + replayStepReadinessSchedule, +} from '@agent-device/replay-port/session-replay-action-runtime'; +import { + SELECTOR_PIPELINE_POLICIES, + readinessScheduleFor, +} from '@agent-device/selectors/selector-pipeline-policy'; import { replayDaemonDependencies } from '../../handlers/session-replay-command.ts'; import { resolveReplayAction } from '@agent-device/ad-script'; +import { + commandAcceptsReadinessBudget, + commandDescriptors, +} from '@agent-device/command-registry/registry'; const REPLAY_REQUEST: DaemonRequest = { token: 'token', @@ -57,3 +68,129 @@ test.each(['', ' '])( expect(session.actions[0]?.result?.text).toBe('${PASSWORD}'); }, ); + +const READINESS_BUDGETED_COMMANDS = commandDescriptors + .map((descriptor) => descriptor.name) + .filter((command) => commandAcceptsReadinessBudget(command)); + +// A replay step has no CLI flag to carry a readiness budget, so `buildReplayActionFlags` defaults +// one in for every readiness-budgeted command. +test.each(READINESS_BUDGETED_COMMANDS)( + 'replay defaults readinessTimeoutMs onto a dispatched %s step', + async (command) => { + const action: SessionAction = { ts: 0, command, positionals: ['label="Continue"'], flags: {} }; + let dispatchedFlags: Record | undefined; + const response = await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + expect(response.ok).toBe(true); + expect(dispatchedFlags?.readinessTimeoutMs).toBe(2_000); + }, +); + +// The dispatch resolves a press/click/longpress target under the promotedTarget row with the +// readinessTimeoutMs it receives; the pre-dispatch gate must poll under that same schedule. +test.each([ + ...READINESS_BUDGETED_COMMANDS.map((command) => ({ command, flags: {} })), + { command: 'click', flags: { readinessTimeoutMs: 700 } }, + { command: 'press', flags: { readinessTimeoutMs: 9_000 } }, +])('the replay gate and the dispatch poll a $command step under one schedule', async (step) => { + const action: SessionAction = { + ts: 0, + command: step.command, + positionals: ['label="Continue"'], + flags: step.flags, + }; + let dispatchedFlags: Record | undefined; + await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + const dispatchSchedule = readinessScheduleFor( + SELECTOR_PIPELINE_POLICIES.promotedTarget.poll, + dispatchedFlags?.readinessTimeoutMs as number | undefined, + ); + expect(dispatchSchedule).toBeDefined(); + expect(await replayStepReadinessSchedule(REPLAY_REQUEST.flags, action)).toEqual(dispatchSchedule); +}); + +test('replay keeps a readinessTimeoutMs the step already carries instead of overwriting it', async () => { + const action: SessionAction = { + ts: 0, + command: 'press', + positionals: ['label="Continue"'], + flags: { readinessTimeoutMs: 500 }, + }; + let dispatchedFlags: Record | undefined; + const response = await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + expect(response.ok).toBe(true); + expect(dispatchedFlags?.readinessTimeoutMs).toBe(500); +}); + +test('replay never defaults readinessTimeoutMs onto a non-acting step', async () => { + const action: SessionAction = { + ts: 0, + command: 'wait', + positionals: ['label="Continue"'], + flags: {}, + }; + let dispatchedFlags: Record | undefined; + const response = await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + expect(response.ok).toBe(true); + expect(dispatchedFlags?.readinessTimeoutMs).toBeUndefined(); +}); diff --git a/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts b/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts index c822ffd05b..f129fc90fb 100644 --- a/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts +++ b/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts @@ -2,6 +2,7 @@ import path from 'node:path'; import { mkdtempForTestSync } from '../../../__tests__/test-utils/tmp-dir.ts'; import { makeIosAppSession } from '../../../__tests__/test-utils/session-factories.ts'; import { SessionStore } from '../../session-store.ts'; +import type { ObservationClock } from '@agent-device/capture-kit/observe-until'; import type { DaemonInvokeFn, DaemonRequest, DaemonResponse } from '../../daemon-request.ts'; import { runReplayForTest } from '../replay-runtime/replay-command-fixture.ts'; import { @@ -33,9 +34,14 @@ export type ReplayScriptScene = ReplaySessionScene & /** * Runs the script through `runReplayCommand`. Each request is appended to * `invoked` before `invoke` answers it; without `invoke` every request succeeds - * with empty data. + * with empty data. `clock` paces its target-readiness waits; `requestId` lets a test cancel the + * replay through the request registry. */ - replay(options?: { invoke?: DaemonInvokeFn }): Promise; + replay(options?: { + invoke?: DaemonInvokeFn; + clock?: ObservationClock; + requestId?: string; + }): Promise; }>; /** An iOS app session plus a written `.ad` script, ready to replay. */ @@ -47,9 +53,13 @@ export function replayScriptScene(prefix: string, lines: string[]): ReplayScript ...scene, filePath, invoked, - replay: ({ invoke } = {}) => + replay: ({ invoke, clock, requestId } = {}) => runReplayForTest({ - req: baseReplayRequest({ positionals: [filePath] }), + req: baseReplayRequest({ + positionals: [filePath], + ...(requestId === undefined ? {} : { meta: { requestId } }), + }), + ...(clock === undefined ? {} : { clock }), sessionName: scene.sessionName, logPath: scene.logPath, sessionStore: scene.sessionStore, diff --git a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts index ca97616a8f..ddffb39004 100644 --- a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts +++ b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts @@ -28,7 +28,18 @@ vi.mock('@agent-device/host-kit/retry', async (importOriginal) => { return { ...actual, sleep: vi.fn(async () => {}) }; }); +vi.mock('@agent-device/host-kit/diagnostics', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, emitDiagnostic: vi.fn(actual.emitDiagnostic) }; +}); + import { AppError } from '@agent-device/kernel/errors'; +import { + clearRequestCanceled, + markRequestCanceled, + registerRequestAbort, +} from '@agent-device/host-kit/request'; +import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; import { legacyDispatchCapture, resetLegacySnapshotCapture, @@ -51,8 +62,16 @@ const mockCaptureSnapshotWithInteractor = vi.mocked(captureSnapshotWithInteracto beforeEach(() => { resetLegacySnapshotCapture(mockCaptureSnapshotWithInteractor); + vi.mocked(emitDiagnostic).mockClear(); }); +function readinessDiagnostics(): unknown[] { + return vi + .mocked(emitDiagnostic) + .mock.calls.filter(([event]) => event.phase === 'interaction_target_readiness') + .map(([event]) => event.data); +} + test('an unannotated action executes unchanged (old-script pass-through)', async () => { const scene = replayScriptScene('agent-device-replay-target-verify-passthrough-', [ 'click id="save"', @@ -80,6 +99,9 @@ test('a verified target proceeds to dispatch the action', async () => { expect(response.ok).toBe(true); expect(scene.invoked.map((req) => req.command)).toEqual(['click']); expect(scene.invoked[0]?.positionals).toEqual(['id="save"']); + // Present on the gate's first capture: the dispatch keeps the step's whole budget. + expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(2_000); + expect(readinessDiagnostics()).toEqual([]); }); test('a verified drag guards both source and destination before dispatch', async () => { @@ -177,6 +199,119 @@ test('a selector-miss divergence blocks dispatch and never sends the action', as expect(targetBinding.matchCount).toBe(0); expect(targetBinding.observed).toBeUndefined(); expect(targetBinding.recorded).toEqual({ id: 'save', role: 'button', label: 'Save' }); + // click waits for its target, so the gate re-captured before refusing. + expect(mockDispatchCommand.mock.calls.length).toBeGreaterThan(1); + expect(response.error.details?.readiness).toMatchObject({ waitedMs: 2_000, end: 'expired' }); +}); + +test('an annotated click whose target renders on the second capture waits for it and dispatches', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-late-render-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + + mockDispatchCommand.mockResolvedValueOnce(emptyCapture()).mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay(); + + expect(response.ok).toBe(true); + expect(scene.invoked.map((req) => req.command)).toEqual(['click']); + expect(scene.invoked[0]?.internal?.replayTargetGuard).toMatchObject({ + identity: { id: 'save', role: 'button', label: 'Save' }, + }); + expect(mockDispatchCommand).toHaveBeenCalledTimes(2); + expect(readinessDiagnostics()).toEqual([ + { polls: 2, waitedMs: 200, end: 'done', command: 'click' }, + ]); +}); + +test('the dispatch gets only the readiness budget the gate left', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-shared-budget-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + + // Eight misses, one interval apart, before the target renders 1.6 s into the 2 s budget. + for (let miss = 0; miss < 8; miss += 1) mockDispatchCommand.mockResolvedValueOnce(emptyCapture()); + mockDispatchCommand.mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay(); + + expect(response.ok).toBe(true); + expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(400); +}); + +test('the budget the gate hands the dispatch counts from the end of its first capture', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-budget-origin-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + let nowMs = 0; + const clock = { + now: () => nowMs, + sleep: async (ms: number) => { + nowMs += ms; + }, + }; + + // A 500 ms first capture, then seven more misses one interval apart: 1.4 s of the 2 s budget. + mockDispatchCommand.mockImplementationOnce(async () => { + nowMs += 500; + return emptyCapture(); + }); + for (let miss = 0; miss < 7; miss += 1) mockDispatchCommand.mockResolvedValueOnce(emptyCapture()); + mockDispatchCommand.mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay({ clock }); + + expect(response.ok).toBe(true); + expect(readinessDiagnostics()).toEqual([ + { polls: 9, waitedMs: 2_100, end: 'done', command: 'click' }, + ]); + expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(400); +}); + +test('cancelling the request during the gate wait ends it at the next poll with no dispatch', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-gate-cancel-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + const requestId = 'replay-gate-cancel'; + const registration = registerRequestAbort(requestId); + mockDispatchCommand.mockResolvedValueOnce(emptyCapture()).mockImplementationOnce(async () => { + markRequestCanceled(requestId); + return emptyCapture(); + }); + mockDispatchCommand.mockResolvedValue(saveButtonCapture()); + + try { + const response = await scene.replay({ requestId }); + + expect(response.ok).toBe(false); + if (response.ok) return; + expect(response.error.details?.reason).toBe('request_canceled'); + expect(mockDispatchCommand).toHaveBeenCalledTimes(2); + expect(scene.invoked).toEqual([]); + } finally { + clearRequestCanceled(requestId, registration); + } +}); + +test('an annotated step whose command does not wait for its target refuses a selector miss on one capture', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-no-wait-', [ + SAVE_ANNOTATION, + 'fill id="save" "hello"', + ]); + + mockDispatchCommand.mockResolvedValueOnce(emptyCapture()).mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay(); + + expect(scene.invoked.length).toBe(0); + expect(response.ok).toBe(false); + if (response.ok) return; + const divergence = response.error.details?.divergence as Record; + expect(divergence.kind).toBe('selector-miss'); }); test('an identity-mismatch divergence reports matchCount and an observed identity', async () => { diff --git a/src/daemon/interaction/internal/interaction-touch-press.ts b/src/daemon/interaction/internal/interaction-touch-press.ts index f5649912b6..cd6287edea 100644 --- a/src/daemon/interaction/internal/interaction-touch-press.ts +++ b/src/daemon/interaction/internal/interaction-touch-press.ts @@ -177,6 +177,7 @@ async function runTargetedTouchInteraction(params: { return await runtime.interactions.longPress(target, { ...shared, durationMs: params.durationMs, + readinessTimeoutMs: flags?.readinessTimeoutMs, }); case 'hover': return await runtime.interactions.hover(target, shared); @@ -210,6 +211,7 @@ function pressRuntimeOptions( jitterPx: flags?.jitterPx, doubleTap: flags?.doubleTap, verify: flags?.verify, + readinessTimeoutMs: flags?.readinessTimeoutMs, // Only click/press take it: `find` dispatches click and fill, never // longpress or hover, so declaring it on their options would be an // unconsumed claim (#1649 review). diff --git a/src/mcp/__tests__/command-tools-operator-inputs.test.ts b/src/mcp/__tests__/command-tools-operator-inputs.test.ts index 64b6fcbf6f..91b2633b0d 100644 --- a/src/mcp/__tests__/command-tools-operator-inputs.test.ts +++ b/src/mcp/__tests__/command-tools-operator-inputs.test.ts @@ -86,6 +86,33 @@ test('MCP refuses every explicit operator-owned argument with guidance', async ( assert.deepEqual(calls, [], 'a refused operator input must never reach the command route'); }); +test('MCP neither advertises nor admits the operator-only readiness budget', async () => { + for (const tool of listCommandTools()) { + const properties = tool.inputSchema.properties ?? {}; + assert.equal('readinessTimeoutMs' in properties, false, `${tool.name} advertises it`); + } + const calls: unknown[] = []; + const executor = createCommandToolExecutor({ + createClient: () => ({}) as AgentDeviceClient, + runCommand: async (_client, name, input) => { + calls.push({ name, input }); + return {}; + }, + }); + + const result = await executor.execute('press', { + target: { kind: 'selector', selector: 'label=Continue' }, + readinessTimeoutMs: 2_000, + }); + + assert.equal(result.isError, true); + assert.match( + result.content[0]?.text ?? '', + /readinessTimeoutMs is not accepted as a tool argument/, + ); + assert.deepEqual(calls, [], 'a refused operator input must never reach the command route'); +}); + test('MCP refuses an explicit daemonAuthToken argument with env guidance', async () => { const calls: unknown[] = []; const executor = createCommandToolExecutor({ diff --git a/test/integration/interaction-contract/runtime-selector.contract.test.ts b/test/integration/interaction-contract/runtime-selector.contract.test.ts index 07abb5349b..90c03366a8 100644 --- a/test/integration/interaction-contract/runtime-selector.contract.test.ts +++ b/test/integration/interaction-contract/runtime-selector.contract.test.ts @@ -19,6 +19,7 @@ import { nonHittableButtonSnapshot, RUNNER_CONTINUE_NODES, settledWelcomeSnapshot, + viewportOnlySnapshot, } from './fixtures.ts'; import { createContractDevice } from './runtime-harness.ts'; import { runnerSnapshotEntry, runnerTapEntry, withIosContractDaemon } from './daemon-harness.ts'; @@ -197,6 +198,35 @@ test(scenario('nonHittable'), async () => { assert.match(result.hint ?? '', /hittable: false/); }); +test(scenario('targetReadiness'), async () => { + const taps: Point[] = []; + let captures = 0; + const device = createContractDevice(viewportOnlySnapshot(), { + captureSnapshot: async () => { + captures += 1; + return { + // The first poll's interactive capture and its full-capture fallback both miss; the button + // appears only from the second poll on, so a loop that never polled again leaves the tap + // unsent. + snapshot: captures >= 3 ? continueButtonSnapshot() : viewportOnlySnapshot(), + }; + }, + tap: async (_context, point) => { + taps.push(point); + }, + }); + + const result = await device.interactions.press(selector('label=Continue'), { + session: 'default', + // Never CLI- or model-writable, so the scenario supplies it directly. + readinessTimeoutMs: 2_000, + }); + + assert.equal(result.kind, 'selector'); + assert.ok(captures >= 3, `expected at least 2 polls (>=3 captures), got ${captures}`); + assert.deepEqual(taps, [{ x: 60, y: 40 }]); +}); + test(scenario('verifyEvidence'), async () => { const device = createContractDevice(continueButtonSnapshot(), { tap: async () => ({ ok: true }), diff --git a/test/integration/interaction-contract/runtime-selector.coverage.ts b/test/integration/interaction-contract/runtime-selector.coverage.ts index b11f593905..0d968bc93f 100644 --- a/test/integration/interaction-contract/runtime-selector.coverage.ts +++ b/test/integration/interaction-contract/runtime-selector.coverage.ts @@ -29,4 +29,6 @@ export const RUNTIME_SELECTOR_COVERAGE = definePathCoverage('runtime-selector', 'runtime-selector resolutionDisclosure: a unique match discloses the unique runtime shape', 'runtime-selector resolutionDisclosure: an equivalent wrapper chain discloses matchCount, winnerDiagnostic, and structural equivalence', ], + targetReadiness: + 'runtime-selector targetReadiness: a target missing on the first capture resolves once a later poll observes it', }); diff --git a/test/integration/provider-scenarios/press-target-readiness.test.ts b/test/integration/provider-scenarios/press-target-readiness.test.ts new file mode 100644 index 0000000000..d98e43766a --- /dev/null +++ b/test/integration/provider-scenarios/press-target-readiness.test.ts @@ -0,0 +1,241 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { assertRpcError, assertRpcOk } from './assertions.ts'; +import { PROVIDER_SCENARIO_IOS_SIMULATOR } from './fixtures.ts'; +import { createProviderScenarioHarness, withProviderScenarioResource } from './harness.ts'; +import { + createAppleRunnerProviderFromTranscript, + createRecordingAppleToolProvider, + simctlDeviceLifecycleHandler, +} from './providers.ts'; +import { createProviderTranscript, type ProviderScenarioProviderEntry } from './transcript.ts'; + +// promotedTarget readiness (press/click/longpress poll for the target to exist and become +// actionable before refusing): end-to-end proof through the real daemon stack, not the plain +// runtime harness. The press route captures through the interaction backend, which keeps no +// selector capture cache, so every poll reaches the runner transcript below. + +const APP = 'com.example.app'; +const DEVICE_ID = PROVIDER_SCENARIO_IOS_SIMULATOR.id; + +const APPLICATION_ONLY_NODES = [ + { + index: 0, + type: 'Application', + label: 'Example', + rect: { x: 0, y: 0, width: 400, height: 800 }, + }, +]; + +const CONTINUE_BUTTON_NODES = [ + { + index: 0, + type: 'Application', + label: 'Example', + rect: { x: 0, y: 0, width: 400, height: 800 }, + }, + { + index: 1, + parentIndex: 0, + type: 'Button', + label: 'Continue', + hittable: true, + rect: { x: 100, y: 300, width: 200, height: 44 }, + }, +]; + +function snapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { + return { + command: 'ios.runner.snapshot', + deviceId: DEVICE_ID, + platform: 'apple', + result: { nodes, truncated: false }, + }; +} + +function repeatSnapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { + return { ...snapshotEntry(nodes), repeat: true }; +} + +function tapEntry(x: number, y: number): ProviderScenarioProviderEntry { + return { + command: 'ios.runner.tap', + deviceId: DEVICE_ID, + platform: 'apple', + result: { x, y }, + }; +} + +async function withPressReadinessDaemon( + entries: readonly ProviderScenarioProviderEntry[], + run: ( + daemon: Awaited>, + transcript: ReturnType, + ) => Promise, +): Promise { + const runnerTranscript = createProviderTranscript(entries); + const appleRunnerProvider = createAppleRunnerProviderFromTranscript( + runnerTranscript, + 'ios.runner', + ); + const appleTool = createRecordingAppleToolProvider({ + simctl: simctlDeviceLifecycleHandler('com.apple.CoreSimulator.SimRuntime.iOS-18-0', [ + { name: PROVIDER_SCENARIO_IOS_SIMULATOR.name, udid: DEVICE_ID }, + ]), + }); + + await withProviderScenarioResource( + async () => + await createProviderScenarioHarness({ + appleRunnerProvider: () => appleRunnerProvider, + appleToolProvider: () => appleTool.provider, + deviceInventoryProvider: async () => [PROVIDER_SCENARIO_IOS_SIMULATOR], + }), + async (daemon) => { + const open = await daemon.callCommand('open', [APP], { platform: 'ios', udid: DEVICE_ID }); + assertRpcOk(open); + await run(daemon, runnerTranscript); + }, + ); +} + +test('press waits for a selector missing on the first two captures, then taps once it appears', async () => { + await withPressReadinessDaemon( + [ + // Poll 1: interactive capture, then the interactive->full fallback — both miss because the + // button is not in the tree yet. + snapshotEntry(APPLICATION_ONLY_NODES), + snapshotEntry(APPLICATION_ONLY_NODES), + // Poll 2 onward: the button has appeared. + repeatSnapshotEntry(CONTINUE_BUTTON_NODES), + tapEntry(200, 322), + ], + async (daemon, transcript) => { + // Never CLI- or model-writable, so the scenario supplies it directly as a request flag. + const press = await daemon.callCommand('press', ['label=Continue'], { + readinessTimeoutMs: 2_000, + }); + const data = assertRpcOk(press); + assert.equal(data.x, 200); + assert.equal(data.y, 322); + + const commands = transcript.calls.map((call) => call.command); + const tapIndex = commands.indexOf('ios.runner.tap'); + assert.ok(tapIndex >= 0, 'expected the runner to receive a tap'); + assert.equal( + commands.filter((command) => command === 'ios.runner.tap').length, + 1, + 'expected exactly one tap dispatch', + ); + // Every snapshot call precedes the single tap: the loop never taps before a poll resolves. + assert.ok( + commands.slice(0, tapIndex).every((command) => command === 'ios.runner.snapshot'), + `expected only snapshot calls before the tap, got ${JSON.stringify(commands)}`, + ); + assert.ok( + tapIndex >= 3, + `expected at least 3 snapshot calls before the tap (interactive+fallback misses, then a resolved poll), got ${tapIndex}`, + ); + + transcript.assertComplete(); + }, + ); +}); + +test('press fails with the standard no-match error, carrying readiness evidence, when the target never appears', async () => { + await withPressReadinessDaemon([repeatSnapshotEntry(APPLICATION_ONLY_NODES)], async (daemon) => { + // No fake clock is wired into this daemon-composed runtime path (AgentDeviceRuntime.clock is + // never set by daemon composition), so this genuinely spends the ~2s promotedTarget readiness + // budget in wall-clock time before failing. + const press = await daemon.callCommand('press', ['label=Continue'], { + readinessTimeoutMs: 2_000, + }); + const error = assertRpcError(press, 'COMMAND_FAILED', /Selector did not match/); + const details = error.details as { + reason: unknown; + readiness: { polls: number; waitedMs: number; end: string }; + }; + assert.equal(details.reason, 'selector_not_found'); + assert.ok(details.readiness.polls >= 2, `expected >=2 polls, got ${details.readiness.polls}`); + assert.ok( + details.readiness.waitedMs >= 2000, + `expected >=2000ms waited, got ${details.readiness.waitedMs}`, + ); + assert.equal(details.readiness.end, 'expired'); + }); +}, 15_000); + +test('press resolving on the first capture costs exactly one snapshot call (zero readiness cost on the success path)', async () => { + await withPressReadinessDaemon( + [snapshotEntry(CONTINUE_BUTTON_NODES), tapEntry(200, 322)], + async (daemon) => { + const press = await daemon.callCommand('press', ['label=Continue']); + assertRpcOk(press); + }, + ); +}); + +// (d) Agent misses are usually a wrong selector, so fast feedback beats absorbing a render race. +// Without an explicit readinessTimeoutMs (never CLI- or model-writable), a miss takes the +// one-attempt path: exactly one capture-and-resolve attempt (the interactive-then-full-capture +// fallback, not the readiness loop's repeated polling), and no readiness evidence to attach. +test('press without a readinessTimeoutMs flag fails on the first capture attempt, with no readiness poll', async () => { + await withPressReadinessDaemon( + // Interactive capture, then the interactive->full fallback — both miss, exactly like poll 1 of + // the readiness loop above, because this IS that same one attempt, just never repeated. + [snapshotEntry(APPLICATION_ONLY_NODES), snapshotEntry(APPLICATION_ONLY_NODES)], + async (daemon, transcript) => { + const callsBeforePress = transcript.calls.length; + const press = await daemon.callCommand('press', ['label=Continue']); + const error = assertRpcError(press, 'COMMAND_FAILED', /Selector did not match/); + const details = error.details as { reason: unknown; readiness: unknown }; + assert.equal(details.reason, 'selector_not_found'); + assert.equal(details.readiness, undefined); + + const commands = transcript.calls.slice(callsBeforePress).map((call) => call.command); + assert.deepEqual(commands, ['ios.runner.snapshot', 'ios.runner.snapshot']); + transcript.assertComplete(); + }, + ); +}); + +test('press ends the wait at once on a sparse capture with capture_sparse, and never taps', async () => { + await withPressReadinessDaemon( + [ + { + command: 'ios.runner.snapshot', + deviceId: DEVICE_ID, + platform: 'apple', + repeat: true, + result: { + nodes: APPLICATION_ONLY_NODES, + truncated: false, + snapshotQuality: { + state: 'sparse', + backend: 'tree', + reasonCode: 'sparse-tree', + }, + }, + }, + ], + async (daemon, transcript) => { + const callsBeforePress = transcript.calls.length; + const startedAt = Date.now(); + const press = await daemon.callCommand('press', ['label=Continue'], { + readinessTimeoutMs: 2_000, + }); + const error = assertRpcError(press, 'COMMAND_FAILED', /sparse capture/); + const details = error.details as { + reason: unknown; + snapshotQuality: { state: string }; + readiness: { polls: number; end: string }; + }; + assert.equal(details.reason, 'capture_sparse'); + assert.equal(details.snapshotQuality.state, 'sparse'); + assert.equal(details.readiness.end, 'sparse'); + assert.ok(Date.now() - startedAt < 1_500, 'a sparse capture must not burn the budget'); + const commands = transcript.calls.slice(callsBeforePress).map((call) => call.command); + assert.ok(!commands.includes('ios.runner.tap'), 'no tap on a sparse capture'); + }, + ); +}); diff --git a/website/docs/docs/client-api.md b/website/docs/docs/client-api.md index 2bcbc00342..08cd4d31de 100644 --- a/website/docs/docs/client-api.md +++ b/website/docs/docs/client-api.md @@ -288,6 +288,17 @@ await client.command.fold({ `fold` accepts either `pose` or `keyframes`. Keyframes use linear interpolation at roughly 60 updates per second; repeat an angle to hold it. Timestamps must start at zero and increase strictly, with 2–64 frames and a final timestamp no greater than 60,000ms. Angles must be finite and between 0° and 180°. The final timestamp bounds motion, excluding helper preparation and final hinge verification. A custom final angle is verified within 0.5°; interior angles must also settle. Cancellation stops the motion at its current angle. Re-snapshot afterwards, including after interrupted motion. +`press`, `click`, and `longpress` take `readinessTimeoutMs`. With it, the command waits up to that many milliseconds for a target that is not on screen yet, then performs the requested interaction. Without it, the command looks once and fails at once, which is the right choice for an agent that most often misses because the selector is wrong. Use it in scripted flows, where a step can land a render early: + +```ts +await client.interactions.press({ + selector: 'label="Continue"', + readinessTimeoutMs: 2_000, +}); +``` + +The wait is capped at 2 seconds and covers only a target that has not appeared. When the target is still missing after the wait, the error carries `error.details.readiness` with `waitedMs`, `polls`, and `end` (`expired` or `stalled`). A capture that shows an empty accessibility tree ends the wait at once with `capture_sparse` and `readiness.end: sparse`. A covered, off-screen, or ambiguous target fails at once, and a screen that stays unreadable for the whole wait fails with its own error; neither carries `readiness`. `readinessTimeoutMs` is not an MCP tool argument and has no CLI flag. + Vega OS client support is currently VVD-only and covers device discovery, app open/close, `back`, `home`, and `tvRemote`. Physical Fire TV, capture, selector, install, logging, and performance methods report unsupported for Vega targets. Supported command methods: diff --git a/website/docs/docs/replay-e2e.md b/website/docs/docs/replay-e2e.md index 304ec141da..3023fa48d1 100644 --- a/website/docs/docs/replay-e2e.md +++ b/website/docs/docs/replay-e2e.md @@ -58,6 +58,17 @@ agent-device replay ~/.agent-device/sessions/e2e-2026-02-09T12-00-00-000Z.ad --s Interior `close` actions still run. The flag is intentionally unavailable to `test` because suite attempts own cleanup, and it is rejected for Maestro YAML because that runtime owns its lifecycle. +- `press`, `click`, and `longpress` steps wait up to 2 seconds for their target to appear before + they fail. A step recorded against a screen that was still loading passes on replay once the + target shows up. The wait covers only a target that is not on screen yet: a target that is + covered, off-screen, or matched by more than one element fails at once, as it does live. +- When the target never appears, replay stops with `REPLAY_DIVERGENCE`, and + `error.details.readiness` says how long the step waited and how many times it looked (`waitedMs`, + `polls`, `end`). For a step recorded with a target annotation (the `# agent-device:target-v1` + line above it), `error.details.divergence.kind` is `selector-miss` and the step is never sent. + For a step without an annotation, `error.details.reason` is `selector_not_found`, as for a live + command. If the app shows an empty accessibility tree during that wait, `error.details.reason` + is `capture_sparse` instead; take a snapshot to see where the app is. ## Run Maestro compatibility flows @@ -384,5 +395,12 @@ Passing `--plan-digest` that no longer matches the current script — because yo - Leave the replay plan unchanged, repair app state so the reported failed step can be retried, then use its `--from`/`--plan-digest`. Resume starts at `--from`; it does not skip that step. - Replay file parse error: - Validate quoting in `.ad` lines (unclosed quotes are rejected). +- A `press` or `click` step fails because its target was not found, but the element is on the + screenshot: + - A `selector-miss` divergence, or `error.details.readiness.end: expired`, means the element was + not in the accessibility tree for the whole 2-second wait: the selector is wrong for this + build, or the element is not exposed to accessibility. `readiness.end: sparse` means the app + showed an empty tree: the screen was mid-transition or the app had left. Add a `wait` step for + a landmark on the new screen before the press. - Maestro compatibility flow fails on unsupported syntax: - Check [ADR 0015](https://github.com/callstack/agent-device/blob/main/docs/adr/0015-direct-maestro-engine.md). If the missing feature matters to your suite, open a focused issue with a small flow snippet.