From 0d0030d5ce472f8df0c72a2647f49ac136c28869 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Sun, 27 Sep 2026 13:36:15 +0200 Subject: [PATCH 01/23] proto(reliability): observation engine, readiness poll on promotedTarget, two loops migrated Prototype only. Shared observeUntil loop in capture-kit with one ride-out classifier in contracts; promotedTarget declares a 2 s / 200 ms readiness budget consumed by selector resolution with the first capture unbounded; post-gesture stability and scroll movement run as verdict predicates over the engine; targetReadiness and outcomeObservation guarantee cells. --- packages/capture-kit/package.json | 4 + .../capture-kit/src/observe-until.test.ts | 217 ++++++++++++++ packages/capture-kit/src/observe-until.ts | 277 ++++++++++++++++++ .../capture-kit/src/post-gesture-stability.ts | 113 +++---- packages/contracts/package.json | 4 + .../contracts/src/interaction-guarantees.ts | 99 +++++++ packages/contracts/src/observation.ts | 24 ++ .../selectors/src/selector-pipeline-policy.ts | 19 +- .../selectors/src/selector-pipeline.test.ts | 10 +- .../layering/contracts-exports.snapshot.json | 1 + scripts/layering/package-boundaries.test.ts | 1 + .../contracts/interaction-guarantees.test.ts | 6 + src/backend-snapshot-options.ts | 5 +- src/backend.ts | 6 + .../interaction/runtime/resolution.ts | 274 +++++++++++++++-- src/daemon/scroll-movement.ts | 77 ++--- src/daemon/selector-runtime-backend.ts | 19 +- .../runtime-selector.contract.test.ts | 28 ++ .../runtime-selector.coverage.ts | 2 + .../press-target-readiness.test.ts | 171 +++++++++++ .../scroll-movement-observation.test.ts | 215 ++++++++++++++ 21 files changed, 1446 insertions(+), 126 deletions(-) create mode 100644 packages/capture-kit/src/observe-until.test.ts create mode 100644 packages/capture-kit/src/observe-until.ts create mode 100644 packages/contracts/src/observation.ts create mode 100644 test/integration/provider-scenarios/press-target-readiness.test.ts create mode 100644 test/integration/provider-scenarios/scroll-movement-observation.test.ts diff --git a/packages/capture-kit/package.json b/packages/capture-kit/package.json index f22fb90cc9..556ef65fb3 100644 --- a/packages/capture-kit/package.json +++ b/packages/capture-kit/package.json @@ -46,6 +46,10 @@ "types": "./src/capture-admission/durable-capture-runtime-recovery.ts", "default": "./src/capture-admission/durable-capture-runtime-recovery.ts" }, + "./observe-until": { + "types": "./src/observe-until.ts", + "default": "./src/observe-until.ts" + }, "./perf-capture-admission-ledger": { "types": "./src/capture-admission/perf-capture-admission-ledger.ts", "default": "./src/capture-admission/perf-capture-admission-ledger.ts" diff --git a/packages/capture-kit/src/observe-until.test.ts b/packages/capture-kit/src/observe-until.test.ts new file mode 100644 index 0000000000..bf00cb7222 --- /dev/null +++ b/packages/capture-kit/src/observe-until.test.ts @@ -0,0 +1,217 @@ +import assert from 'node:assert/strict'; +import { describe, test } from 'vitest'; +import { AppError } from '@agent-device/kernel/errors'; +import { isRideOutCaptureError } from '@agent-device/contracts/observation'; +import { observeUntil, type ObservationClock } from './observe-until.ts'; + +/** A clock that advances only when the loop sleeps or a capture declares its own cost. */ +function fakeClock(): ObservationClock & { advance(ms: number): void; slept: number[] } { + let nowMs = 1_000; + const slept: number[] = []; + return { + slept, + now: () => nowMs, + advance: (ms) => { + nowMs += ms; + }, + sleep: async (ms) => { + slept.push(ms); + nowMs += ms; + }, + }; +} + +const SCHEDULE = { intervalMs: 200, budgetMs: 1_000 }; + +describe('observeUntil', () => { + test('answers done on the poll whose verdict accepts, with the timeline', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + clock.advance(50); + return captures; + }, + verdict: (latest) => + latest >= 3 ? { kind: 'done', result: `seen ${latest}` } : { kind: 'continue' }, + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.equal(observed.kind === 'done' && observed.result, 'seen 3'); + assert.equal(observed.polls.length, 3); + assert.deepEqual(clock.slept, [200, 200]); + assert.equal(observed.waitedMs, 550); + }); + + test('expires when the budget runs out while the verdict still continues, keeping the last value', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: () => ({ kind: 'continue' }), + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.kind === 'expired' && observed.last, 6); + assert.equal(observed.polls.length, 6); + assert.equal(observed.waitedMs, 1_000); + }); + + test('rides out only the errors the caller classifies, and keeps polling', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + if (captures === 1) throw new AppError('COMMAND_FAILED', 'busy', { retriable: true }); + return captures; + }, + verdict: (latest) => ({ kind: 'done', result: latest }), + schedule: SCHEDULE, + rideOut: isRideOutCaptureError, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['rode-out', 'observed'], + ); + }); + + test('fails on the first error the caller does not ride out', async () => { + const clock = fakeClock(); + const failure = new AppError('COMMAND_FAILED', 'wedged'); + const observed = await observeUntil({ + capture: async () => { + throw failure; + }, + verdict: () => ({ kind: 'continue' }), + schedule: SCHEDULE, + rideOut: isRideOutCaptureError, + clock, + }); + assert.equal(observed.kind, 'failed'); + assert.equal(observed.kind === 'failed' && observed.error, failure); + assert.equal(observed.polls.length, 1); + }); + + test('cancels and joins a capture still in flight at the deadline', async () => { + let aborted = false; + const observed = await observeUntil({ + capture: (signal) => + new Promise((resolve) => { + signal.addEventListener('abort', () => { + aborted = true; + setTimeout(() => resolve(1), 5); + }); + }), + verdict: () => ({ kind: 'done', result: true }), + schedule: { intervalMs: 5, budgetMs: 20 }, + }); + assert.equal(observed.kind, 'stalled'); + assert.equal(aborted, true); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['stalled'], + ); + }); + + test('always completes minPolls even past the budget, so a quiet pair can form', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + clock.advance(900); + return captures; + }, + verdict: (latest, previous) => + previous !== undefined + ? { kind: 'done', result: [previous, latest] } + : { kind: 'continue' }, + schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2 }, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.deepEqual(observed.kind === 'done' && observed.result, [1, 2]); + }); + + test('a verdict can raise the budget but never lower it', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: (latest) => + latest === 1 ? { kind: 'continue', budgetMs: 2_000 } : { kind: 'continue', budgetMs: 100 }, + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.waitedMs, 2_000); + }); + + test('judges an initial value before spending a poll', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: (latest) => ({ kind: 'done', result: latest }), + schedule: SCHEDULE, + initial: 42, + clock, + }); + assert.equal(observed.kind === 'done' && observed.value, 42); + assert.equal(captures, 0); + assert.equal(observed.polls.length, 0); + }); +}); + +describe('observeUntil budgetFrom first-capture', () => { + test('never bounds the first capture and spends the budget only on retries', async () => { + const clock = fakeClock(); + let captures = 0; + const observed = await observeUntil({ + capture: async () => { + captures += 1; + if (captures === 1) clock.advance(5_000); + return captures; + }, + verdict: (latest) => (latest === 3 ? { kind: 'done', result: latest } : { kind: 'continue' }), + schedule: { intervalMs: 200, budgetMs: 1_000, budgetFrom: 'first-capture' }, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.equal(observed.polls.length, 3); + assert.equal(observed.polls[0]?.durationMs, 5_000); + assert.equal(observed.waitedMs, 5_400); + }); + + test('sleeps the interval between an initial observation and the first poll', async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: async () => 2, + verdict: (latest, previous) => + previous === undefined + ? { kind: 'continue' } + : { kind: 'done', result: [previous, latest] }, + schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2 }, + initial: 1, + clock, + }); + assert.deepEqual(observed.kind === 'done' && observed.result, [1, 2]); + assert.deepEqual(clock.slept, [200]); + assert.equal(observed.polls.length, 1); + }); +}); diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts new file mode 100644 index 0000000000..5ce305518f --- /dev/null +++ b/packages/capture-kit/src/observe-until.ts @@ -0,0 +1,277 @@ +import { createRequestCanceledError } from '@agent-device/kernel/errors'; +import type { ObservationEnd } from '@agent-device/contracts/observation'; +import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; + +/** + * The one observation loop: capture until a verdict accepts, on a schedule, under a budget that is + * enforced by cancellation. Owns exactly what every loop in this codebase used to re-implement — + * cadence, the deadline, abort-and-join of an in-flight capture, which capture errors are ridden + * out, and the per-poll timeline. Owns nothing about WHAT is observed: the verdict is the caller's, + * and so is the equality rule behind it (digest, signature, pixel). + */ + +export type ObservationSchedule = Readonly<{ + /** Delay between polls, clamped to the remaining budget. */ + intervalMs: number; + /** + * Wall-clock budget from the first capture. It bounds when a poll may start, and a capture that + * starts inside it may overrun it by at most one interval: the sample after the last sleep is + * always taken, so a verdict that reads elapsed time against a cap sees a poll at or past it. + */ + budgetMs: number; + /** Observations (an `initial` counts) the loop always completes before the budget may end it. */ + minPolls?: number; + /** + * `'start'` (default) bounds every capture, the first included. `'first-capture'` leaves the first + * capture unbounded and spends the budget only on retries, so a caller whose first attempt is the + * pre-existing one-shot behaviour pays nothing on its success path. + */ + budgetFrom?: 'start' | 'first-capture'; +}>; + +export type ObservationClock = Readonly<{ + now(): number; + sleep(ms: number): Promise; +}>; + +export type ObservationVerdict = + | Readonly<{ kind: 'done'; result: R }> + /** Keep polling; `budgetMs` raises the budget measured from the first capture (never lowers it). */ + | Readonly<{ kind: 'continue'; budgetMs?: number }>; + +export type ObservationPollOutcome = 'observed' | 'rode-out' | 'stalled'; + +export type ObservationPoll = Readonly<{ + startedMs: number; + durationMs: number; + outcome: ObservationPollOutcome; +}>; + +export type ObservationEvidence = Readonly<{ + polls: readonly ObservationPoll[]; + waitedMs: number; +}>; + +type ObservedEnd = + | Readonly<{ kind: 'done'; result: R; value: T }> + | Readonly<{ kind: 'expired'; last: T | undefined; lastError: unknown }> + | Readonly<{ kind: 'stalled'; last: T | undefined; error: unknown }> + | Readonly<{ kind: 'failed'; last: T | undefined; error: unknown }>; + +export type Observed = ObservationEvidence & ObservedEnd; + +export type ObserveUntilParams = Readonly<{ + capture: (signal: AbortSignal) => Promise; + /** Judges the latest capture; `previous` is the last observed value, for quiet-pair predicates. */ + verdict: (latest: T, previous: T | undefined, polls: number) => ObservationVerdict; + schedule: ObservationSchedule; + /** True keeps polling past this capture error. Default: no error is ridden out. */ + rideOut?: (error: unknown) => boolean; + signal?: AbortSignal; + clock?: ObservationClock; + /** A capture the caller already holds; judged first, without a poll. */ + initial?: T; + /** Diagnostic phase for the end-of-loop line; nothing is logged without it. */ + phase?: string; +}>; + +const REAL_CLOCK: ObservationClock = { + now: () => Date.now(), + sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)), +}; + +type LoopState = { + readonly startedMs: number; + budgetStartedMs: number; + budgetMs: number; + observations: number; + readonly polls: ObservationPoll[]; + previous: T | undefined; + last: T | undefined; + lastError: unknown; +}; + +type Loop = Readonly<{ + params: ObserveUntilParams; + clock: ObservationClock; + state: LoopState; +}>; + +export async function observeUntil( + params: ObserveUntilParams, +): Promise> { + const clock = params.clock ?? REAL_CLOCK; + const startedMs = clock.now(); + const loop: Loop = { + params, + clock, + state: { + startedMs, + budgetStartedMs: startedMs, + budgetMs: params.schedule.budgetMs, + observations: 0, + polls: [], + previous: undefined, + last: undefined, + lastError: undefined, + }, + }; + if (params.initial !== undefined) { + loop.state.observations += 1; + const done = judge(loop, params.initial); + if (done) return done; + } + while (true) { + if (params.signal?.aborted) throw createRequestCanceledError(); + const expired = await waitBeforePoll(loop); + if (expired) return expired; + const ended = await pollOnce(loop); + if (ended) return ended; + } +} + +function remainingMs({ state, clock }: Loop): number { + return state.budgetStartedMs + state.budgetMs - clock.now(); +} + +function mustPoll({ params, state }: Loop): boolean { + return state.observations < (params.schedule.minPolls ?? 1); +} + +/** Sleeps one interval between observations; ends the loop when the budget is spent first. */ +async function waitBeforePoll(loop: Loop): Promise | undefined> { + const { state, clock, params } = loop; + if (state.observations === 0) return undefined; + const forced = mustPoll(loop); + if (!forced && remainingMs(loop) <= 0) + return finish(loop, { kind: 'expired', last: state.last, lastError: state.lastError }); + const { intervalMs } = params.schedule; + await clock.sleep(forced ? intervalMs : Math.max(0, Math.min(intervalMs, remainingMs(loop)))); + return undefined; +} + +async function pollOnce(loop: Loop): Promise | undefined> { + const { params, clock, state } = loop; + const unbounded = params.schedule.budgetFrom === 'first-capture' && state.polls.length === 0; + const pollStartedMs = clock.now(); + const poll = await captureWithin( + unbounded ? undefined : Math.max(remainingMs(loop), params.schedule.intervalMs), + params.signal, + params.capture, + ); + state.observations += 1; + if (unbounded) state.budgetStartedMs = clock.now(); + const outcome = pollOutcome(loop, poll); + state.polls.push({ + startedMs: pollStartedMs - state.startedMs, + durationMs: clock.now() - pollStartedMs, + outcome, + }); + if (poll.kind === 'observed') return judge(loop, poll.value); + if (outcome === 'rode-out') return undefined; + return finish(loop, { kind: poll.kind, last: state.last, error: poll.error }); +} + +function pollOutcome( + loop: Loop, + poll: CaptureWithinOutcome, +): ObservationPollOutcome { + if (poll.kind === 'stalled') return 'stalled'; + if (poll.kind === 'failed') return loop.params.rideOut?.(poll.error) ? 'rode-out' : 'observed'; + return 'observed'; +} + +function judge(loop: Loop, value: T): Observed | undefined { + const { params, state } = loop; + const verdict = params.verdict(value, state.previous, state.polls.length); + state.previous = value; + state.last = value; + if (verdict.kind === 'done') return finish(loop, { kind: 'done', result: verdict.result, value }); + if (verdict.budgetMs !== undefined) state.budgetMs = Math.max(state.budgetMs, verdict.budgetMs); + return undefined; +} + +function finish(loop: Loop, end: ObservedEnd): Observed { + const { params, state, clock } = loop; + const observed: Observed = { + ...end, + polls: state.polls, + waitedMs: clock.now() - state.startedMs, + }; + if (params.phase) { + emitDiagnostic({ + level: 'debug', + phase: params.phase, + data: { + end: observed.kind satisfies ObservationEnd, + polls: state.polls.length, + waitedMs: observed.waitedMs, + }, + }); + } + return observed; +} + +type CaptureWithinOutcome = + | Readonly<{ kind: 'observed'; value: T }> + | Readonly<{ kind: 'failed'; error: unknown }> + | Readonly<{ kind: 'stalled'; error: unknown }>; + +/** + * Runs one capture under the remaining budget by cancellation, then waits for it to quiesce. + * Deliberately not race-and-abandon: a late capture could otherwise mutate session state or keep a + * platform helper after the loop has returned. + */ +async function captureWithin( + remainingMs: number | undefined, + parent: AbortSignal | undefined, + capture: (signal: AbortSignal) => Promise, +): Promise> { + const deadline = armDeadline(remainingMs); + const signal = parent ? AbortSignal.any([parent, deadline.signal]) : deadline.signal; + try { + const value = await capture(signal); + return endOfCapture(parent, deadline.expired(), undefined) ?? { kind: 'observed', value }; + } catch (error) { + return endOfCapture(parent, deadline.expired(), error) ?? { kind: 'failed', error }; + } finally { + deadline.dispose(); + } +} + +/** A capture that ended after the deadline stalled; one ended by the caller's signal was canceled. */ +function endOfCapture( + parent: AbortSignal | undefined, + deadlineExpired: boolean, + error: unknown, +): CaptureWithinOutcome | undefined { + if (parent?.aborted) throw createRequestCanceledError(); + return deadlineExpired ? { kind: 'stalled', error } : undefined; +} + +function armDeadline(remainingMs: number | undefined): { + signal: AbortSignal; + expired: () => boolean; + dispose: () => void; +} { + const controller = new AbortController(); + let expired = false; + const timer = + remainingMs === undefined + ? undefined + : setTimeout( + () => { + expired = true; + controller.abort(new DOMException('Observation deadline exceeded', 'TimeoutError')); + }, + Math.max(0, remainingMs), + ); + timer?.unref(); + return { + signal: controller.signal, + expired: () => expired, + dispose: () => { + if (timer !== undefined) clearTimeout(timer); + }, + }; +} diff --git a/packages/capture-kit/src/post-gesture-stability.ts b/packages/capture-kit/src/post-gesture-stability.ts index 5ce237f1ce..9610bf5d49 100644 --- a/packages/capture-kit/src/post-gesture-stability.ts +++ b/packages/capture-kit/src/post-gesture-stability.ts @@ -1,21 +1,29 @@ import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; -import { sleep } from '@agent-device/host-kit/retry'; import type { PostGestureAction, PostGestureOutcome } from '@agent-device/kernel/snapshot'; +import { observeUntil, type ObservationSchedule } from './observe-until.ts'; /** - * Pure post-gesture stability mechanics: the quiet-window polling loop and the - * baseline-distrust verdict, parameterized over the capture value and the - * signature comparators. Deliberately a leaf — it imports no cycle owners and - * no `SessionState`, so it stays outside the R9 type cycle while the - * deferred-interaction-outcome owner (which holds the pending record and the - * session mutation) stays the one seam callers see. The owner supplies the - * comparators from interaction-outcome-policy; their semantics (subset - * tolerance, identity keying, discriminating entries) are documented there. + * Pure post-gesture stability mechanics: the quiet-window verdict over the shared + * `observeUntil` engine, plus the baseline-distrust decision, parameterized over + * the capture value and the signature comparators. Deliberately a leaf — it + * imports no cycle owners and no `SessionState`, so it stays outside the R9 + * type cycle while the deferred-interaction-outcome owner (which holds the + * pending record and the session mutation) stays the one seam callers see. The + * owner supplies the comparators from interaction-outcome-policy; their + * semantics (subset tolerance, identity keying, discriminating entries) are + * documented there. */ -const STABILIZATION_DEADLINE_MS = 1_500; -const STABILIZATION_INTERVAL_MS = 200; -const STABILIZATION_MIN_ATTEMPTS = 2; +/** + * Cadence and budget for the quiet-window loop: poll every 200ms, allow 1.5s + * before an unsettled surface times out, and always complete two observations + * (an `initial` counts) so a quiet pair can form even under a tight budget. + */ +const POST_GESTURE_STABILITY_SCHEDULE: ObservationSchedule = { + intervalMs: 200, + budgetMs: 1_500, + minPolls: 2, +}; /** * Defect 2 (#1542): a bounded extra budget used ONLY when a quiet signature @@ -30,7 +38,7 @@ const STABILIZATION_MIN_ATTEMPTS = 2; * settle-zero-margin-flake, a week-long contention-flake root cause), so this * cap is sized to never come close to that trap. */ -const STABILIZATION_DISTRUST_DEADLINE_MS = STABILIZATION_DEADLINE_MS + 2_000; +const STABILIZATION_DISTRUST_DEADLINE_MS = POST_GESTURE_STABILITY_SCHEDULE.budgetMs + 2_000; export type BaselineSurfaceEvidence = 'changed' | 'unchanged' | 'ambiguous'; @@ -116,10 +124,11 @@ export function decidePostGestureStabilityVerdict( } /** - * The quiet-window stability loop: poll until two consecutive captures agree - * and the verdict accepts the agreement, or the (possibly distrust-extended) - * deadline expires. Session state never enters here — the caller owns the - * pending record's lifecycle and clears it when this returns. + * The quiet-window stability verdict over the shared observation engine: keep + * polling until two consecutive captures agree and the decision accepts the + * agreement, or the (possibly distrust-extended) budget expires. Session + * state never enters here — the caller owns the pending record's lifecycle + * and clears it when this returns. */ export async function runPostGestureStabilityLoop(params: { pending: PostGestureStabilityPending; @@ -129,24 +138,34 @@ export async function runPostGestureStabilityLoop> { const { pending, needsBaselineDistrust, hooks } = params; const startedAt = Date.now(); - let attempts = 1; - let previous = await captureSurface(hooks, params.initial); + let attempts = 0; let baselineSignature = pending.baselineSignature; let baselineBackend = pending.baselineBackend; let baselineRebased = false; - // Extended past STABILIZATION_DEADLINE_MS only when the distrust verdict - // fires below; the ordinary (non-distrust) timeout path is unaffected. - let effectiveDeadlineMs = STABILIZATION_DEADLINE_MS; // A rebase or a distrust verdict keeps polling on a pair that DID agree, so - // the deadline can expire on a surface that is already at rest. + // the budget can expire on a surface that is already at rest. let lastPairAgreed = false; + let surfaceCache: CapturedSurface | undefined; + + const surfaceOf = (value: T): CapturedSurface => { + if (surfaceCache?.value === value) return surfaceCache; + surfaceCache = { value, ...hooks.readSurface(value) }; + return surfaceCache; + }; + + const observed = await observeUntil>({ + ...(params.initial !== undefined ? { initial: params.initial } : {}), + capture: () => hooks.capture(), + schedule: POST_GESTURE_STABILITY_SCHEDULE, + verdict: (latest, previousValue) => { + attempts += 1; + const current = surfaceOf(latest); + if (previousValue === undefined) return { kind: 'continue' }; + + const previous = surfaceOf(previousValue); + lastPairAgreed = hooks.signaturesStable(previous.signature, current.signature); + if (!lastPairAgreed) return { kind: 'continue' }; - while (attempts < STABILIZATION_MIN_ATTEMPTS || Date.now() - startedAt < effectiveDeadlineMs) { - await sleep(STABILIZATION_INTERVAL_MS); - attempts += 1; - const current = await captureSurface(hooks); - lastPairAgreed = hooks.signaturesStable(previous.signature, current.signature); - if (lastPairAgreed) { const elapsedMs = Date.now() - startedAt; // A capture plan may fall back or be pre-empted by the XCTest-channel // penalty at any time, so the backend can change mid-poll. Backends do @@ -162,8 +181,7 @@ export async function runPostGestureStabilityLoop = { @@ -204,14 +227,6 @@ type CapturedSurface = { backend: string | undefined; }; -async function captureSurface( - hooks: PostGestureStabilityHooks, - initial?: T, -): Promise> { - const value = initial ?? (await hooks.capture()); - return { value, ...hooks.readSurface(value) }; -} - function emitSettleDiagnostic( verdict: 'trust' | 'accept-stale', action: string, diff --git a/packages/contracts/package.json b/packages/contracts/package.json index 9582f1ada8..8a3cfe9bad 100644 --- a/packages/contracts/package.json +++ b/packages/contracts/package.json @@ -308,6 +308,10 @@ "types": "./src/facades/observability.ts", "default": "./src/facades/observability.ts" }, + "./observation": { + "types": "./src/observation.ts", + "default": "./src/observation.ts" + }, "./orientation-runtime": { "types": "./src/orientation-runtime.ts", "default": "./src/orientation-runtime.ts" diff --git a/packages/contracts/src/interaction-guarantees.ts b/packages/contracts/src/interaction-guarantees.ts index b18f46be48..8224e6ecee 100644 --- a/packages/contracts/src/interaction-guarantees.ts +++ b/packages/contracts/src/interaction-guarantees.ts @@ -73,6 +73,16 @@ export const INTERACTION_GUARANTEES = [ // unique/disambiguated/exact/label-fallback/not-observed provenance, // pre-action diagnostics only — never ref-issued or MCP-pinned. 'resolutionDisclosure', + // How the path waits for the target to exist and become actionable before acting: a declared + // budget under which a plain no-match keeps retrying, versus a target that must already exist at + // dispatch time. Distinct from occlusion/offscreen/nonHittable, which judge a target already + // found; this is about whether one is found at all. + 'targetReadiness', + // What the path observes after dispatch to back its success claim, and whether that observation + // is advisory (a hint the caller may ignore) or can change the response (a corroborated failure, + // a settled diff). Distinct from the opt-in verifyEvidence/settleObservation features; this is the + // path's baseline, always-on claim. + 'outcomeObservation', ] as const; export type InteractionGuarantee = (typeof INTERACTION_GUARANTEES)[number]; @@ -149,6 +159,45 @@ const SHARED_RESPONSE_CONSTRUCTION: GuaranteeEnforcement = { via: 'src/daemon/interaction/internal/interaction-touch-response.ts#buildInteractionResponseData', }; +// runtime-selector and runtime-ref share this waiver by construction: neither path observes the tap +// outcome by default. The deferred/corroborated-failure outcome mark (a recorded XCTest failure +// after a tap is an ambiguous outcome, per docs/agents/selector-capture.md) is scoped to +// navigation-sensitive Android actions and target-authored gestures, not the tap-shaped runtime +// paths; --settle/--verify are the opt-in ways to observe one. +const TAP_OUTCOME_NOT_OBSERVED_GAP: GuaranteeEnforcement = { + kind: 'waived', + reason: + 'gap: neither path observes the tap outcome by default; only --settle/--verify capture post-action evidence, and the deferred/corroborated-failure outcome mark applies only to navigation-sensitive Android actions and target-authored gestures.', + trackingIssue: GAPS_UMBRELLA_ISSUE, +}; + +// Both Maestro-compatible fast paths (src/daemon/interaction/internal/interaction-touch-direct-ios.ts) +// dispatch ONE fused XCTest runner request (`command: 'tap'` with a selectorKey/selectorValue) built +// in packages/platform-apple/src/interactions.ts#tapElementSelector — not the separate +// packages/maestro/ flow-script engine, which drives standalone `.yaml` Maestro flows and never +// participates in an ordinary daemon click. Verified by reading the dispatch call graph +// (interaction-touch-direct-ios.ts -> BoundTouchExecutor.tapElementSelector -> +// platform-apple/src/interactions.ts) before writing this cell (AGENTS.md: verify each claim +// against the native implementation). +const DIRECT_IOS_SINGLE_QUERY_READINESS: GuaranteeEnforcement = { + kind: 'waived', + reason: + "Intentional: the fused runner request performs a single XCTest selector/frame query per dispatch and does not itself retry for an as-yet-nonexistent target; a miss is a runner failure (see errorTaxonomy) rather than a wait, and delegation-on-error (ADR 0011) is a different path's guarantee, not a retry of this one.", +}; + +// Shares the SAME reasoning as TAP_OUTCOME_NOT_OBSERVED_GAP: the tap outcome is not observed on the +// success path here either. The shared ambiguous-XCTest-failure corroboration +// (src/daemon/interaction/internal/interaction-ios-tap-outcome.ts#corroborateIosTapFailure) is +// invoked identically from this fast path and from the tree-based runtime dispatch, but only ever +// reconsiders a THROWN error — it never validates a call that returned success, so it is not a +// baseline outcome claim for either path. +const DIRECT_IOS_OUTCOME_NOT_OBSERVED_GAP: GuaranteeEnforcement = { + kind: 'waived', + reason: + "gap: this replay-only route observes no outcome beyond the runner's own report; --verify/--settle do not apply here (see the inapplicable cells on this row), and the shared ambiguous-failure corroboration only reconsiders a thrown error, never a successful dispatch.", + trackingIssue: GAPS_UMBRELLA_ISSUE, +}; + // The two runtime tree paths (selector and ref resolution) run the SAME shared // guard/observation implementations; only how the target is found // (disambiguation) and how failures are described (errorTaxonomy) differ. @@ -238,6 +287,14 @@ export const INTERACTION_DISPATCH_PATHS: Record { intervalMs: 300, }); } - for (const row of NODE_STAGE_ROWS.filter((name) => name !== 'wait' && name !== 'findWait')) { + assert.deepEqual(selectorPollBudget(SELECTOR_PIPELINE_POLICIES.promotedTarget), { + defaultTimeoutMs: 2_000, + intervalMs: 200, + }); + for (const row of NODE_STAGE_ROWS.filter( + (name) => name !== 'wait' && name !== 'findWait' && name !== 'promotedTarget', + )) { assert.throws(() => selectorPollBudget(SELECTOR_PIPELINE_POLICIES[row]), /no poll budget/, row); } }); @@ -390,7 +396,7 @@ test('the documented per-caller pipelines are the ones declared', () => { }), ), { - promotedTarget: ['exclude-and-refuse', 'refuse', 'hittable-ancestor', 'no-poll'], + promotedTarget: ['exclude-and-refuse', 'refuse', 'hittable-ancestor', 'poll'], resolvedTarget: ['exclude-and-refuse', 'refuse', 'none', 'no-poll'], coveredDiagnosis: ['refuse', 'ignore', 'none', 'no-poll'], readText: ['ignore', 'ignore', 'none', 'no-poll'], diff --git a/scripts/layering/contracts-exports.snapshot.json b/scripts/layering/contracts-exports.snapshot.json index 3095a7699e..14967bed29 100644 --- a/scripts/layering/contracts-exports.snapshot.json +++ b/scripts/layering/contracts-exports.snapshot.json @@ -74,6 +74,7 @@ "@agent-device/contracts/network-runtime-plan", "@agent-device/contracts/network-traffic", "@agent-device/contracts/observability", + "@agent-device/contracts/observation", "@agent-device/contracts/orientation-runtime", "@agent-device/contracts/perf-runtime", "@agent-device/contracts/perf-runtime-host", diff --git a/scripts/layering/package-boundaries.test.ts b/scripts/layering/package-boundaries.test.ts index ea6b4f56ef..9657a0a870 100644 --- a/scripts/layering/package-boundaries.test.ts +++ b/scripts/layering/package-boundaries.test.ts @@ -408,6 +408,7 @@ test('the real tree parses, declares, and passes R11', () => { '@agent-device/capture-kit/ios-snapshot-runtime', '@agent-device/capture-kit/ios-snapshot-tree', '@agent-device/capture-kit/mobile-snapshot-semantics', + '@agent-device/capture-kit/observe-until', '@agent-device/capture-kit/perf-capture-admission-ledger', '@agent-device/capture-kit/perf-capture-recovery', '@agent-device/capture-kit/perf-capture-resource-store', diff --git a/src/__tests__/contracts/interaction-guarantees.test.ts b/src/__tests__/contracts/interaction-guarantees.test.ts index 1f618303ef..f98690db5f 100644 --- a/src/__tests__/contracts/interaction-guarantees.test.ts +++ b/src/__tests__/contracts/interaction-guarantees.test.ts @@ -164,9 +164,15 @@ test('acknowledged gaps are visible and bounded', () => { // (umbrella: https://github.com/callstack/agent-device/issues/1081). assert.deepEqual(gaps.sort(), [ 'maestro-direct-selector/errorTaxonomy', + 'maestro-direct-selector/outcomeObservation', 'maestro-direct-selector/parentOwnedTouchPoint', 'maestro-direct-selector/responseIdentity', 'maestro-non-hittable-fallback/errorTaxonomy', + 'maestro-non-hittable-fallback/outcomeObservation', 'maestro-non-hittable-fallback/parentOwnedTouchPoint', + 'native-ref/outcomeObservation', + 'runtime-ref/outcomeObservation', + 'runtime-selector/outcomeObservation', + 'target-drag/outcomeObservation', ]); }); diff --git a/src/backend-snapshot-options.ts b/src/backend-snapshot-options.ts index e60f192310..04f8564aa4 100644 --- a/src/backend-snapshot-options.ts +++ b/src/backend-snapshot-options.ts @@ -1,7 +1,10 @@ import { snapshotFlagsFromOptions } from '@agent-device/kernel/snapshot'; import type { BackendSnapshotOptions } from './backend.ts'; -type RoutedSnapshotOption = keyof Omit; +type RoutedSnapshotOption = keyof Omit< + BackendSnapshotOptions, + 'includeRects' | 'outPath' | 'forceFresh' +>; /** * Which backend snapshot options route as request flags. Still pinned to diff --git a/src/backend.ts b/src/backend.ts index 2533e8f4d1..57d3d7140b 100644 --- a/src/backend.ts +++ b/src/backend.ts @@ -64,6 +64,12 @@ export type BackendSnapshotOptions = SnapshotOptions & { includeRects?: boolean; includeHiddenContentHints?: boolean; outPath?: string; + /** + * Bypass a cache-serving backend's short-lived stored capture, the way `wait`/`find` already force + * one (`src/daemon/selector-runtime-backend.ts`). A backend without such a cache has nothing to + * bypass and may ignore this. + */ + forceFresh?: boolean; }; export type BackendReadTextResult = { diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index 48bf82e61b..ac95c5bd36 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -32,7 +32,10 @@ import { SELECTOR_PIPELINE_POLICIES, type ActingPipelinePolicy, type SelectorPipelinePolicy, + type SelectorPollBudget, } from '@agent-device/selectors/selector-pipeline-policy'; +import { observeUntil } from '@agent-device/capture-kit/observe-until'; +import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; import { resolvePressRecordingTarget } from '@agent-device/selectors/press-retarget'; import { requireSnapshotSession } from './selector-read-shared.ts'; import { findNodeByLabel, resolveRefLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; @@ -379,14 +382,39 @@ async function resolveRefInteractionTarget( }; } -async function resolveSelectorInteractionTarget( +/** One poll's outcome: the capture it read the tree from, and what it resolved (if anything). */ +type SelectorResolutionAttempt = { + capture: InteractionSnapshot; + resolved: SelectorResolution | null; +}; + +/** The narrowed attempt a readiness poll accepts: a resolution with a usable rect. */ +type ResolvedSelectorAttempt = { + capture: InteractionSnapshot; + resolved: SelectorResolution; +}; + +/** + * One capture-and-resolve attempt: interactive capture, resolve; on a miss that requires + * interactivity, fall back to a full (non-interactive) capture and resolve again. Identical body + * whether run once (`promotedTarget.poll === 'none'`'s twins, `resolvedTarget`) or repeated under + * `observeUntil` for a row that polls — this function draws no distinction, and does not decide + * whether the miss is a refusal (ambiguity, occlusion, off-screen) or a plain no-match: those stay + * the caller's decisions, made from `resolved`/thrown errors exactly as before. + */ +async function attemptSelectorResolution( runtime: AgentDeviceRuntime, options: CommandContext, - target: Extract, + selectorExpression: string, params: ResolveInteractionTargetParams, -): Promise { - const selectorExpression = target.selector; - let capture = await captureInteractionSnapshot(runtime, options, params.requireInteractive); + forceFresh: boolean, +): Promise { + let capture = await captureInteractionSnapshot( + runtime, + options, + params.requireInteractive, + forceFresh, + ); let resolved = resolveActionSelector( capture.snapshot.nodes, selectorExpression, @@ -395,7 +423,7 @@ async function resolveSelectorInteractionTarget( ); if ((!resolved || !resolved.node.rect) && params.requireInteractive) { const interactive = capture.snapshot; - capture = await captureInteractionSnapshot(runtime, options, false); + capture = await captureInteractionSnapshot(runtime, options, false, forceFresh); inheritPostGestureOutcome(interactive, capture.snapshot); resolved = resolveActionSelector( capture.snapshot.nodes, @@ -404,14 +432,184 @@ async function resolveSelectorInteractionTarget( params.pipeline, ); } - if (!resolved || !resolved.node.rect) { - throw await selectorInteractionFailure({ + return { capture, resolved }; +} + +/** The readiness poll's evidence, attached to a target-not-found failure only (never to a refusal). */ +type SelectorReadinessDetails = { + polls: number; + waitedMs: number; + end: 'expired' | 'stalled'; +}; + +/** + * `promotedTarget`'s readiness budget: the target may not exist yet, so a + * plain no-match (no resolution, or a resolution with no usable rect) keeps polling under the row's + * budget instead of refusing on the first capture. Every other outcome stays terminal exactly as a + * single attempt would produce it — an ambiguity throw from `resolveActionSelector`, or a capture + * error the loop does not ride out, ends the loop on this same poll, and is rethrown unchanged. + * Occlusion, off-screen, non-hittable, and keyboard refusals are not judged here: they run once, + * after this loop returns a rect-bearing resolution, exactly as `runInteractionPipelineStages` + * always has. + * + * The first poll is unbounded and reuses today's snapshot cache (`forceFresh: false`), so a caller + * whose first capture already matches pays the same one-or-two-capture cost as before. Only a poll + * that follows a miss (poll 2+) forces a fresh capture, mirroring how `wait` bypasses the cache. + */ +async function pollForSelectorReadiness( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + poll: SelectorPollBudget, +): Promise { + const signal = options.signal ?? runtime.signal; + let previousPoll: InteractionSnapshot | undefined; + const observed = await observeUntil({ + capture: async (pollSignal) => { + const attempt = await pollSelectorReadinessOnce( + runtime, + { ...options, signal: pollSignal }, + selectorExpression, + params, + previousPoll, + ); + previousPoll = attempt.capture; + return attempt; + }, + verdict: (latest) => + latest.resolved && latest.resolved.node.rect + ? { kind: 'done', result: { capture: latest.capture, resolved: latest.resolved } } + : { kind: 'continue' }, + schedule: { + intervalMs: poll.intervalMs, + budgetMs: poll.defaultTimeoutMs, + budgetFrom: 'first-capture', + }, + rideOut: isUnreadableCaptureContentError, + ...(signal ? { signal } : {}), + ...(runtime.clock ? { clock: runtime.clock } : {}), + phase: 'interaction_target_readiness', + }); + if (observed.kind === 'done') return observed.result; + if (observed.kind === 'failed') throw observed.error; + throw await readinessExhaustedFailure(runtime, selectorExpression, params, observed); +} + +/** + * One readiness poll: today's capture-and-resolve attempt, the previous poll's post-gesture outcome + * carried forward, and the covered-target probe on a miss. A covered candidate is excluded by + * promotedTarget's own occlusion stage before it reaches `resolved`, so it looks identical to + * "not found yet"; detecting it here ends the loop on this poll instead of spending the budget on a + * target that will never stop being covered. + */ +async function pollSelectorReadinessOnce( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + previousPoll: InteractionSnapshot | undefined, +): Promise { + const attempt = await attemptSelectorResolution( + runtime, + options, + selectorExpression, + params, + previousPoll !== undefined, + ); + if (previousPoll) inheritPostGestureOutcome(previousPoll.snapshot, attempt.capture.snapshot); + if (attempt.resolved?.node.rect) return attempt; + const covered = await detectCoveredSelectorTarget({ + runtime, + nodes: attempt.capture.snapshot.nodes, + selectorExpression, + action: params.action, + }); + if (covered) throw covered; + return attempt; +} + +/** + * The budget is spent or a capture stalled at the deadline. When no poll ever observed the tree, the + * last ridden-out capture error is the cause and is raised as such; otherwise the ordinary + * selector failure carries the readiness evidence. + */ +async function readinessExhaustedFailure( + runtime: AgentDeviceRuntime, + selectorExpression: string, + params: ResolveInteractionTargetParams, + observed: Extract< + Awaited>>, + { kind: 'expired' | 'stalled' } + >, +): Promise { + if ( + observed.last === undefined && + observed.kind === 'expired' && + observed.lastError !== undefined + ) { + return observed.lastError; + } + const readiness: SelectorReadinessDetails = { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: observed.kind, + }; + const failure = await selectorInteractionFailure({ + runtime, + nodes: observed.last?.capture.snapshot.nodes ?? [], + selectorExpression, + action: params.action, + resolved: observed.last?.resolved ?? null, + }); + failure.details = { ...failure.details, readiness }; + return failure; +} + +/** + * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites this as the + * runtime-selector `targetReadiness` `via` symbol, and the gate test imports it dynamically, which + * fallow cannot trace statically. + */ +// fallow-ignore-next-line unused-export +export async function resolveSelectorInteractionTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, + params: ResolveInteractionTargetParams, +): Promise { + const selectorExpression = target.selector; + let capture: InteractionSnapshot; + let resolved: SelectorResolution | null; + if (params.pipeline.poll === 'none') { + const attempt = await attemptSelectorResolution( runtime, - nodes: capture.snapshot.nodes, + options, selectorExpression, - action: params.action, - resolved, - }); + params, + false, + ); + capture = attempt.capture; + resolved = attempt.resolved; + if (!resolved || !resolved.node.rect) { + throw await selectorInteractionFailure({ + runtime, + nodes: capture.snapshot.nodes, + selectorExpression, + action: params.action, + resolved, + }); + } + } else { + const ready = await pollForSelectorReadiness( + runtime, + options, + selectorExpression, + params, + params.pipeline.poll, + ); + capture = ready.capture; + resolved = ready.resolved; } // #1542: see the ref-target twin above. const selected = resolved; @@ -456,31 +654,46 @@ async function resolveSelectorInteractionTarget( * that. Both probes name a policy row, so the two contracts stay visible side * by side instead of as two sets of engine knobs (#1630). */ -async function selectorInteractionFailure(params: { +/** + * The diagnosis row keeps covered nodes as candidates precisely so its occlusion stage can report + * them: "matched but covered" is a different failure than "did not match", produced by re-probing + * the same tree under `coveredDiagnosis` rather than by the acting row's own (candidate-excluding) + * resolution. Shared by the single-attempt failure path and the readiness poll: a covered target is + * a refusal, not an absence, and must end a poll loop rather than spend its budget (a `promotedTarget` + * poll cannot otherwise tell "excluded because covered" apart from "does not exist yet"). + */ +async function detectCoveredSelectorTarget(params: { runtime: AgentDeviceRuntime; nodes: SnapshotState['nodes']; selectorExpression: string; action: InteractionAction; - resolved: SelectorResolution | null; -}): Promise { - const { runtime, nodes, selectorExpression, action, resolved } = params; - // The diagnosis row keeps covered nodes as candidates precisely so its - // occlusion stage can report them: "matched but covered" is a different - // failure with a different recovery than "did not match". +}): Promise { + const { runtime, nodes, selectorExpression, action } = params; const covered = await resolveSelectorPipeline( SELECTOR_PIPELINE_POLICIES.coveredDiagnosis, nodes, selectorExpression, { platform: runtime.backend.platform }, ); - if (covered.kind === 'occluded') { - return buildCoveredInteractionError({ - label: `Selector ${covered.selector}`, - node: covered.node, - action, - selector: covered.selector, - }); - } + if (covered.kind !== 'occluded') return undefined; + return buildCoveredInteractionError({ + label: `Selector ${covered.selector}`, + node: covered.node, + action, + selector: covered.selector, + }); +} + +async function selectorInteractionFailure(params: { + runtime: AgentDeviceRuntime; + nodes: SnapshotState['nodes']; + selectorExpression: string; + action: InteractionAction; + resolved: SelectorResolution | null; +}): Promise { + const { runtime, nodes, selectorExpression, action, resolved } = params; + const covered = await detectCoveredSelectorTarget({ runtime, nodes, selectorExpression, action }); + if (covered) return covered; const diagnostics = resolved?.diagnostics ?? []; return new AppError( 'COMMAND_FAILED', @@ -736,6 +949,12 @@ export async function captureInteractionSnapshot( runtime: AgentDeviceRuntime, options: CommandContext, interactiveOnly: boolean, + /** + * Bypasses the daemon's short-lived selector capture cache, the way `wait`'s capture already does + * (`src/daemon/selector-runtime-backend.ts`). Defaults to false: every caller before the readiness + * poll relied on the cache, and the poll itself only sets this from its second capture on. + */ + forceFresh?: boolean, ): Promise { if (!runtime.backend.captureSnapshot) { throw new AppError('UNSUPPORTED_OPERATION', 'snapshot is not supported by this backend'); @@ -746,6 +965,7 @@ export async function captureInteractionSnapshot( const result = await runtime.backend.captureSnapshot(toBackendContext(runtime, options), { interactiveOnly, includeRects: true, + ...(forceFresh ? { forceFresh: true } : {}), }); const snapshot = result.snapshot ?? diff --git a/src/daemon/scroll-movement.ts b/src/daemon/scroll-movement.ts index 7bc96f6ca0..c9e9383397 100644 --- a/src/daemon/scroll-movement.ts +++ b/src/daemon/scroll-movement.ts @@ -14,7 +14,7 @@ import { containsPoint } from '@agent-device/kernel/rect'; import { AppError } from '@agent-device/kernel/errors'; import type { Point, Rect, SnapshotState } from '@agent-device/kernel/snapshot'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; -import { sleep } from '@agent-device/host-kit/retry'; +import { observeUntil, type ObservationSchedule } from '@agent-device/capture-kit/observe-until'; import { areInteractionSurfaceSignaturesStable, buildInteractionSurfaceSignature, @@ -62,8 +62,10 @@ import type { SessionState } from './session-state.ts'; */ /** How long a scroll keeps asking whether an untouched surface is really untouched (#1542's window). */ -const MOVEMENT_VERDICT_BUDGET_MS = 1_500; -const MOVEMENT_POLL_MS = 200; +const SCROLL_MOVEMENT_SCHEDULE: ObservationSchedule = { + intervalMs: 200, + budgetMs: 1_500, +}; /** The pre-gesture surface, already in hand: no capture is spent to produce it. */ export type ScrollSurfaceBaseline = Readonly<{ @@ -217,6 +219,12 @@ type SurfaceVerdict = startedAt: number; }>; +/** What one capture's verdict decides, before the loop's own timing is stitched back on. */ +type SurfaceJudgement = + | Readonly<{ kind: 'blind'; reason: SurfaceBlindReason }> + | Readonly<{ kind: 'moved'; observed: ObservedSurface }> + | Readonly<{ kind: 'settled'; observed: ObservedSurface; evidence: InteractionSurfaceChange }>; + /** Why this capture cannot be compared against the pre-gesture tree at all, movement included. */ type SurfaceBlindReason = 'capture-unreadable' | 'surface-unsettled' | ScrollSurfacePairDrift; @@ -257,30 +265,32 @@ async function pollForSurfaceVerdict( swipe: ScrollSwipeEvidence; }, ): Promise { - const startedAt = Date.now(); - const deadline = startedAt + (params.budgetMs ?? MOVEMENT_VERDICT_BUDGET_MS); + // The edge question is asked only of the pre-gesture (baseline) tree, so it never depends on a + // poll's outcome and can be answered once, before the loop starts, rather than lazily on the + // first `changed` reading. + const changeNeedsRest = await baselineEndsInDirection(baseline, params.direction, params.swipe); let previous: InteractionSurfaceSignature | undefined; - let attempts = 0; - let changeNeedsRest: boolean | undefined; - - while (true) { - const reading = await readOneCapture(baseline, params.capture); - attempts += 1; - if (reading.kind === 'blind') return { kind: 'blind', reason: reading.reason }; - if (reading.kind === 'changed') { - changeNeedsRest ??= await baselineEndsInDirection(baseline, params.direction, params.swipe); - } - const verdict = settledVerdict(reading, { - previous, - changeNeedsRest: changeNeedsRest === true, - attempts, - startedAt, - }); - if (verdict) return verdict; - if (Date.now() >= deadline) return budgetExpiredVerdict(params, attempts, startedAt); - previous = reading.observed.signature; - await sleep(params.pollMs ?? MOVEMENT_POLL_MS); - } + const observed = await observeUntil({ + capture: () => readOneCapture(baseline, params.capture), + schedule: { + intervalMs: params.pollMs ?? SCROLL_MOVEMENT_SCHEDULE.intervalMs, + budgetMs: params.budgetMs ?? SCROLL_MOVEMENT_SCHEDULE.budgetMs, + }, + verdict: (latest) => { + if (latest.kind === 'blind') + return { kind: 'done', result: { kind: 'blind', reason: latest.reason } }; + const judged = settledVerdict(latest, { previous, changeNeedsRest }); + previous = latest.observed.signature; + return judged ? { kind: 'done', result: judged } : { kind: 'continue' }; + }, + }); + + const attempts = observed.polls.length; + const startedAt = Date.now() - observed.waitedMs; + if (observed.kind !== 'done') return budgetExpiredVerdict(params, attempts, startedAt); + return observed.result.kind === 'blind' + ? observed.result + : { ...observed.result, attempts, startedAt }; } /** @@ -293,26 +303,17 @@ function settledVerdict( poll: { previous: InteractionSurfaceSignature | undefined; changeNeedsRest: boolean; - attempts: number; - startedAt: number; }, -): SurfaceVerdict | undefined { +): SurfaceJudgement | undefined { const atRest = surfaceIsAtRest(poll.previous, reading.observed.signature); - const { attempts, startedAt } = poll; if (reading.kind === 'changed') { if (!poll.changeNeedsRest || atRest) { - return { kind: 'moved', observed: reading.observed, attempts, startedAt }; + return { kind: 'moved', observed: reading.observed }; } return undefined; } if (!atRest) return undefined; - return { - kind: 'settled', - observed: reading.observed, - evidence: reading.evidence, - attempts, - startedAt, - }; + return { kind: 'settled', observed: reading.observed, evidence: reading.evidence }; } /** diff --git a/src/daemon/selector-runtime-backend.ts b/src/daemon/selector-runtime-backend.ts index 4af5d63305..80bd6af98a 100644 --- a/src/daemon/selector-runtime-backend.ts +++ b/src/daemon/selector-runtime-backend.ts @@ -188,10 +188,7 @@ function createSelectorBackend(params: SelectorRuntimeDeviceParams): AgentDevice const includeRects = options?.includeRects === true; const snapshotScope = options?.scope ?? req.flags?.snapshotScope; const needsFreshSnapshot = - req.command === 'wait' || - req.command === 'find' || - isAbsentPredicateRequest(req) || - (includeRects && device.platform === 'web'); + options?.forceFresh === true || requestNeedsFreshSnapshot(req, device, includeRects); return await captureRuntime.capture({ flags, signal: context.signal, @@ -241,3 +238,17 @@ function isAbsentPredicateRequest(req: DaemonRequest): boolean { const checked = checkIsArgs(req.positionals ?? []); return checked.ok && checked.predicate === 'absent'; } + +/** The requests whose read must not be served from the short-lived selector capture cache. */ +function requestNeedsFreshSnapshot( + req: Parameters[0], + device: { platform: string }, + includeRects: boolean, +): boolean { + return ( + req.command === 'wait' || + req.command === 'find' || + isAbsentPredicateRequest(req) || + (includeRects && device.platform === 'web') + ); +} diff --git a/test/integration/interaction-contract/runtime-selector.contract.test.ts b/test/integration/interaction-contract/runtime-selector.contract.test.ts index 07abb5349b..240185a48c 100644 --- a/test/integration/interaction-contract/runtime-selector.contract.test.ts +++ b/test/integration/interaction-contract/runtime-selector.contract.test.ts @@ -19,6 +19,7 @@ import { nonHittableButtonSnapshot, RUNNER_CONTINUE_NODES, settledWelcomeSnapshot, + viewportOnlySnapshot, } from './fixtures.ts'; import { createContractDevice } from './runtime-harness.ts'; import { runnerSnapshotEntry, runnerTapEntry, withIosContractDaemon } from './daemon-harness.ts'; @@ -197,6 +198,33 @@ test(scenario('nonHittable'), async () => { assert.match(result.hint ?? '', /hittable: false/); }); +test(scenario('targetReadiness'), async () => { + const taps: Point[] = []; + let captures = 0; + const device = createContractDevice(viewportOnlySnapshot(), { + captureSnapshot: async (_context, options) => { + captures += 1; + return { + // Polls after the first bypass the selector capture cache (forceFresh); the button appears + // only once that bypass is in effect, so a resolution the loop never actually polled for + // would leave `captures` at 1 and the tap unsent. + snapshot: options?.forceFresh ? continueButtonSnapshot() : viewportOnlySnapshot(), + }; + }, + tap: async (_context, point) => { + taps.push(point); + }, + }); + + const result = await device.interactions.press(selector('label=Continue'), { + session: 'default', + }); + + assert.equal(result.kind, 'selector'); + assert.ok(captures >= 3, `expected at least 2 polls (>=3 captures), got ${captures}`); + assert.deepEqual(taps, [{ x: 60, y: 40 }]); +}); + test(scenario('verifyEvidence'), async () => { const device = createContractDevice(continueButtonSnapshot(), { tap: async () => ({ ok: true }), diff --git a/test/integration/interaction-contract/runtime-selector.coverage.ts b/test/integration/interaction-contract/runtime-selector.coverage.ts index b11f593905..0d968bc93f 100644 --- a/test/integration/interaction-contract/runtime-selector.coverage.ts +++ b/test/integration/interaction-contract/runtime-selector.coverage.ts @@ -29,4 +29,6 @@ export const RUNTIME_SELECTOR_COVERAGE = definePathCoverage('runtime-selector', 'runtime-selector resolutionDisclosure: a unique match discloses the unique runtime shape', 'runtime-selector resolutionDisclosure: an equivalent wrapper chain discloses matchCount, winnerDiagnostic, and structural equivalence', ], + targetReadiness: + 'runtime-selector targetReadiness: a target missing on the first capture resolves once a later poll observes it', }); diff --git a/test/integration/provider-scenarios/press-target-readiness.test.ts b/test/integration/provider-scenarios/press-target-readiness.test.ts new file mode 100644 index 0000000000..aed99994ac --- /dev/null +++ b/test/integration/provider-scenarios/press-target-readiness.test.ts @@ -0,0 +1,171 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { assertRpcError, assertRpcOk } from './assertions.ts'; +import { PROVIDER_SCENARIO_IOS_SIMULATOR } from './fixtures.ts'; +import { createProviderScenarioHarness, withProviderScenarioResource } from './harness.ts'; +import { + createAppleRunnerProviderFromTranscript, + createRecordingAppleToolProvider, + simctlDeviceLifecycleHandler, +} from './providers.ts'; +import { createProviderTranscript, type ProviderScenarioProviderEntry } from './transcript.ts'; + +// promotedTarget readiness (press/click/longpress poll for the target to exist and become +// actionable before refusing): end-to-end proof through the real daemon stack, not the plain +// runtime harness — this is the only suite that exercises captureSnapshot's `forceFresh` threading +// through the daemon's selector-runtime-backend cache decision. + +const APP = 'com.example.app'; +const DEVICE_ID = PROVIDER_SCENARIO_IOS_SIMULATOR.id; + +const APPLICATION_ONLY_NODES = [ + { + index: 0, + type: 'Application', + label: 'Example', + rect: { x: 0, y: 0, width: 400, height: 800 }, + }, +]; + +const CONTINUE_BUTTON_NODES = [ + { + index: 0, + type: 'Application', + label: 'Example', + rect: { x: 0, y: 0, width: 400, height: 800 }, + }, + { + index: 1, + parentIndex: 0, + type: 'Button', + label: 'Continue', + hittable: true, + rect: { x: 100, y: 300, width: 200, height: 44 }, + }, +]; + +function snapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { + return { + command: 'ios.runner.snapshot', + deviceId: DEVICE_ID, + platform: 'apple', + result: { nodes, truncated: false }, + }; +} + +function repeatSnapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { + return { ...snapshotEntry(nodes), repeat: true }; +} + +function tapEntry(x: number, y: number): ProviderScenarioProviderEntry { + return { + command: 'ios.runner.tap', + deviceId: DEVICE_ID, + platform: 'apple', + result: { x, y }, + }; +} + +async function withPressReadinessDaemon( + entries: readonly ProviderScenarioProviderEntry[], + run: ( + daemon: Awaited>, + transcript: ReturnType, + ) => Promise, +): Promise { + const runnerTranscript = createProviderTranscript(entries); + const appleRunnerProvider = createAppleRunnerProviderFromTranscript( + runnerTranscript, + 'ios.runner', + ); + const appleTool = createRecordingAppleToolProvider({ + simctl: simctlDeviceLifecycleHandler('com.apple.CoreSimulator.SimRuntime.iOS-18-0', [ + { name: PROVIDER_SCENARIO_IOS_SIMULATOR.name, udid: DEVICE_ID }, + ]), + }); + + await withProviderScenarioResource( + async () => + await createProviderScenarioHarness({ + appleRunnerProvider: () => appleRunnerProvider, + appleToolProvider: () => appleTool.provider, + deviceInventoryProvider: async () => [PROVIDER_SCENARIO_IOS_SIMULATOR], + }), + async (daemon) => { + const open = await daemon.callCommand('open', [APP], { platform: 'ios', udid: DEVICE_ID }); + assertRpcOk(open); + await run(daemon, runnerTranscript); + }, + ); +} + +test('press waits for a selector missing on the first two captures, then taps once it appears', async () => { + await withPressReadinessDaemon( + [ + // Poll 1 (not yet forceFresh): interactive capture, then the interactive->full fallback — + // both miss because the button is not in the tree yet. + snapshotEntry(APPLICATION_ONLY_NODES), + snapshotEntry(APPLICATION_ONLY_NODES), + // Poll 2 onward (forceFresh: bypasses the selector capture cache): the button has appeared. + repeatSnapshotEntry(CONTINUE_BUTTON_NODES), + tapEntry(200, 322), + ], + async (daemon, transcript) => { + const press = await daemon.callCommand('press', ['label=Continue']); + const data = assertRpcOk(press); + assert.equal(data.x, 200); + assert.equal(data.y, 322); + + const commands = transcript.calls.map((call) => call.command); + const tapIndex = commands.indexOf('ios.runner.tap'); + assert.ok(tapIndex >= 0, 'expected the runner to receive a tap'); + assert.equal( + commands.filter((command) => command === 'ios.runner.tap').length, + 1, + 'expected exactly one tap dispatch', + ); + // Every snapshot call precedes the single tap: the loop never taps before a poll resolves. + assert.ok( + commands.slice(0, tapIndex).every((command) => command === 'ios.runner.snapshot'), + `expected only snapshot calls before the tap, got ${JSON.stringify(commands)}`, + ); + assert.ok( + tapIndex >= 3, + `expected at least 3 snapshot calls before the tap (interactive+fallback misses, then a resolved poll), got ${tapIndex}`, + ); + + transcript.assertComplete(); + }, + ); +}); + +test('press fails with the standard no-match error, carrying readiness evidence, when the target never appears', async () => { + await withPressReadinessDaemon([repeatSnapshotEntry(APPLICATION_ONLY_NODES)], async (daemon) => { + // No fake clock is wired into this daemon-composed runtime path (AgentDeviceRuntime.clock is + // never set by daemon composition), so this genuinely spends the ~2s promotedTarget readiness + // budget in wall-clock time before failing. + const press = await daemon.callCommand('press', ['label=Continue']); + const error = assertRpcError(press, 'COMMAND_FAILED', /Selector did not match/); + const details = error.details as { + reason: unknown; + readiness: { polls: number; waitedMs: number; end: string }; + }; + assert.equal(details.reason, 'selector_not_found'); + assert.ok(details.readiness.polls >= 2, `expected >=2 polls, got ${details.readiness.polls}`); + assert.ok( + details.readiness.waitedMs >= 2000, + `expected >=2000ms waited, got ${details.readiness.waitedMs}`, + ); + assert.equal(details.readiness.end, 'expired'); + }); +}, 15_000); + +test('press resolving on the first capture costs exactly one snapshot call (zero readiness cost on the success path)', async () => { + await withPressReadinessDaemon( + [snapshotEntry(CONTINUE_BUTTON_NODES), tapEntry(200, 322)], + async (daemon) => { + const press = await daemon.callCommand('press', ['label=Continue']); + assertRpcOk(press); + }, + ); +}); diff --git a/test/integration/provider-scenarios/scroll-movement-observation.test.ts b/test/integration/provider-scenarios/scroll-movement-observation.test.ts new file mode 100644 index 0000000000..90c4cf203f --- /dev/null +++ b/test/integration/provider-scenarios/scroll-movement-observation.test.ts @@ -0,0 +1,215 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import type { AppleToolProvider } from '@agent-device/platform-apple/tool-provider'; +import type { ExecOptions, ExecResult } from '@agent-device/host-kit/command'; +import { assertRpcError, assertRpcOk } from './assertions.ts'; +import { PROVIDER_SCENARIO_IOS_SIMULATOR } from './fixtures.ts'; +import { createProviderScenarioHarness, withProviderScenarioResource } from './harness.ts'; +import { + createAppleRunnerProviderFromTranscript, + createRecordingAppleToolProvider, + simctlDeviceLifecycleHandler, +} from './providers.ts'; +import { createProviderTranscript, type ProviderScenarioProviderEntry } from './transcript.ts'; + +const APP = 'com.example.app'; +const DEVICE_ID = PROVIDER_SCENARIO_IOS_SIMULATOR.id; + +// #2714 directional-scroll observation, ported onto `observeUntil` (packages/capture-kit/src/observe-until.ts): +// `scroll down` gates its own `movement` claim on the tree it held before the gesture against one +// (or more, while the surface still looks untouched) post-gesture capture. The unit-level coverage +// in src/daemon/__tests__/scroll-movement.test.ts pins the verdict against a scripted capture +// function directly; these two scenarios drive the SAME loop end to end through the daemon's real +// command routing and a scripted Apple runner, the way `settle-observation.test.ts` drives `--settle`. + +// Fixed tab-bar-free list: a ScrollView reporting hidden content below, holding two rows. `rowOffset` +// moves the rows to model a scroll that actually shifted content; `hiddenBelow` toggles whether the +// container still reports more to reveal, which is what the no-progress refusal keys on. +const CONTAINER = { x: 18, y: 178, width: 366, height: 662 }; + +function screen(rowOffset: number, hiddenBelow: boolean) { + return [ + { + index: 0, + type: 'Application', + label: 'Example', + rect: { x: 0, y: 0, width: 393, height: 852 }, + }, + { + index: 1, + parentIndex: 0, + type: 'ScrollView', + identifier: 'lab-list', + rect: CONTAINER, + ...(hiddenBelow ? { hiddenContentBelow: true } : {}), + }, + { + index: 2, + parentIndex: 1, + type: 'StaticText', + label: 'Row one', + // 300 (+ up to -100 offset) stays inside CONTAINER's clip (y 178..840): the presentation + // validator refuses a child frame that escapes its container's cumulative clip. + rect: { x: 24, y: 300 + rowOffset, width: 300, height: 20 }, + }, + { + index: 3, + parentIndex: 1, + type: 'StaticText', + label: 'Row two', + rect: { x: 24, y: 360 + rowOffset, width: 300, height: 20 }, + }, + ]; +} + +function snapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { + return { + command: 'ios.runner.snapshot', + deviceId: DEVICE_ID, + platform: 'apple', + result: { nodes, truncated: false }, + }; +} + +// A UI that never moves: every post-gesture capture returns the same tree, so the loop can never +// see a quiet-then-still-hidden pair on a fixed count. How many captures it takes to prove that is +// wall-clock (200ms poll floor), so the count is not scripted — a repeat entry serves every call. +function repeatSnapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { + return { ...snapshotEntry(nodes), repeat: true }; +} + +// Midpoint (201, 437) sits inside CONTAINER, matching the swipe fixture +// `src/daemon/__tests__/scroll-movement.test.ts` uses for the same geometry. +function scrollEntry(): ProviderScenarioProviderEntry { + return { + command: 'ios.runner.scroll', + deviceId: DEVICE_ID, + platform: 'apple', + result: { x: 201, y: 487, x2: 201, y2: 387 }, + }; +} + +/** + * The route ahead of a directional scroll's own capture resolves a live simulator target (a real + * host process lookup) before it ever reaches the scripted runner, so a bare `simctlDeviceLifecycleHandler` + * leaves every capture with a fresh, per-call "unknown generation" identity — comparable to nothing, + * which is `capture-lineage-drift` on the very first post-gesture read. Naming one fixed, stable + * "running app" job (a `launchctl list` line) and a fixed process start time (`ps -p -o lstart=`) + * lets the route resolve ONE identity and reuse it every capture, exactly like a real simulator whose + * app process never restarts mid-scroll — the AX bridge itself still isn't modeled, so every capture + * still falls back to the scripted runner, carrying that one stable identity instead of a random one. + */ +function iosSimulatorTool(): { provider: AppleToolProvider } { + const baseSimctl = simctlDeviceLifecycleHandler('com.apple.CoreSimulator.SimRuntime.iOS-18-0', [ + { name: PROVIDER_SCENARIO_IOS_SIMULATOR.name, udid: DEVICE_ID }, + ]); + const recorded = createRecordingAppleToolProvider({ + simctl: async (args, options) => { + if (args[0] === 'spawn' && args[1] === DEVICE_ID && args[2] === 'launchctl') { + return { stdout: `4242\t0\tUIKitApplication:${APP}[0x1]\n`, stderr: '', exitCode: 0 }; + } + return await baseSimctl(args, options); + }, + }); + const runCommand = async ( + cmd: string, + args: string[], + options?: ExecOptions, + ): Promise => { + // #2438's system-surface presence probe runs before target resolution: report every + // registered host absent so the route proceeds to resolve the (scripted) app target below, + // instead of reading the probe as 'unknown' and falling back with a fresh random identity. + if (cmd === 'pgrep') return { stdout: '', stderr: '', exitCode: 1 }; + if (cmd === 'ps' && args[0] === '-p') { + return { stdout: 'Thu Jan 1 00:00:00 1970\n', stderr: '', exitCode: 0 }; + } + return await recorded.provider.runCommand(cmd, args, options); + }; + return { provider: { ...recorded.provider, runCommand } }; +} + +test('Provider-backed integration scroll down answers moved on the first post-gesture capture', async () => { + const runnerTranscript = createProviderTranscript([ + snapshotEntry(screen(0, true)), // pre-scroll baseline + scrollEntry(), + snapshotEntry(screen(-100, true)), // post-gesture: rows shifted into view on the first read + ]); + const appleRunnerProvider = createAppleRunnerProviderFromTranscript( + runnerTranscript, + 'ios.runner', + ); + const appleTool = iosSimulatorTool(); + + await withProviderScenarioResource( + async () => + await createProviderScenarioHarness({ + appleRunnerProvider: () => appleRunnerProvider, + appleToolProvider: () => appleTool.provider, + deviceInventoryProvider: async () => [PROVIDER_SCENARIO_IOS_SIMULATOR], + }), + async (daemon) => { + assertRpcOk(await daemon.callCommand('open', [APP], { platform: 'ios', udid: DEVICE_ID })); + assertRpcOk(await daemon.callCommand('snapshot')); + + const callsBeforeScroll = runnerTranscript.calls.length; + const scroll = await daemon.callCommand('scroll', ['down']); + const scrollData = assertRpcOk<{ movement?: string }>(scroll); + + assert.equal(scrollData.movement, 'moved'); + // Gesture first, then exactly one post-gesture snapshot: a scroll that worked is confirmed by + // the cheapest possible read, and the order proves the read followed the gesture rather than + // racing it. + assert.deepEqual( + runnerTranscript.calls.slice(callsBeforeScroll).map((call) => call.command), + ['ios.runner.scroll', 'ios.runner.snapshot'], + ); + + runnerTranscript.assertComplete(); + }, + ); +}); + +test('Provider-backed integration scroll down that never moves a hidden-content container refuses with scroll_no_progress', async () => { + const runnerTranscript = createProviderTranscript([ + snapshotEntry(screen(0, true)), // pre-scroll baseline + scrollEntry(), + // Every post-gesture read is byte-identical to the baseline and the container still reports + // hidden content below: the gesture never reached it. + repeatSnapshotEntry(screen(0, true)), + ]); + const appleRunnerProvider = createAppleRunnerProviderFromTranscript( + runnerTranscript, + 'ios.runner', + ); + const appleTool = iosSimulatorTool(); + + await withProviderScenarioResource( + async () => + await createProviderScenarioHarness({ + appleRunnerProvider: () => appleRunnerProvider, + appleToolProvider: () => appleTool.provider, + deviceInventoryProvider: async () => [PROVIDER_SCENARIO_IOS_SIMULATOR], + }), + async (daemon) => { + assertRpcOk(await daemon.callCommand('open', [APP], { platform: 'ios', udid: DEVICE_ID })); + assertRpcOk(await daemon.callCommand('snapshot')); + + const callsBeforeScroll = runnerTranscript.calls.length; + const scroll = await daemon.callCommand('scroll', ['down']); + const errorData = assertRpcError(scroll, 'COMMAND_FAILED', /scroll down moved nothing/); + assert.equal( + errorData.details && (errorData.details as Record).reason, + 'scroll_no_progress', + ); + + const callsDuringScroll = runnerTranscript.calls.slice(callsBeforeScroll); + // Gesture first, then at least the quiet-pair minimum of post-gesture snapshots — an exact + // count would assert the loop's polling speed rather than the refusal itself. + assert.equal(callsDuringScroll[0]?.command, 'ios.runner.scroll'); + assert.ok( + callsDuringScroll.filter((call) => call.command === 'ios.runner.snapshot').length >= 2, + `expected at least two post-gesture captures, saw ${callsDuringScroll.length - 1}`, + ); + }, + ); +}); From 0d85953a3e98f9a75c120983399e09c9a7f702b4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Mon, 28 Sep 2026 13:26:52 +0200 Subject: [PATCH 02/23] feat(interaction): operator-only readiness budget for press/click/longpress promotedTarget.poll becomes {intervalMs, maxTimeoutMs} (no row default). A new operator-only readinessTimeoutMs common input field (no CLI flag, no MCP exposure) gates resolveSelectorInteractionTarget's poll: a positive integer caps at the row's maxTimeoutMs and runs the readiness loop; absent (MCP, CLI, and every caller today) takes the one-attempt path, so a wrong-selector miss fails on the first capture instead of paying a 2s wait. Replay defaults readinessTimeoutMs: 2_000 onto dispatched press/click/longpress steps unless already set. Split resolution.ts (1,411 lines) three ways: selector-readiness.ts owns the readiness poll and its capture-attempt/failure machinery, interaction-snapshot-capture.ts and covered-interaction-error.ts are new leaves shared with resolution.ts without a value-import cycle (verified via the layering scan). interaction-guarantees.ts's targetReadiness via now points at selector-readiness.ts#pollForSelectorReadiness. --- packages/contracts/src/client-gesture.ts | 19 +- packages/contracts/src/command-flags.ts | 6 + packages/contracts/src/facades/client.ts | 1 + .../contracts/src/interaction-guarantees.ts | 8 +- packages/contracts/src/request-envelope.ts | 6 + .../session-replay-action-runtime.ts | 23 +- .../selectors/src/selector-pipeline-policy.ts | 49 ++- .../selectors/src/selector-pipeline.test.ts | 17 +- packages/selectors/src/selector-pipeline.ts | 15 +- src/__tests__/client.test.ts | 17 + src/commands/__tests__/input-audience.test.ts | 1 + src/commands/command-flags.ts | 1 + src/commands/common-input-fields.ts | 23 ++ src/commands/interaction/index.ts | 3 + .../runtime/__tests__/test-utils/index.ts | 3 + .../runtime/covered-interaction-error.ts | 28 ++ src/commands/interaction/runtime/gestures.ts | 8 +- .../runtime/interaction-snapshot-capture.ts | 50 +++ .../interaction/runtime/interactions.ts | 6 + .../runtime/post-action-observation.ts | 2 +- .../interaction/runtime/resolution.ts | 357 ++---------------- .../runtime/selector-readiness.test.ts | 96 +++++ .../interaction/runtime/selector-readiness.ts | 291 ++++++++++++++ .../session-replay-action-runtime.test.ts | 86 +++++ .../internal/interaction-touch-press.ts | 2 + .../command-tools-operator-inputs.test.ts | 1 + .../runtime-selector.contract.test.ts | 3 + .../press-target-readiness.test.ts | 36 +- 28 files changed, 807 insertions(+), 351 deletions(-) create mode 100644 src/commands/interaction/runtime/covered-interaction-error.ts create mode 100644 src/commands/interaction/runtime/interaction-snapshot-capture.ts create mode 100644 src/commands/interaction/runtime/selector-readiness.test.ts create mode 100644 src/commands/interaction/runtime/selector-readiness.ts diff --git a/packages/contracts/src/client-gesture.ts b/packages/contracts/src/client-gesture.ts index 5928b727a9..bc3675723d 100644 --- a/packages/contracts/src/client-gesture.ts +++ b/packages/contracts/src/client-gesture.ts @@ -35,11 +35,22 @@ export type SettleCommandOptions = { timeoutMs?: number; }; +/** + * Operator-only readiness budget (#1656 promotedTarget row): how long a tap-shaped interaction may + * poll for a target that does not exist yet, capped at the row's maxTimeoutMs. Never model- or + * CLI-writable; omitted means the one-attempt resolution path unchanged from before this option + * existed. + */ +export type ReadinessBudgetOptions = { + readinessTimeoutMs?: number; +}; + export type ClickOptions = DeviceCommandBaseOptions & SelectorSnapshotCommandOptions & InteractionTarget & RepeatedPressOptions & - SettleCommandOptions & { + SettleCommandOptions & + ReadinessBudgetOptions & { button?: ClickButton; /** * Opt-in (#1047): return cheap post-action evidence (AX digest, node counts, @@ -53,14 +64,16 @@ export type PressOptions = DeviceCommandBaseOptions & SelectorSnapshotCommandOptions & InteractionTarget & RepeatedPressOptions & - SettleCommandOptions & { + SettleCommandOptions & + ReadinessBudgetOptions & { verify?: boolean; }; export type LongPressOptions = DeviceCommandBaseOptions & SelectorSnapshotCommandOptions & InteractionTarget & - SettleCommandOptions & { + SettleCommandOptions & + ReadinessBudgetOptions & { durationMs?: number; }; diff --git a/packages/contracts/src/command-flags.ts b/packages/contracts/src/command-flags.ts index f7777c1bd5..c9805621da 100644 --- a/packages/contracts/src/command-flags.ts +++ b/packages/contracts/src/command-flags.ts @@ -35,6 +35,12 @@ export type CommandFlags = Omit & { kind?: string; maestro?: MaestroRuntimeFlags; postGestureStabilization?: boolean; + /** + * Operator-only readiness budget for press/click/longpress (#1656 promotedTarget row), capped at + * its maxTimeoutMs. No CliFlags counterpart: never CLI- or model-writable, so it is a daemon-side + * addition here rather than an `Omit` member. + */ + readinessTimeoutMs?: number; snapshotIncludeHiddenContentHints?: boolean; leaseProvider?: string; provider?: string; diff --git a/packages/contracts/src/facades/client.ts b/packages/contracts/src/facades/client.ts index 8cd5d9dad2..f7ec417f78 100644 --- a/packages/contracts/src/facades/client.ts +++ b/packages/contracts/src/facades/client.ts @@ -55,6 +55,7 @@ export type { PanOptions, PinchOptions, PressOptions, + ReadinessBudgetOptions, RepeatedPressOptions, RotateGestureOptions, ScrollOptions, diff --git a/packages/contracts/src/interaction-guarantees.ts b/packages/contracts/src/interaction-guarantees.ts index 8224e6ecee..e816c1a295 100644 --- a/packages/contracts/src/interaction-guarantees.ts +++ b/packages/contracts/src/interaction-guarantees.ts @@ -287,11 +287,13 @@ export const INTERACTION_DISPATCH_PATHS: Record & durationMs?: number; holdMs?: number; jitterPx?: number; + /** + * Operator-only: the readiness budget a tap-shaped interaction (press/click/longpress) may + * spend polling for a target that does not exist yet, capped at the promotedTarget row's + * maxTimeoutMs. Never model- or CLI-writable; absent means the one-attempt resolution path. + */ + readinessTimeoutMs?: number; pixels?: number; /** Scroll: repeat passes until this selector is visible on screen. */ until?: string; diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index 3221deb61c..c1f0fca6d9 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -145,7 +145,7 @@ async function invokeResolvedReplayAction(params: { }): Promise { const { req, sessionName, resolved, sourceAction, invoke, resolvedSessionScope, dependencies } = params; - const flags = buildReplayActionFlags(req.flags, resolved.flags); + const flags = buildReplayActionFlags(req.flags, resolved.flags, resolved.command); const recordedInputVariable = sourceAction.command === 'fill' ? readRecordedInputVariableName(inferFillText(sourceAction)) @@ -215,9 +215,28 @@ function readResponseTiming(data: unknown): Record | undefined ); } +/** + * A press/click/longpress step replays against a UI that may still be a render or two behind where + * it was recorded — the same render race #1656's operator-only readiness budget absorbs live. A + * replayed step carries no recorded budget of its own (there is no CLI flag for it), so every such + * step defaults to one here unless the merged flags already name one — future-proofing against a + * flag source this PR does not add, rather than a live possibility today. + */ +const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; +const READINESS_BUDGETED_REPLAY_COMMANDS: ReadonlySet = new Set([ + 'press', + 'click', + 'longpress', +]); + function buildReplayActionFlags( parentFlags: CommandFlags | undefined, actionFlags: SessionAction['flags'] | undefined, + command: string, ): CommandFlags { - return mergeParentFlags(parentFlags, { ...(actionFlags ?? {}) }); + const flags = mergeParentFlags(parentFlags, { ...(actionFlags ?? {}) }); + if (READINESS_BUDGETED_REPLAY_COMMANDS.has(command) && flags.readinessTimeoutMs === undefined) { + flags.readinessTimeoutMs = REPLAY_DEFAULT_READINESS_TIMEOUT_MS; + } + return flags; } diff --git a/packages/selectors/src/selector-pipeline-policy.ts b/packages/selectors/src/selector-pipeline-policy.ts index b140f9bad5..e44ac24d06 100644 --- a/packages/selectors/src/selector-pipeline-policy.ts +++ b/packages/selectors/src/selector-pipeline-policy.ts @@ -63,8 +63,12 @@ export type SelectorOffscreenStage = 'refuse' | 'ignore'; */ export type SelectorPromotionStage = 'hittable-ancestor' | 'hittable-ancestor-below-root' | 'none'; -/** The poll budget of a row that polls, read by `selectorPollBudget`. */ -export type SelectorPollBudget = { +/** + * `wait`/`findWait`: the row supplies its own default deadline, spent whenever the caller passes + * no explicit timeout. Read by `selectorPollBudget`, which `createWaitPolling` derives its deadline + * and sleep from. + */ +export type WaitPollBudget = { /** Used when the caller passes no explicit timeout. */ defaultTimeoutMs: number; /** Delay between polls, clamped to the remaining budget. */ @@ -72,13 +76,27 @@ export type SelectorPollBudget = { }; /** - * Consumed by `selectorPollBudget`, which `createWaitPolling` derives its - * deadline and sleep from, and — for `promotedTarget` — by - * `resolveSelectorInteractionTarget`'s target-readiness loop. `'none'` is an - * answer, not an omission: a row that resolves against one capture has no - * polling contract, so asking for its budget is a caller bug — never a place - * to default one in. Acting rows are not all `'none'`: `promotedTarget` polls - * for the target to exist, while `resolvedTarget` and every read row still + * `promotedTarget`: the row states only a ceiling and a cadence, never a default — a miss with no + * caller-supplied budget takes the one-attempt path instead of polling (`resolveSelectorInteractionTarget`). + * When the caller does supply one (an operator-only `readinessTimeoutMs`, never model- or + * CLI-writable), it is capped at `maxTimeoutMs` before it bounds the readiness loop. + */ +export type ReadinessPollBudget = { + /** The ceiling a caller-supplied readiness timeout may not exceed. */ + maxTimeoutMs: number; + /** Delay between polls, clamped to the remaining budget. */ + intervalMs: number; +}; + +/** The poll budget of a row that polls: `wait`/`findWait`'s own default, or `promotedTarget`'s cap. */ +export type SelectorPollBudget = WaitPollBudget | ReadinessPollBudget; + +/** + * Consumed by `selectorPollBudget` (wait-shaped rows only) and — for `promotedTarget` — by + * `resolveSelectorInteractionTarget`'s target-readiness loop. `'none'` is an answer, not an + * omission: a row that resolves against one capture has no polling contract, so asking for its + * budget is a caller bug — never a place to default one in. Acting rows are not all `'none'`: + * `promotedTarget` polls for the target to exist, while `resolvedTarget` and every read row still * resolve against a single capture. */ export type SelectorPollStage = SelectorPollBudget | 'none'; @@ -109,21 +127,24 @@ export type SelectorPipelinePolicy = SelectorListPolicy & { }; /** Shared by both wait loops; they differ only in what each poll resolves. */ -const WAIT_POLL_BUDGET: SelectorPollBudget = { defaultTimeoutMs: 10_000, intervalMs: 300 }; +const WAIT_POLL_BUDGET: WaitPollBudget = { defaultTimeoutMs: 10_000, intervalMs: 300 }; export const SELECTOR_PIPELINE_POLICIES = { /** * `click`/`press`/`longpress`: the tap lands on the actionable owner of the - * match. Polls for the target to appear and become resolvable (a rect) before - * refusing — resolution only; occlusion, off-screen, and promotion still run - * once, against the winning capture, after the loop ends (#1656). + * match. When the caller supplies an operator-only readiness budget, polls for the target to + * appear and become resolvable (a rect), capped at this row's `maxTimeoutMs`, before refusing — + * resolution only; occlusion, off-screen, and promotion still run once, against the winning + * capture, after the loop ends (#1656). A miss with no caller-supplied budget takes the + * one-attempt path instead (`resolveSelectorInteractionTarget`), unchanged from before this + * budget existed. */ promotedTarget: { resolution: SELECTOR_RESOLUTION_POLICIES.act, occlusion: 'exclude-and-refuse', offscreen: 'refuse', promotion: 'hittable-ancestor', - poll: { defaultTimeoutMs: 2_000, intervalMs: 200 }, + poll: { maxTimeoutMs: 2_000, intervalMs: 200 }, }, /** * `fill`/`focus`/`scroll`/gesture endpoints, and the native-ref preflight — diff --git a/packages/selectors/src/selector-pipeline.test.ts b/packages/selectors/src/selector-pipeline.test.ts index 7d574a7363..1300459d06 100644 --- a/packages/selectors/src/selector-pipeline.test.ts +++ b/packages/selectors/src/selector-pipeline.test.ts @@ -316,17 +316,13 @@ test('the uniqueness rows still refuse matches that are not one wrapper chain', } }); -test('the poll stage answers only for the rows that poll', () => { +test('the poll stage answers only for the rows that poll a caller-default budget', () => { for (const row of ['wait', 'findWait'] as const) { assert.deepEqual(selectorPollBudget(SELECTOR_PIPELINE_POLICIES[row]), { defaultTimeoutMs: 10_000, intervalMs: 300, }); } - assert.deepEqual(selectorPollBudget(SELECTOR_PIPELINE_POLICIES.promotedTarget), { - defaultTimeoutMs: 2_000, - intervalMs: 200, - }); for (const row of NODE_STAGE_ROWS.filter( (name) => name !== 'wait' && name !== 'findWait' && name !== 'promotedTarget', )) { @@ -334,6 +330,17 @@ test('the poll stage answers only for the rows that poll', () => { } }); +test('promotedTarget states a caller-capped readiness budget, not a caller-default one', () => { + assert.deepEqual(SELECTOR_PIPELINE_POLICIES.promotedTarget.poll, { + maxTimeoutMs: 2_000, + intervalMs: 200, + }); + assert.throws( + () => selectorPollBudget(SELECTOR_PIPELINE_POLICIES.promotedTarget), + /caller-capped readiness budget/, + ); +}); + test('a listing row cannot be handed the node stages it does not declare', () => { // The `@ts-expect-error` directives ARE the assertion: a listing has no // single target to retarget, keep on screen, or wait for, so widening diff --git a/packages/selectors/src/selector-pipeline.ts b/packages/selectors/src/selector-pipeline.ts index f7d8518cba..f43ce7d408 100644 --- a/packages/selectors/src/selector-pipeline.ts +++ b/packages/selectors/src/selector-pipeline.ts @@ -14,9 +14,9 @@ import type { CandidateSetPipelinePolicy, SelectorListPolicy, SelectorPipelinePolicy, - SelectorPollBudget, SelectorPromotionStage, SingleTargetPipelinePolicy, + WaitPollBudget, } from './selector-pipeline-policy.ts'; /** @@ -262,12 +262,21 @@ async function guardOffscreen( return await hooks.offscreen(node, nodes); } -/** The poll stage: the budget `createWaitPolling` runs the row under. */ -export function selectorPollBudget(policy: SelectorPipelinePolicy): SelectorPollBudget { +/** + * The poll stage: the caller-default budget `createWaitPolling` runs a wait-shaped row under. + * `promotedTarget`'s row polls too, but under a caller-supplied, row-capped budget rather than a + * row-owned default — it has no `defaultTimeoutMs` to hand back, so it is not a legal input here. + */ +export function selectorPollBudget(policy: SelectorPipelinePolicy): WaitPollBudget { if (policy.poll === 'none') { throw new Error( 'selector pipeline row resolves against one capture and declares no poll budget', ); } + if (!('defaultTimeoutMs' in policy.poll)) { + throw new Error( + 'selector pipeline row polls under a caller-capped readiness budget, not a caller-default one', + ); + } return policy.poll; } diff --git a/src/__tests__/client.test.ts b/src/__tests__/client.test.ts index 7ae6d47b05..97864da213 100644 --- a/src/__tests__/client.test.ts +++ b/src/__tests__/client.test.ts @@ -1454,6 +1454,23 @@ test('interactions expose targetKind-discriminated public response data', async ); }); +// #1656 follow-up: the operator-only readiness budget rides the same common-input seam as every +// other option (`commonToClientOptions`), so it must reach `req.flags` for press/click/longpress +// exactly like `button`/`durationMs` do above. Never CLI- or model-writable — this is the SDK-only +// route the option exists for. +test('readinessTimeoutMs on press/click/longpress reaches the request flags', async () => { + const setup = createTransport(async () => ({ ok: true, data: {} })); + const client = createAgentDeviceClient(setup.config, { transport: setup.transport }); + + await client.interactions.press({ selector: 'label=Foo', readinessTimeoutMs: 2_000 }); + await client.interactions.click({ selector: 'label=Foo', readinessTimeoutMs: 1_500 }); + await client.interactions.longPress({ selector: 'label=Foo', readinessTimeoutMs: 900 }); + + assert.equal(setup.calls[0]?.flags?.readinessTimeoutMs, 2_000); + assert.equal(setup.calls[1]?.flags?.readinessTimeoutMs, 1_500); + assert.equal(setup.calls[2]?.flags?.readinessTimeoutMs, 900); +}); + // #1625: `find … list` through the typed client — `matches` is part of the // public response type and rides with the issuing generation, so a caller can // pin any listed ref for its next command. diff --git a/src/commands/__tests__/input-audience.test.ts b/src/commands/__tests__/input-audience.test.ts index eb4eb67562..18387d0203 100644 --- a/src/commands/__tests__/input-audience.test.ts +++ b/src/commands/__tests__/input-audience.test.ts @@ -85,5 +85,6 @@ function commonInputValueFor(key: string): unknown { if (key === 'platform') return 'ios'; if (key === 'deviceTarget' || key === 'target') return 'mobile'; if (key === 'debug') return true; + if (key === 'readinessTimeoutMs') return 2_000; return `value-${key}`; } diff --git a/src/commands/command-flags.ts b/src/commands/command-flags.ts index 4699c2c0bc..f7b44a4618 100644 --- a/src/commands/command-flags.ts +++ b/src/commands/command-flags.ts @@ -87,6 +87,7 @@ function buildFlags(options: InternalRequestOptions): CommandFlags { durationMs: options.durationMs, holdMs: options.holdMs, jitterPx: options.jitterPx, + readinessTimeoutMs: options.readinessTimeoutMs, pixels: options.pixels, until: options.until, doubleTap: options.doubleTap, diff --git a/src/commands/common-input-fields.ts b/src/commands/common-input-fields.ts index e53672c5c9..bf266c8827 100644 --- a/src/commands/common-input-fields.ts +++ b/src/commands/common-input-fields.ts @@ -1,4 +1,5 @@ import type { CliFlags } from '@agent-device/contracts/command'; +import { readOptionalInteger } from '@agent-device/contracts/command'; import type { AgentDeviceRequestOverrides, AgentDeviceSelectionOptions, @@ -30,6 +31,12 @@ export type CommonCommandInput = Pick< androidDeviceAllowlist?: string; /** `--no-record`: common to every recordable command (see `commonInputFromFlags`). */ noRecord?: boolean; + /** + * Operator-only readiness budget for a tap-shaped interaction (press/click/longpress), capped at + * the promotedTarget row's maxTimeoutMs. No `flagIn`/`flagKey`: it never rides a CLI flag or the + * client "selection" projection, only the CLI/Node structured-input and SDK client option seams. + */ + readinessTimeoutMs?: number; }; export type CommonInputReadOptions = { readTargetAlias?: boolean }; @@ -212,6 +219,22 @@ const COMMON_INPUT_FIELDS = { flagKey: 'noRecord', flagIn: ['input', 'selection'], }, + readinessTimeoutMs: { + schema: { + type: 'integer', + minimum: 1, + description: + "Operator-only: how long press/click/longpress may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", + }, + read: (record) => readOptionalInteger(record, 'readinessTimeoutMs', { min: 1 }), + // Not accepted as a tool argument, and no CLI flag exists for it (deliberately: agents mostly + // miss on a wrong selector, where fast feedback matters more than absorbing a render race) — an + // operator supplies it directly through the CLI/Node structured input or the SDK client option. + audience: operatorAudience({ + operatorPath: + 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', + }), + }, daemonBaseUrl: { schema: { type: 'string', description: 'Remote daemon base URL.' }, read: (record) => optionalString(record, 'daemonBaseUrl'), diff --git a/src/commands/interaction/index.ts b/src/commands/interaction/index.ts index 6ab009896b..e17f46c9e3 100644 --- a/src/commands/interaction/index.ts +++ b/src/commands/interaction/index.ts @@ -667,6 +667,7 @@ function toClickOptions(input: ClickInput): ClickOptions { ...toRepeatedOptions(input), button: input.button, verify: input.verify, + readinessTimeoutMs: input.readinessTimeoutMs, ...toSettleOptions(input), }; } @@ -678,6 +679,7 @@ function toPressOptions(input: PressInput): PressOptions { ...toSelectorSnapshotOptions(input), ...toRepeatedOptions(input), verify: input.verify, + readinessTimeoutMs: input.readinessTimeoutMs, ...toSettleOptions(input), }; } @@ -701,6 +703,7 @@ function toLongPressOptions(input: LongPressInput): LongPressOptions { ...toClientInteractionTarget(input.target), ...toSelectorSnapshotOptions(input), durationMs: input.durationMs, + readinessTimeoutMs: input.readinessTimeoutMs, ...toSettleOptions(input), }; } diff --git a/src/commands/interaction/runtime/__tests__/test-utils/index.ts b/src/commands/interaction/runtime/__tests__/test-utils/index.ts index c6013d40aa..83a48df40e 100644 --- a/src/commands/interaction/runtime/__tests__/test-utils/index.ts +++ b/src/commands/interaction/runtime/__tests__/test-utils/index.ts @@ -342,6 +342,8 @@ export function createInteractionDevice( > & { platform?: AgentDeviceBackend['platform']; sessionMetadata?: Record; + /** An advancing clock, for interactions that poll (the promotedTarget readiness loop). */ + clock?: { now: () => number; sleep: (ms: number) => Promise }; } = {}, ) { return createAgentDevice({ @@ -378,6 +380,7 @@ export function createInteractionDevice( { name: 'default', snapshot, metadata: overrides.sessionMetadata }, ]), policy: localCommandPolicy(), + ...(overrides.clock ? { clock: overrides.clock } : {}), }); } diff --git a/src/commands/interaction/runtime/covered-interaction-error.ts b/src/commands/interaction/runtime/covered-interaction-error.ts new file mode 100644 index 0000000000..170946ad01 --- /dev/null +++ b/src/commands/interaction/runtime/covered-interaction-error.ts @@ -0,0 +1,28 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotNode } from '@agent-device/kernel/snapshot'; +import type { InteractionAction } from './resolution.ts'; +import { interactionVerb } from './interaction-verb.ts'; + +/** + * The one construction site for "covered by another visible element" refusals, shared by + * `resolution.ts`'s node-stage runner (ref, selector, native-ref preflight) and + * `selector-readiness.ts`'s covered-target diagnosis probe. A leaf so both can import it without + * either owning the other. + */ +export function buildCoveredInteractionError(params: { + label: string; + node: SnapshotNode; + action: InteractionAction; + selector?: string; +}): AppError { + return new AppError( + 'COMMAND_FAILED', + `${params.label} is covered by another visible element and cannot ${interactionVerb(params.action)} safely`, + { + hint: 'Use a different visible target, scroll it clear of the overlay, or inspect with snapshot/screenshot before retrying.', + ...(params.selector ? { selector: params.selector } : {}), + ref: `@${params.node.ref}`, + interactionBlocked: params.node.interactionBlocked, + }, + ); +} diff --git a/src/commands/interaction/runtime/gestures.ts b/src/commands/interaction/runtime/gestures.ts index a4b045e79e..eee9585532 100644 --- a/src/commands/interaction/runtime/gestures.ts +++ b/src/commands/interaction/runtime/gestures.ts @@ -27,13 +27,13 @@ import { } from './post-action-observation.ts'; import { assertSupportedInteractionSurface, - captureInteractionSnapshot, dispatchNativeRefInteraction, resolveInteractionTarget, type ExpectedResolvedTarget, type InteractionTarget, type ResolvedInteractionTarget, } from './resolution.ts'; +import { captureInteractionSnapshot } from './interaction-snapshot-capture.ts'; import { resolveVisibleSnapshotViewport } from './viewport.ts'; type DragRecordingTarget = { @@ -95,6 +95,11 @@ export type FocusCommandResult = ResolvedInteractionTarget & BackendResultEnvelo export type LongPressCommandOptions = CommandContext & { target: InteractionTarget; durationMs?: number; + /** + * Operator-only readiness budget (#1656 promotedTarget row): polls for the target to exist and + * become actionable, capped at the row's maxTimeoutMs. Absent takes the one-attempt path. + */ + readinessTimeoutMs?: number; /** ADR 0012 step 4: replay-only post-resolution guard; see resolution.ts. */ expectedResolvedTarget?: ExpectedResolvedTarget; } & SettlePostActionObservationOptions; @@ -142,6 +147,7 @@ export const longPressCommand: RuntimeCommand< pipeline: SELECTOR_PIPELINE_POLICIES.promotedTarget, captureEvidenceBaseline: observation.needsPreActionBaseline, expectedResolvedTarget: options.expectedResolvedTarget, + readinessTimeoutMs: options.readinessTimeoutMs, }); if (!runtime.backend.longPress) { throw new AppError('UNSUPPORTED_OPERATION', 'longPress is not supported by this backend'); diff --git a/src/commands/interaction/runtime/interaction-snapshot-capture.ts b/src/commands/interaction/runtime/interaction-snapshot-capture.ts new file mode 100644 index 0000000000..9f2da3ab25 --- /dev/null +++ b/src/commands/interaction/runtime/interaction-snapshot-capture.ts @@ -0,0 +1,50 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotState } from '@agent-device/kernel/snapshot'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { now, toBackendContext } from '../../runtime-common.ts'; + +/** + * A leaf shared by every interaction runtime module that needs a fresh (or session-cached) tree: + * `resolution.ts`'s ref/selector resolution, `selector-readiness.ts`'s readiness poll, + * `gestures.ts`'s viewport resolution, and `post-action-observation.ts`'s `--verify` baseline. + * Deliberately self-contained — it imports no sibling of this directory, so those siblings can + * import it without ever risking a cycle back. + */ + +export type InteractionSnapshot = { + snapshot: SnapshotState; +}; + +export async function captureInteractionSnapshot( + runtime: AgentDeviceRuntime, + options: CommandContext, + interactiveOnly: boolean, + /** + * Bypasses the daemon's short-lived selector capture cache, the way `wait`'s capture already does + * (`src/daemon/selector-runtime-backend.ts`). Defaults to false: every caller before the readiness + * poll relied on the cache, and the poll itself only sets this from its second capture on. + */ + forceFresh?: boolean, +): Promise { + if (!runtime.backend.captureSnapshot) { + throw new AppError('UNSUPPORTED_OPERATION', 'snapshot is not supported by this backend'); + } + const sessionName = options.session ?? 'default'; + const session = await runtime.sessions.get(sessionName); + if (!session) throw new AppError('SESSION_NOT_FOUND', 'No active session. Run open first.'); + const result = await runtime.backend.captureSnapshot(toBackendContext(runtime, options), { + interactiveOnly, + includeRects: true, + ...(forceFresh ? { forceFresh: true } : {}), + }); + const snapshot = + result.snapshot ?? + ({ + nodes: result.nodes ?? [], + truncated: result.truncated, + backend: result.backend as SnapshotState['backend'], + createdAt: now(runtime), + } satisfies SnapshotState); + await runtime.sessions.set({ ...session, snapshot }); + return { snapshot }; +} diff --git a/src/commands/interaction/runtime/interactions.ts b/src/commands/interaction/runtime/interactions.ts index 5f5c18afe9..ac8656eb2f 100644 --- a/src/commands/interaction/runtime/interactions.ts +++ b/src/commands/interaction/runtime/interactions.ts @@ -41,6 +41,11 @@ export type PressCommandOptions = CommandContext & RepeatedInput & { target: InteractionTarget; button?: ClickButton; + /** + * Operator-only readiness budget (#1656 promotedTarget row): polls for the target to exist and + * become actionable, capped at the row's maxTimeoutMs. Absent takes the one-attempt path. + */ + readinessTimeoutMs?: number; /** ADR 0012 step 4: replay-only post-resolution guard; see resolution.ts. */ expectedResolvedTarget?: ExpectedResolvedTarget; /** #1654: a mutating `find`'s already-resolved node; see resolution.ts. */ @@ -152,6 +157,7 @@ async function tapCommand( captureEvidenceBaseline: observation.needsPreActionBaseline, expectedResolvedTarget: options.expectedResolvedTarget, preresolvedTarget: options.preresolvedTarget, + readinessTimeoutMs: options.readinessTimeoutMs, }); if (!runtime.backend.tap) { throw new AppError('UNSUPPORTED_OPERATION', 'tap is not supported by this backend'); diff --git a/src/commands/interaction/runtime/post-action-observation.ts b/src/commands/interaction/runtime/post-action-observation.ts index 68f78de53b..a0b513f192 100644 --- a/src/commands/interaction/runtime/post-action-observation.ts +++ b/src/commands/interaction/runtime/post-action-observation.ts @@ -5,7 +5,7 @@ import type { SettleObservation, SettleParams, } from '@agent-device/contracts/interaction'; -import { captureInteractionSnapshot } from './resolution.ts'; +import { captureInteractionSnapshot } from './interaction-snapshot-capture.ts'; import { summarizePostActionEvidence, surfaceScopedNodes } from './post-action-surface.ts'; import { settleAfterInteraction, settleEvidence } from './settle.ts'; diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index ac95c5bd36..8a7fc041fb 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -5,11 +5,7 @@ import type { SnapshotNode, SnapshotState, } from '@agent-device/kernel/snapshot'; -import { - findNodeByRef, - inheritPostGestureOutcome, - normalizeRef, -} from '@agent-device/kernel/snapshot'; +import { findNodeByRef, normalizeRef } from '@agent-device/kernel/snapshot'; import { resolveRectCenter } from '@agent-device/kernel/rect-center'; import type { AgentDeviceRuntime, @@ -17,14 +13,11 @@ import type { CommandSessionRecord, } from '../../../runtime-contract.ts'; import { - formatSelectorFailure, - selectorFailureHint, STALE_REF_HINT, type SelectorResolution, buildSelectorChainForNode, } from '@agent-device/selectors'; import { - resolveSelectorPipeline, runNodePipelineStages, type SelectorPipelineHooks, } from '@agent-device/selectors/selector-pipeline'; @@ -32,11 +25,19 @@ import { SELECTOR_PIPELINE_POLICIES, type ActingPipelinePolicy, type SelectorPipelinePolicy, - type SelectorPollBudget, } from '@agent-device/selectors/selector-pipeline-policy'; -import { observeUntil } from '@agent-device/capture-kit/observe-until'; -import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; import { resolvePressRecordingTarget } from '@agent-device/selectors/press-retarget'; +import { + attemptSelectorResolution, + pollForSelectorReadiness, + selectorInteractionFailure, + type ReadinessSchedule, +} from './selector-readiness.ts'; +import { + captureInteractionSnapshot, + type InteractionSnapshot, +} from './interaction-snapshot-capture.ts'; +import { buildCoveredInteractionError } from './covered-interaction-error.ts'; import { requireSnapshotSession } from './selector-read-shared.ts'; import { findNodeByLabel, resolveRefLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; import { containsPoint } from '@agent-device/kernel/rect'; @@ -63,7 +64,7 @@ import type { BackendCommandContext, BackendRefTarget, } from '../../../backend.ts'; -import { now, toBackendContext } from '../../runtime-common.ts'; +import { toBackendContext } from '../../runtime-common.ts'; import { toBackendResult } from '../../runtime-types.ts'; import { resolveInteractionTouchPoint } from '@agent-device/selectors/interaction-touch-point'; import { @@ -76,12 +77,10 @@ import { REPLAY_TARGET_GUARD_MISMATCH_REASON, type ReplayTargetGuardDenotation, } from '@agent-device/contracts/replay'; -import { resolveActionSelector } from './selector-action-resolution.ts'; import { assertTapTargetClearOfVisibleKeyboard, describeKeyboardOccludedPointWarning, } from './keyboard-occlusion.ts'; -import { interactionVerb } from './interaction-verb.ts'; export type { InteractionTarget, ResolvedInteractionTarget }; @@ -156,11 +155,7 @@ export type InteractionAction = | 'rotate' | 'transform'; -export type InteractionSnapshot = { - snapshot: SnapshotState; -}; - -type ResolveInteractionTargetParams = { +export type ResolveInteractionTargetParams = { action: InteractionAction; requireInteractive: boolean; /** @@ -170,6 +165,14 @@ type ResolveInteractionTargetParams = { * the element they resolved. */ pipeline: ActingPipelinePolicy; + /** + * Operator-only readiness budget (never model- or CLI-writable): how long a `promotedTarget` row + * may poll for a target that does not exist yet, before `resolveSelectorInteractionTarget` caps it + * at the row's `maxTimeoutMs`. Anything other than a positive integer — including `resolvedTarget` + * rows, which declare no poll budget at all — takes the one-attempt path unchanged from before + * this budget existed. + */ + readinessTimeoutMs?: number; /** * `--verify` (#1047): also capture the pre-action node set for a `point` target * so `changedFromBefore` evidence has a baseline. Ref/selector targets already @@ -382,197 +385,27 @@ async function resolveRefInteractionTarget( }; } -/** One poll's outcome: the capture it read the tree from, and what it resolved (if anything). */ -type SelectorResolutionAttempt = { - capture: InteractionSnapshot; - resolved: SelectorResolution | null; -}; - -/** The narrowed attempt a readiness poll accepts: a resolution with a usable rect. */ -type ResolvedSelectorAttempt = { - capture: InteractionSnapshot; - resolved: SelectorResolution; -}; - -/** - * One capture-and-resolve attempt: interactive capture, resolve; on a miss that requires - * interactivity, fall back to a full (non-interactive) capture and resolve again. Identical body - * whether run once (`promotedTarget.poll === 'none'`'s twins, `resolvedTarget`) or repeated under - * `observeUntil` for a row that polls — this function draws no distinction, and does not decide - * whether the miss is a refusal (ambiguity, occlusion, off-screen) or a plain no-match: those stay - * the caller's decisions, made from `resolved`/thrown errors exactly as before. - */ -async function attemptSelectorResolution( - runtime: AgentDeviceRuntime, - options: CommandContext, - selectorExpression: string, - params: ResolveInteractionTargetParams, - forceFresh: boolean, -): Promise { - let capture = await captureInteractionSnapshot( - runtime, - options, - params.requireInteractive, - forceFresh, - ); - let resolved = resolveActionSelector( - capture.snapshot.nodes, - selectorExpression, - runtime.backend.platform, - params.pipeline, - ); - if ((!resolved || !resolved.node.rect) && params.requireInteractive) { - const interactive = capture.snapshot; - capture = await captureInteractionSnapshot(runtime, options, false, forceFresh); - inheritPostGestureOutcome(interactive, capture.snapshot); - resolved = resolveActionSelector( - capture.snapshot.nodes, - selectorExpression, - runtime.backend.platform, - params.pipeline, - ); - } - return { capture, resolved }; +function isPositiveInteger(value: number | undefined): value is number { + return typeof value === 'number' && Number.isInteger(value) && value > 0; } -/** The readiness poll's evidence, attached to a target-not-found failure only (never to a refusal). */ -type SelectorReadinessDetails = { - polls: number; - waitedMs: number; - end: 'expired' | 'stalled'; -}; - /** - * `promotedTarget`'s readiness budget: the target may not exist yet, so a - * plain no-match (no resolution, or a resolution with no usable rect) keeps polling under the row's - * budget instead of refusing on the first capture. Every other outcome stays terminal exactly as a - * single attempt would produce it — an ambiguity throw from `resolveActionSelector`, or a capture - * error the loop does not ride out, ends the loop on this same poll, and is rethrown unchanged. - * Occlusion, off-screen, non-hittable, and keyboard refusals are not judged here: they run once, - * after this loop returns a rect-bearing resolution, exactly as `runInteractionPipelineStages` - * always has. - * - * The first poll is unbounded and reuses today's snapshot cache (`forceFresh: false`), so a caller - * whose first capture already matches pays the same one-or-two-capture cost as before. Only a poll - * that follows a miss (poll 2+) forces a fresh capture, mirroring how `wait` bypasses the cache. + * The schedule THIS call may poll `promotedTarget` under: `undefined` when the row resolves + * against one capture (`resolvedTarget`'s `poll: 'none'`), or when the caller supplied no + * `readinessTimeoutMs` — both take the one-attempt path exactly as before this budget existed + * (MCP and CLI never supply one, so neither surface changes). Otherwise the row's own cadence, and + * a budget capped at the row's `maxTimeoutMs` so an operator-supplied value can shrink the wait but + * never stretch past what the row allows. */ -async function pollForSelectorReadiness( - runtime: AgentDeviceRuntime, - options: CommandContext, - selectorExpression: string, - params: ResolveInteractionTargetParams, - poll: SelectorPollBudget, -): Promise { - const signal = options.signal ?? runtime.signal; - let previousPoll: InteractionSnapshot | undefined; - const observed = await observeUntil({ - capture: async (pollSignal) => { - const attempt = await pollSelectorReadinessOnce( - runtime, - { ...options, signal: pollSignal }, - selectorExpression, - params, - previousPoll, - ); - previousPoll = attempt.capture; - return attempt; - }, - verdict: (latest) => - latest.resolved && latest.resolved.node.rect - ? { kind: 'done', result: { capture: latest.capture, resolved: latest.resolved } } - : { kind: 'continue' }, - schedule: { - intervalMs: poll.intervalMs, - budgetMs: poll.defaultTimeoutMs, - budgetFrom: 'first-capture', - }, - rideOut: isUnreadableCaptureContentError, - ...(signal ? { signal } : {}), - ...(runtime.clock ? { clock: runtime.clock } : {}), - phase: 'interaction_target_readiness', - }); - if (observed.kind === 'done') return observed.result; - if (observed.kind === 'failed') throw observed.error; - throw await readinessExhaustedFailure(runtime, selectorExpression, params, observed); +function readinessScheduleFor( + poll: ActingPipelinePolicy['poll'], + readinessTimeoutMs: number | undefined, +): ReadinessSchedule | undefined { + if (poll === 'none' || !isPositiveInteger(readinessTimeoutMs)) return undefined; + return { intervalMs: poll.intervalMs, budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs) }; } -/** - * One readiness poll: today's capture-and-resolve attempt, the previous poll's post-gesture outcome - * carried forward, and the covered-target probe on a miss. A covered candidate is excluded by - * promotedTarget's own occlusion stage before it reaches `resolved`, so it looks identical to - * "not found yet"; detecting it here ends the loop on this poll instead of spending the budget on a - * target that will never stop being covered. - */ -async function pollSelectorReadinessOnce( - runtime: AgentDeviceRuntime, - options: CommandContext, - selectorExpression: string, - params: ResolveInteractionTargetParams, - previousPoll: InteractionSnapshot | undefined, -): Promise { - const attempt = await attemptSelectorResolution( - runtime, - options, - selectorExpression, - params, - previousPoll !== undefined, - ); - if (previousPoll) inheritPostGestureOutcome(previousPoll.snapshot, attempt.capture.snapshot); - if (attempt.resolved?.node.rect) return attempt; - const covered = await detectCoveredSelectorTarget({ - runtime, - nodes: attempt.capture.snapshot.nodes, - selectorExpression, - action: params.action, - }); - if (covered) throw covered; - return attempt; -} - -/** - * The budget is spent or a capture stalled at the deadline. When no poll ever observed the tree, the - * last ridden-out capture error is the cause and is raised as such; otherwise the ordinary - * selector failure carries the readiness evidence. - */ -async function readinessExhaustedFailure( - runtime: AgentDeviceRuntime, - selectorExpression: string, - params: ResolveInteractionTargetParams, - observed: Extract< - Awaited>>, - { kind: 'expired' | 'stalled' } - >, -): Promise { - if ( - observed.last === undefined && - observed.kind === 'expired' && - observed.lastError !== undefined - ) { - return observed.lastError; - } - const readiness: SelectorReadinessDetails = { - polls: observed.polls.length, - waitedMs: observed.waitedMs, - end: observed.kind, - }; - const failure = await selectorInteractionFailure({ - runtime, - nodes: observed.last?.capture.snapshot.nodes ?? [], - selectorExpression, - action: params.action, - resolved: observed.last?.resolved ?? null, - }); - failure.details = { ...failure.details, readiness }; - return failure; -} - -/** - * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites this as the - * runtime-selector `targetReadiness` `via` symbol, and the gate test imports it dynamically, which - * fallow cannot trace statically. - */ -// fallow-ignore-next-line unused-export -export async function resolveSelectorInteractionTarget( +async function resolveSelectorInteractionTarget( runtime: AgentDeviceRuntime, options: CommandContext, target: Extract, @@ -581,7 +414,8 @@ export async function resolveSelectorInteractionTarget( const selectorExpression = target.selector; let capture: InteractionSnapshot; let resolved: SelectorResolution | null; - if (params.pipeline.poll === 'none') { + const readinessSchedule = readinessScheduleFor(params.pipeline.poll, params.readinessTimeoutMs); + if (!readinessSchedule) { const attempt = await attemptSelectorResolution( runtime, options, @@ -606,7 +440,7 @@ export async function resolveSelectorInteractionTarget( options, selectorExpression, params, - params.pipeline.poll, + readinessSchedule, ); capture = ready.capture; resolved = ready.resolved; @@ -646,65 +480,6 @@ export async function resolveSelectorInteractionTarget( }; } -/** - * No usable acting target. Before reporting "did not match", re-probe the same - * tree through the diagnosis row: a selector that DOES match but landed on a - * covered node is a different failure with a different recovery, and the - * acting row — rect-required, candidates rejected — cannot tell the caller - * that. Both probes name a policy row, so the two contracts stay visible side - * by side instead of as two sets of engine knobs (#1630). - */ -/** - * The diagnosis row keeps covered nodes as candidates precisely so its occlusion stage can report - * them: "matched but covered" is a different failure than "did not match", produced by re-probing - * the same tree under `coveredDiagnosis` rather than by the acting row's own (candidate-excluding) - * resolution. Shared by the single-attempt failure path and the readiness poll: a covered target is - * a refusal, not an absence, and must end a poll loop rather than spend its budget (a `promotedTarget` - * poll cannot otherwise tell "excluded because covered" apart from "does not exist yet"). - */ -async function detectCoveredSelectorTarget(params: { - runtime: AgentDeviceRuntime; - nodes: SnapshotState['nodes']; - selectorExpression: string; - action: InteractionAction; -}): Promise { - const { runtime, nodes, selectorExpression, action } = params; - const covered = await resolveSelectorPipeline( - SELECTOR_PIPELINE_POLICIES.coveredDiagnosis, - nodes, - selectorExpression, - { platform: runtime.backend.platform }, - ); - if (covered.kind !== 'occluded') return undefined; - return buildCoveredInteractionError({ - label: `Selector ${covered.selector}`, - node: covered.node, - action, - selector: covered.selector, - }); -} - -async function selectorInteractionFailure(params: { - runtime: AgentDeviceRuntime; - nodes: SnapshotState['nodes']; - selectorExpression: string; - action: InteractionAction; - resolved: SelectorResolution | null; -}): Promise { - const { runtime, nodes, selectorExpression, action, resolved } = params; - const covered = await detectCoveredSelectorTarget({ runtime, nodes, selectorExpression, action }); - if (covered) return covered; - const diagnostics = resolved?.diagnostics ?? []; - return new AppError( - 'COMMAND_FAILED', - formatSelectorFailure(selectorExpression, diagnostics, { unique: true }), - { - reason: INTERACTION_ERROR_REASONS.selectorNotFound, - hint: selectorFailureHint(diagnostics), - }, - ); -} - function assertReplayTargetResolution( node: SnapshotNode, nodes: SnapshotState['nodes'], @@ -927,58 +702,6 @@ async function runInteractionPipelineStages(params: return { node: target.node, tapPoint }; } -function buildCoveredInteractionError(params: { - label: string; - node: SnapshotNode; - action: InteractionAction; - selector?: string; -}): AppError { - return new AppError( - 'COMMAND_FAILED', - `${params.label} is covered by another visible element and cannot ${interactionVerb(params.action)} safely`, - { - hint: 'Use a different visible target, scroll it clear of the overlay, or inspect with snapshot/screenshot before retrying.', - ...(params.selector ? { selector: params.selector } : {}), - ref: `@${params.node.ref}`, - interactionBlocked: params.node.interactionBlocked, - }, - ); -} - -export async function captureInteractionSnapshot( - runtime: AgentDeviceRuntime, - options: CommandContext, - interactiveOnly: boolean, - /** - * Bypasses the daemon's short-lived selector capture cache, the way `wait`'s capture already does - * (`src/daemon/selector-runtime-backend.ts`). Defaults to false: every caller before the readiness - * poll relied on the cache, and the poll itself only sets this from its second capture on. - */ - forceFresh?: boolean, -): Promise { - if (!runtime.backend.captureSnapshot) { - throw new AppError('UNSUPPORTED_OPERATION', 'snapshot is not supported by this backend'); - } - const sessionName = options.session ?? 'default'; - const session = await runtime.sessions.get(sessionName); - if (!session) throw new AppError('SESSION_NOT_FOUND', 'No active session. Run open first.'); - const result = await runtime.backend.captureSnapshot(toBackendContext(runtime, options), { - interactiveOnly, - includeRects: true, - ...(forceFresh ? { forceFresh: true } : {}), - }); - const snapshot = - result.snapshot ?? - ({ - nodes: result.nodes ?? [], - truncated: result.truncated, - backend: result.backend as SnapshotState['backend'], - createdAt: now(runtime), - } satisfies SnapshotState); - await runtime.sessions.set({ ...session, snapshot }); - return { snapshot }; -} - export async function assertSupportedInteractionSurface( runtime: AgentDeviceRuntime, options: CommandContext, diff --git a/src/commands/interaction/runtime/selector-readiness.test.ts b/src/commands/interaction/runtime/selector-readiness.test.ts new file mode 100644 index 0000000000..fad786fab5 --- /dev/null +++ b/src/commands/interaction/runtime/selector-readiness.test.ts @@ -0,0 +1,96 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { AppError } from '@agent-device/kernel/errors'; +import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; +import { selector } from './selector-read-utils.ts'; +import { createFakeClock, createInteractionDevice } from './__tests__/test-utils/index.ts'; + +// #1656 follow-up: promotedTarget's readiness poll now runs only when the caller supplies an +// operator-only `readinessTimeoutMs`, capped at the row's `maxTimeoutMs` (2_000). These pin the +// gating/capping decision at the runtime layer; the end-to-end poll mechanics (interactive/fresh +// capture sequencing, covered-target diagnosis, ridden-out capture errors) are unit-tested at the +// daemon level in test/integration/provider-scenarios/press-target-readiness.test.ts. + +const CONTINUE_BUTTON = { + index: 0, + depth: 0, + type: 'Button', + label: 'Continue', + rect: { x: 10, y: 20, width: 100, height: 40 }, + hittable: true, +}; + +test('runtime press without readinessTimeoutMs takes the one-attempt path and reports no readiness evidence', async () => { + let captures = 0; + const device = createInteractionDevice(makeSnapshotState([]), { + captureSnapshot: async () => { + captures += 1; + return { snapshot: makeSnapshotState([]) }; + }, + }); + + await assert.rejects( + () => device.interactions.press(selector('label=Continue'), { session: 'default' }), + (error: unknown) => { + assert.ok(error instanceof AppError); + assert.equal(error.details?.readiness, undefined); + return true; + }, + ); + // The pre-existing interactive-capture-then-full-capture-fallback pair (unchanged from before + // this budget existed) — not the readiness loop's repeated polling. + assert.equal(captures, 2); +}); + +test('runtime press with readinessTimeoutMs polls until the target appears', async () => { + let captures = 0; + const device = createInteractionDevice(makeSnapshotState([]), { + clock: createFakeClock(), + captureSnapshot: async () => { + captures += 1; + // Every capture before the fourth misses; the fourth (and every one after) has the button. + return { snapshot: makeSnapshotState(captures >= 4 ? [CONTINUE_BUTTON] : []) }; + }, + tap: async () => ({ ok: true }), + }); + + const result = await device.interactions.press(selector('label=Continue'), { + session: 'default', + readinessTimeoutMs: 2_000, + }); + + assert.equal(result.kind, 'selector'); + assert.equal(result.node?.label, 'Continue'); + assert.ok( + captures >= 4, + `expected at least 4 captures before the target appeared, got ${captures}`, + ); +}); + +test('runtime press caps a readinessTimeoutMs larger than the row maxTimeoutMs at 2_000ms', async () => { + const device = createInteractionDevice(makeSnapshotState([]), { + clock: createFakeClock(), + captureSnapshot: async () => ({ snapshot: makeSnapshotState([]) }), + }); + + await assert.rejects( + () => + device.interactions.press(selector('label=Continue'), { + session: 'default', + // Far past the promotedTarget row's own maxTimeoutMs (2_000): the row's cap governs, not + // this caller-supplied value. + readinessTimeoutMs: 999_000, + }), + (error: unknown) => { + assert.ok(error instanceof AppError); + const readiness = error.details?.readiness as { waitedMs: number; end: string } | undefined; + assert.ok(readiness, 'expected readiness evidence on an exhausted poll'); + assert.equal(readiness.end, 'expired'); + assert.ok( + readiness.waitedMs >= 2_000 && readiness.waitedMs <= 2_200, + `expected waitedMs capped near 2_000ms, got ${readiness.waitedMs}`, + ); + return true; + }, + ); +}); diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts new file mode 100644 index 0000000000..338d4455eb --- /dev/null +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -0,0 +1,291 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotState } from '@agent-device/kernel/snapshot'; +import { inheritPostGestureOutcome } from '@agent-device/kernel/snapshot'; +import { + formatSelectorFailure, + selectorFailureHint, + type SelectorResolution, +} from '@agent-device/selectors'; +import { resolveSelectorPipeline } from '@agent-device/selectors/selector-pipeline'; +import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; +import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { observeUntil } from '@agent-device/capture-kit/observe-until'; +import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { + captureInteractionSnapshot, + type InteractionSnapshot, +} from './interaction-snapshot-capture.ts'; +import { buildCoveredInteractionError } from './covered-interaction-error.ts'; +import { resolveActionSelector } from './selector-action-resolution.ts'; +import type { InteractionAction, ResolveInteractionTargetParams } from './resolution.ts'; + +/** + * `promotedTarget`'s readiness poll (#1656 / operator-only `readinessTimeoutMs`): the one-attempt + * capture-and-resolve, the covered-target diagnosis it shares with the exhausted-budget failure, and + * the `observeUntil`-driven loop itself. Split out of `resolution.ts` (#1656 follow-up) once it grew + * past 1,300 lines; `resolution.ts#resolveSelectorInteractionTarget` is this module's one caller, + * deciding whether a call polls at all and, if so, under what capped budget. + */ + +/** One poll's outcome: the capture it read the tree from, and what it resolved (if anything). */ +export type SelectorResolutionAttempt = { + capture: InteractionSnapshot; + resolved: SelectorResolution | null; +}; + +/** The narrowed attempt a readiness poll accepts: a resolution with a usable rect. */ +export type ResolvedSelectorAttempt = { + capture: InteractionSnapshot; + resolved: SelectorResolution; +}; + +/** The readiness poll's evidence, attached to a target-not-found failure only (never to a refusal). */ +export type SelectorReadinessDetails = { + polls: number; + waitedMs: number; + end: 'expired' | 'stalled'; +}; + +/** + * The schedule one call may poll under: the row's own cadence, and a budget the caller + * (`resolution.ts#resolveSelectorInteractionTarget`) has already capped at the row's own ceiling. + * This module never reads the row or the caller-supplied `readinessTimeoutMs` directly — deciding + * whether and how long to poll is the caller's job; this module only runs the loop. + */ +export type ReadinessSchedule = { + intervalMs: number; + budgetMs: number; +}; + +/** + * One capture-and-resolve attempt: interactive capture, resolve; on a miss that requires + * interactivity, fall back to a full (non-interactive) capture and resolve again. Identical body + * whether run once (`promotedTarget.poll`'s one-attempt twin, `resolvedTarget`) or repeated under + * `observeUntil` for a call that polls — this function draws no distinction, and does not decide + * whether the miss is a refusal (ambiguity, occlusion, off-screen) or a plain no-match: those stay + * the caller's decisions, made from `resolved`/thrown errors exactly as before. + */ +export async function attemptSelectorResolution( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + forceFresh: boolean, +): Promise { + let capture = await captureInteractionSnapshot( + runtime, + options, + params.requireInteractive, + forceFresh, + ); + let resolved = resolveActionSelector( + capture.snapshot.nodes, + selectorExpression, + runtime.backend.platform, + params.pipeline, + ); + if ((!resolved || !resolved.node.rect) && params.requireInteractive) { + const interactive = capture.snapshot; + capture = await captureInteractionSnapshot(runtime, options, false, forceFresh); + inheritPostGestureOutcome(interactive, capture.snapshot); + resolved = resolveActionSelector( + capture.snapshot.nodes, + selectorExpression, + runtime.backend.platform, + params.pipeline, + ); + } + return { capture, resolved }; +} + +/** + * No usable acting target. Before reporting "did not match", re-probe the same + * tree through the diagnosis row: a selector that DOES match but landed on a + * covered node is a different failure with a different recovery, and the + * acting row — rect-required, candidates rejected — cannot tell the caller + * that. Both probes name a policy row, so the two contracts stay visible side + * by side instead of as two sets of engine knobs (#1630). + * + * The diagnosis row keeps covered nodes as candidates precisely so its occlusion stage can report + * them: "matched but covered" is a different failure than "did not match", produced by re-probing + * the same tree under `coveredDiagnosis` rather than by the acting row's own (candidate-excluding) + * resolution. Shared by the single-attempt failure path and the readiness poll: a covered target is + * a refusal, not an absence, and must end a poll loop rather than spend its budget (a `promotedTarget` + * poll cannot otherwise tell "excluded because covered" apart from "does not exist yet"). + */ +async function detectCoveredSelectorTarget(params: { + runtime: AgentDeviceRuntime; + nodes: SnapshotState['nodes']; + selectorExpression: string; + action: InteractionAction; +}): Promise { + const { runtime, nodes, selectorExpression, action } = params; + const covered = await resolveSelectorPipeline( + SELECTOR_PIPELINE_POLICIES.coveredDiagnosis, + nodes, + selectorExpression, + { platform: runtime.backend.platform }, + ); + if (covered.kind !== 'occluded') return undefined; + return buildCoveredInteractionError({ + label: `Selector ${covered.selector}`, + node: covered.node, + action, + selector: covered.selector, + }); +} + +/** + * Shared by the one-attempt path (`resolution.ts#resolveSelectorInteractionTarget`) and this + * module's own exhausted-budget path: the same "selector not found" shape either way, covered + * targets disclosed through `detectCoveredSelectorTarget` first. + */ +export async function selectorInteractionFailure(params: { + runtime: AgentDeviceRuntime; + nodes: SnapshotState['nodes']; + selectorExpression: string; + action: InteractionAction; + resolved: SelectorResolution | null; +}): Promise { + const { runtime, nodes, selectorExpression, action, resolved } = params; + const covered = await detectCoveredSelectorTarget({ runtime, nodes, selectorExpression, action }); + if (covered) return covered; + const diagnostics = resolved?.diagnostics ?? []; + return new AppError( + 'COMMAND_FAILED', + formatSelectorFailure(selectorExpression, diagnostics, { unique: true }), + { + reason: INTERACTION_ERROR_REASONS.selectorNotFound, + hint: selectorFailureHint(diagnostics), + }, + ); +} + +/** + * One readiness poll: today's capture-and-resolve attempt, the previous poll's post-gesture outcome + * carried forward, and the covered-target probe on a miss. A covered candidate is excluded by + * promotedTarget's own occlusion stage before it reaches `resolved`, so it looks identical to + * "not found yet"; detecting it here ends the loop on this poll instead of spending the budget on a + * target that will never stop being covered. + */ +async function pollSelectorReadinessOnce( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + previousPoll: InteractionSnapshot | undefined, +): Promise { + const attempt = await attemptSelectorResolution( + runtime, + options, + selectorExpression, + params, + previousPoll !== undefined, + ); + if (previousPoll) inheritPostGestureOutcome(previousPoll.snapshot, attempt.capture.snapshot); + if (attempt.resolved?.node.rect) return attempt; + const covered = await detectCoveredSelectorTarget({ + runtime, + nodes: attempt.capture.snapshot.nodes, + selectorExpression, + action: params.action, + }); + if (covered) throw covered; + return attempt; +} + +/** + * The budget is spent or a capture stalled at the deadline. When no poll ever observed the tree, the + * last ridden-out capture error is the cause and is raised as such; otherwise the ordinary + * selector failure carries the readiness evidence. + */ +async function readinessExhaustedFailure( + runtime: AgentDeviceRuntime, + selectorExpression: string, + params: ResolveInteractionTargetParams, + observed: Extract< + Awaited>>, + { kind: 'expired' | 'stalled' } + >, +): Promise { + if ( + observed.last === undefined && + observed.kind === 'expired' && + observed.lastError !== undefined + ) { + return observed.lastError; + } + const readiness: SelectorReadinessDetails = { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: observed.kind, + }; + const failure = await selectorInteractionFailure({ + runtime, + nodes: observed.last?.capture.snapshot.nodes ?? [], + selectorExpression, + action: params.action, + resolved: observed.last?.resolved ?? null, + }); + failure.details = { ...failure.details, readiness }; + return failure; +} + +/** + * `promotedTarget`'s readiness budget: the target may not exist yet, so a plain no-match (no + * resolution, or a resolution with no usable rect) keeps polling under `schedule` instead of + * refusing on the first capture. Every other outcome stays terminal exactly as a single attempt + * would produce it — an ambiguity throw from `resolveActionSelector`, or a capture error the loop + * does not ride out, ends the loop on this same poll, and is rethrown unchanged. Occlusion, + * off-screen, non-hittable, and keyboard refusals are not judged here: they run once, after this + * loop returns a rect-bearing resolution, exactly as `runInteractionPipelineStages` always has. + * + * The first poll is unbounded and reuses today's snapshot cache (`forceFresh: false`), so a caller + * whose first capture already matches pays the same one-or-two-capture cost as before. Only a poll + * that follows a miss (poll 2+) forces a fresh capture, mirroring how `wait` bypasses the cache. + * + * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites this as the + * runtime-selector `targetReadiness` `via` symbol, and the gate test imports it dynamically, which + * fallow cannot trace statically. + */ +// fallow-ignore-next-line unused-export +export async function pollForSelectorReadiness( + runtime: AgentDeviceRuntime, + options: CommandContext, + selectorExpression: string, + params: ResolveInteractionTargetParams, + schedule: ReadinessSchedule, +): Promise { + const signal = options.signal ?? runtime.signal; + let previousPoll: InteractionSnapshot | undefined; + const observed = await observeUntil({ + capture: async (pollSignal) => { + const attempt = await pollSelectorReadinessOnce( + runtime, + { ...options, signal: pollSignal }, + selectorExpression, + params, + previousPoll, + ); + previousPoll = attempt.capture; + return attempt; + }, + verdict: (latest) => + latest.resolved && latest.resolved.node.rect + ? { kind: 'done', result: { capture: latest.capture, resolved: latest.resolved } } + : { kind: 'continue' }, + schedule: { + intervalMs: schedule.intervalMs, + budgetMs: schedule.budgetMs, + budgetFrom: 'first-capture', + }, + rideOut: isUnreadableCaptureContentError, + ...(signal ? { signal } : {}), + ...(runtime.clock ? { clock: runtime.clock } : {}), + phase: 'interaction_target_readiness', + }); + if (observed.kind === 'done') return observed.result; + if (observed.kind === 'failed') throw observed.error; + throw await readinessExhaustedFailure(runtime, selectorExpression, params, observed); +} diff --git a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts index 2234c80b44..5358cdfa5b 100644 --- a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts +++ b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts @@ -57,3 +57,89 @@ test.each(['', ' '])( expect(session.actions[0]?.result?.text).toBe('${PASSWORD}'); }, ); + +// #1656 follow-up: a replayed press/click/longpress step races the same render gap the operator- +// only readiness budget exists for, but a replay step has no CLI flag to carry one — so the one +// place replay builds a step's dispatched flags (`invokeResolvedReplayAction`'s +// `buildReplayActionFlags`) defaults it in for those three commands. +test.each(['press', 'click', 'longpress'])( + 'replay defaults readinessTimeoutMs onto a dispatched %s step', + async (command) => { + const action: SessionAction = { ts: 0, command, positionals: ['label="Continue"'], flags: {} }; + let dispatchedFlags: Record | undefined; + const response = await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + expect(response.ok).toBe(true); + expect(dispatchedFlags?.readinessTimeoutMs).toBe(2_000); + }, +); + +test('replay keeps a readinessTimeoutMs the step already carries instead of overwriting it', async () => { + const action: SessionAction = { + ts: 0, + command: 'press', + positionals: ['label="Continue"'], + flags: { readinessTimeoutMs: 500 }, + }; + let dispatchedFlags: Record | undefined; + const response = await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + expect(response.ok).toBe(true); + expect(dispatchedFlags?.readinessTimeoutMs).toBe(500); +}); + +test('replay never defaults readinessTimeoutMs onto a non-acting step', async () => { + const action: SessionAction = { + ts: 0, + command: 'wait', + positionals: ['label="Continue"'], + flags: {}, + }; + let dispatchedFlags: Record | undefined; + const response = await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + expect(response.ok).toBe(true); + expect(dispatchedFlags?.readinessTimeoutMs).toBeUndefined(); +}); diff --git a/src/daemon/interaction/internal/interaction-touch-press.ts b/src/daemon/interaction/internal/interaction-touch-press.ts index f5649912b6..cd6287edea 100644 --- a/src/daemon/interaction/internal/interaction-touch-press.ts +++ b/src/daemon/interaction/internal/interaction-touch-press.ts @@ -177,6 +177,7 @@ async function runTargetedTouchInteraction(params: { return await runtime.interactions.longPress(target, { ...shared, durationMs: params.durationMs, + readinessTimeoutMs: flags?.readinessTimeoutMs, }); case 'hover': return await runtime.interactions.hover(target, shared); @@ -210,6 +211,7 @@ function pressRuntimeOptions( jitterPx: flags?.jitterPx, doubleTap: flags?.doubleTap, verify: flags?.verify, + readinessTimeoutMs: flags?.readinessTimeoutMs, // Only click/press take it: `find` dispatches click and fill, never // longpress or hover, so declaring it on their options would be an // unconsumed claim (#1649 review). diff --git a/src/mcp/__tests__/command-tools-operator-inputs.test.ts b/src/mcp/__tests__/command-tools-operator-inputs.test.ts index 64b6fcbf6f..01f0228b27 100644 --- a/src/mcp/__tests__/command-tools-operator-inputs.test.ts +++ b/src/mcp/__tests__/command-tools-operator-inputs.test.ts @@ -24,6 +24,7 @@ const OPERATOR_OWNED_KEYS = [ 'iosXctestrunFile', 'iosXctestDerivedDataPath', 'iosXctestEnvDir', + 'readinessTimeoutMs', ] as const; // The rest of the shared common input: keys a model may write, because they diff --git a/test/integration/interaction-contract/runtime-selector.contract.test.ts b/test/integration/interaction-contract/runtime-selector.contract.test.ts index 240185a48c..fd2d103891 100644 --- a/test/integration/interaction-contract/runtime-selector.contract.test.ts +++ b/test/integration/interaction-contract/runtime-selector.contract.test.ts @@ -218,6 +218,9 @@ test(scenario('targetReadiness'), async () => { const result = await device.interactions.press(selector('label=Continue'), { session: 'default', + // Operator-only readiness budget (#1656): never CLI- or model-writable, so this contract + // scenario supplies it directly, the one route it reaches a call through in this PR. + readinessTimeoutMs: 2_000, }); assert.equal(result.kind, 'selector'); diff --git a/test/integration/provider-scenarios/press-target-readiness.test.ts b/test/integration/provider-scenarios/press-target-readiness.test.ts index aed99994ac..ea6ecc382b 100644 --- a/test/integration/provider-scenarios/press-target-readiness.test.ts +++ b/test/integration/provider-scenarios/press-target-readiness.test.ts @@ -111,7 +111,11 @@ test('press waits for a selector missing on the first two captures, then taps on tapEntry(200, 322), ], async (daemon, transcript) => { - const press = await daemon.callCommand('press', ['label=Continue']); + // Operator-only readiness budget: never CLI- or model-writable, so this scenario supplies it + // directly as a request flag, the one route it reaches a call through in this PR. + const press = await daemon.callCommand('press', ['label=Continue'], { + readinessTimeoutMs: 2_000, + }); const data = assertRpcOk(press); assert.equal(data.x, 200); assert.equal(data.y, 322); @@ -144,7 +148,9 @@ test('press fails with the standard no-match error, carrying readiness evidence, // No fake clock is wired into this daemon-composed runtime path (AgentDeviceRuntime.clock is // never set by daemon composition), so this genuinely spends the ~2s promotedTarget readiness // budget in wall-clock time before failing. - const press = await daemon.callCommand('press', ['label=Continue']); + const press = await daemon.callCommand('press', ['label=Continue'], { + readinessTimeoutMs: 2_000, + }); const error = assertRpcError(press, 'COMMAND_FAILED', /Selector did not match/); const details = error.details as { reason: unknown; @@ -169,3 +175,29 @@ test('press resolving on the first capture costs exactly one snapshot call (zero }, ); }); + +// (d) #1656 follow-up: reviewers found the row-wide 2s wait wrong for agents, whose misses are +// usually a wrong selector — fast feedback matters more than absorbing a render race. Without an +// explicit readinessTimeoutMs (never CLI- or model-writable, so MCP and CLI never supply one), a +// miss takes the one-attempt path: exactly one capture-and-resolve attempt (the pre-existing +// interactive-then-full-capture fallback, unchanged from main — not the readiness loop's repeated +// polling), and no readiness evidence to attach. +test('press without a readinessTimeoutMs flag fails on the first capture attempt, with no readiness poll', async () => { + await withPressReadinessDaemon( + // Interactive capture, then the interactive->full fallback — both miss, exactly like poll 1 of + // the readiness loop above, because this IS that same one attempt, just never repeated. + [snapshotEntry(APPLICATION_ONLY_NODES), snapshotEntry(APPLICATION_ONLY_NODES)], + async (daemon, transcript) => { + const callsBeforePress = transcript.calls.length; + const press = await daemon.callCommand('press', ['label=Continue']); + const error = assertRpcError(press, 'COMMAND_FAILED', /Selector did not match/); + const details = error.details as { reason: unknown; readiness: unknown }; + assert.equal(details.reason, 'selector_not_found'); + assert.equal(details.readiness, undefined); + + const commands = transcript.calls.slice(callsBeforePress).map((call) => call.command); + assert.deepEqual(commands, ['ios.runner.snapshot', 'ios.runner.snapshot']); + transcript.assertComplete(); + }, + ); +}); From 49ca3eb5514dd4e6f8eb86259dc55714ccc36c93 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Mon, 28 Sep 2026 13:32:44 +0200 Subject: [PATCH 03/23] test(client): split the readinessTimeoutMs SDK test out of client.test.ts client.test.ts is already over the test-file-size-ratchet tripwire and may not grow; the new press/click/longpress readinessTimeoutMs test moves to its own small file instead. --- .../client-interactions-readiness.test.ts | 22 +++++++++++++++++++ src/__tests__/client.test.ts | 17 -------------- 2 files changed, 22 insertions(+), 17 deletions(-) create mode 100644 src/__tests__/client-interactions-readiness.test.ts diff --git a/src/__tests__/client-interactions-readiness.test.ts b/src/__tests__/client-interactions-readiness.test.ts new file mode 100644 index 0000000000..6fe45fec78 --- /dev/null +++ b/src/__tests__/client-interactions-readiness.test.ts @@ -0,0 +1,22 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { createAgentDeviceClient } from '../agent-device-client.ts'; +import { createTransport } from './client-transport-fixture.ts'; + +// #1656 follow-up: the operator-only readiness budget rides the same common-input seam as every +// other option (`commonToClientOptions`), so it must reach `req.flags` for press/click/longpress +// exactly like `button`/`durationMs` do (see client.test.ts). Never CLI- or model-writable — this +// is the SDK-only route the option exists for. Split out of client.test.ts, which is already over +// the test-file-size-ratchet tripwire and may not grow (docs/agents/testing.md). +test('readinessTimeoutMs on press/click/longpress reaches the request flags', async () => { + const setup = createTransport(async () => ({ ok: true, data: {} })); + const client = createAgentDeviceClient(setup.config, { transport: setup.transport }); + + await client.interactions.press({ selector: 'label=Foo', readinessTimeoutMs: 2_000 }); + await client.interactions.click({ selector: 'label=Foo', readinessTimeoutMs: 1_500 }); + await client.interactions.longPress({ selector: 'label=Foo', readinessTimeoutMs: 900 }); + + assert.equal(setup.calls[0]?.flags?.readinessTimeoutMs, 2_000); + assert.equal(setup.calls[1]?.flags?.readinessTimeoutMs, 1_500); + assert.equal(setup.calls[2]?.flags?.readinessTimeoutMs, 900); +}); diff --git a/src/__tests__/client.test.ts b/src/__tests__/client.test.ts index 97864da213..7ae6d47b05 100644 --- a/src/__tests__/client.test.ts +++ b/src/__tests__/client.test.ts @@ -1454,23 +1454,6 @@ test('interactions expose targetKind-discriminated public response data', async ); }); -// #1656 follow-up: the operator-only readiness budget rides the same common-input seam as every -// other option (`commonToClientOptions`), so it must reach `req.flags` for press/click/longpress -// exactly like `button`/`durationMs` do above. Never CLI- or model-writable — this is the SDK-only -// route the option exists for. -test('readinessTimeoutMs on press/click/longpress reaches the request flags', async () => { - const setup = createTransport(async () => ({ ok: true, data: {} })); - const client = createAgentDeviceClient(setup.config, { transport: setup.transport }); - - await client.interactions.press({ selector: 'label=Foo', readinessTimeoutMs: 2_000 }); - await client.interactions.click({ selector: 'label=Foo', readinessTimeoutMs: 1_500 }); - await client.interactions.longPress({ selector: 'label=Foo', readinessTimeoutMs: 900 }); - - assert.equal(setup.calls[0]?.flags?.readinessTimeoutMs, 2_000); - assert.equal(setup.calls[1]?.flags?.readinessTimeoutMs, 1_500); - assert.equal(setup.calls[2]?.flags?.readinessTimeoutMs, 900); -}); - // #1625: `find … list` through the typed client — `matches` is part of the // public response type and rides with the issuing generation, so a caller can // pin any listed ref for its next command. From 5fc066997ceea2a7aeadc9f1a909619cf4529513 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 10:51:57 +0200 Subject: [PATCH 04/23] feat(interaction): end a readiness wait on a sparse capture; widen the request envelope by the readiness budget --- .../command-registry/src/timeout-policy.ts | 22 ++++++++-- packages/selectors/src/interaction-error.ts | 6 +++ .../command-descriptor-timeout-policy.test.ts | 21 ++++++++++ .../interaction/runtime/selector-readiness.ts | 36 +++++++++++++++- .../press-target-readiness.test.ts | 41 +++++++++++++++++++ 5 files changed, 120 insertions(+), 6 deletions(-) diff --git a/packages/command-registry/src/timeout-policy.ts b/packages/command-registry/src/timeout-policy.ts index 30e6f58f61..3f09b54d44 100644 --- a/packages/command-registry/src/timeout-policy.ts +++ b/packages/command-registry/src/timeout-policy.ts @@ -64,7 +64,12 @@ type BoundedTimeoutPolicy = CommandTimeoutPolicy & { envelopeMs: number }; type FlagTimeoutBudget = Extract; type RequestTimeoutInput = Readonly<{ positionals?: string[]; - flags?: Readonly<{ timeoutMs?: number; settle?: boolean; waitMs?: number }>; + flags?: Readonly<{ + timeoutMs?: number; + settle?: boolean; + waitMs?: number; + readinessTimeoutMs?: number; + }>; }>; /** Resolves the request envelope from its declared policy and user-supplied budget. */ @@ -74,11 +79,20 @@ export function resolveCommandRequestTimeoutMs( ): number | undefined { if (policy.envelopeMs === 'unbounded') return undefined; const boundedPolicy: BoundedTimeoutPolicy = { ...policy, envelopeMs: policy.envelopeMs }; - return ( + const envelopeMs = resolvePositionalBudgetTimeoutMs(boundedPolicy, input.positionals ?? []) ?? resolveFlagBudgetTimeoutMs(boundedPolicy, input.flags) ?? - boundedPolicy.envelopeMs - ); + boundedPolicy.envelopeMs; + return envelopeMs + readinessBudgetMs(input.flags); +} + +/** + * A readiness budget is target-poll time spent before the action itself, so it extends whatever + * envelope the command otherwise has; the daemon's own poll must never outlive the client's clock. + */ +function readinessBudgetMs(flags: RequestTimeoutInput['flags']): number { + const budgetMs = flags?.readinessTimeoutMs; + return typeof budgetMs === 'number' && Number.isFinite(budgetMs) && budgetMs > 0 ? budgetMs : 0; } function resolvePositionalBudgetTimeoutMs( diff --git a/packages/selectors/src/interaction-error.ts b/packages/selectors/src/interaction-error.ts index 35970844c0..593173ebaf 100644 --- a/packages/selectors/src/interaction-error.ts +++ b/packages/selectors/src/interaction-error.ts @@ -12,4 +12,10 @@ export const INTERACTION_ERROR_REASONS = { refUnlabeled: 'ref_unlabeled', /** The target names a node with no usable centre to touch: a missing, non-finite, or negative rect. */ targetBoundsInvalid: 'target_bounds_invalid', + /** + * A readiness poll captured a tree the snapshot-quality verdict calls sparse and the selector did + * not resolve in it: the tree is untrustworthy, so absence proves nothing and the wait ends now. + * `details.snapshotQuality` carries the verdict. + */ + captureSparse: 'capture_sparse', } as const; diff --git a/src/__tests__/command-descriptor-timeout-policy.test.ts b/src/__tests__/command-descriptor-timeout-policy.test.ts index 03f49a5b83..3bbb2b4498 100644 --- a/src/__tests__/command-descriptor-timeout-policy.test.ts +++ b/src/__tests__/command-descriptor-timeout-policy.test.ts @@ -403,3 +403,24 @@ test('open and prepare startup budgets keep a client-envelope margin over the da 90_000, ); }); + +test('a readiness budget widens the request envelope on top of the settle envelope', () => { + const press = resolveCommandTimeoutPolicy('press'); + const withoutReadiness = resolveCommandRequestTimeoutMs(press, { flags: {} }); + assert.equal(withoutReadiness, 90_000); + assert.equal( + resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 2_000 } }), + 92_000, + ); + assert.equal( + resolveCommandRequestTimeoutMs(press, { flags: { settle: true, readinessTimeoutMs: 2_000 } }), + 90_000 + 10_000 + 30_000 + 2_000, + ); + assert.equal( + resolveCommandRequestTimeoutMs(resolveCommandTimeoutPolicy('longpress'), { + flags: { readinessTimeoutMs: 2_000 }, + }), + 212_000, + ); + assert.equal(resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 0 } }), 90_000); +}); diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts index 338d4455eb..3f3d235f82 100644 --- a/src/commands/interaction/runtime/selector-readiness.ts +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -10,6 +10,7 @@ import { resolveSelectorPipeline } from '@agent-device/selectors/selector-pipeli import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; import { observeUntil } from '@agent-device/capture-kit/observe-until'; +import { isSparseSnapshotQualityVerdict } from '@agent-device/capture-kit/snapshot-quality-verdict'; import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; import { @@ -44,7 +45,7 @@ export type ResolvedSelectorAttempt = { export type SelectorReadinessDetails = { polls: number; waitedMs: number; - end: 'expired' | 'stalled'; + end: 'expired' | 'stalled' | 'sparse'; }; /** @@ -185,6 +186,18 @@ async function pollSelectorReadinessOnce( ); if (previousPoll) inheritPostGestureOutcome(previousPoll.snapshot, attempt.capture.snapshot); if (attempt.resolved?.node.rect) return attempt; + const quality = attempt.capture.snapshot.snapshotQuality; + if (isSparseSnapshotQualityVerdict(quality)) { + throw new AppError( + 'COMMAND_FAILED', + `Selector ${selectorExpression} was not found in a sparse capture; the tree cannot prove it absent`, + { + reason: INTERACTION_ERROR_REASONS.captureSparse, + snapshotQuality: quality, + hint: 'Re-run after the screen settles, or capture a snapshot to inspect the tree.', + }, + ); + } const covered = await detectCoveredSelectorTarget({ runtime, nodes: attempt.capture.snapshot.nodes, @@ -286,6 +299,25 @@ export async function pollForSelectorReadiness( phase: 'interaction_target_readiness', }); if (observed.kind === 'done') return observed.result; - if (observed.kind === 'failed') throw observed.error; + if (observed.kind === 'failed') throw withSparseReadiness(observed.error, observed); throw await readinessExhaustedFailure(runtime, selectorExpression, params, observed); } + +function withSparseReadiness( + error: unknown, + observed: { polls: readonly unknown[]; waitedMs: number }, +): unknown { + if ( + !(error instanceof AppError) || + error.details?.reason !== INTERACTION_ERROR_REASONS.captureSparse + ) { + return error; + } + const readiness: SelectorReadinessDetails = { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: 'sparse', + }; + error.details = { ...error.details, readiness }; + return error; +} diff --git a/test/integration/provider-scenarios/press-target-readiness.test.ts b/test/integration/provider-scenarios/press-target-readiness.test.ts index ea6ecc382b..dca8e84c4e 100644 --- a/test/integration/provider-scenarios/press-target-readiness.test.ts +++ b/test/integration/provider-scenarios/press-target-readiness.test.ts @@ -201,3 +201,44 @@ test('press without a readinessTimeoutMs flag fails on the first capture attempt }, ); }); + +test('press ends the wait at once on a sparse capture with capture_sparse, and never taps', async () => { + await withPressReadinessDaemon( + [ + { + command: 'ios.runner.snapshot', + deviceId: DEVICE_ID, + platform: 'apple', + repeat: true, + result: { + nodes: APPLICATION_ONLY_NODES, + truncated: false, + snapshotQuality: { + state: 'sparse', + backend: 'tree', + reasonCode: 'sparse-tree', + }, + }, + }, + ], + async (daemon, transcript) => { + const callsBeforePress = transcript.calls.length; + const startedAt = Date.now(); + const press = await daemon.callCommand('press', ['label=Continue'], { + readinessTimeoutMs: 2_000, + }); + const error = assertRpcError(press, 'COMMAND_FAILED', /sparse capture/); + const details = error.details as { + reason: unknown; + snapshotQuality: { state: string }; + readiness: { polls: number; end: string }; + }; + assert.equal(details.reason, 'capture_sparse'); + assert.equal(details.snapshotQuality.state, 'sparse'); + assert.equal(details.readiness.end, 'sparse'); + assert.ok(Date.now() - startedAt < 1_500, 'a sparse capture must not burn the budget'); + const commands = transcript.calls.slice(callsBeforePress).map((call) => call.command); + assert.ok(!commands.includes('ios.runner.tap'), 'no tap on a sparse capture'); + }, + ); +}); From 63ebc0493254a719e15f6f02733eec46df32bde4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 14:06:03 +0200 Subject: [PATCH 05/23] docs(interaction): document the replay readiness wait and readinessTimeoutMs --- website/docs/docs/client-api.md | 11 +++++++++++ website/docs/docs/replay-e2e.md | 15 +++++++++++++++ 2 files changed, 26 insertions(+) diff --git a/website/docs/docs/client-api.md b/website/docs/docs/client-api.md index 2bcbc00342..75da1673a8 100644 --- a/website/docs/docs/client-api.md +++ b/website/docs/docs/client-api.md @@ -288,6 +288,17 @@ await client.command.fold({ `fold` accepts either `pose` or `keyframes`. Keyframes use linear interpolation at roughly 60 updates per second; repeat an angle to hold it. Timestamps must start at zero and increase strictly, with 2–64 frames and a final timestamp no greater than 60,000ms. Angles must be finite and between 0° and 180°. The final timestamp bounds motion, excluding helper preparation and final hinge verification. A custom final angle is verified within 0.5°; interior angles must also settle. Cancellation stops the motion at its current angle. Re-snapshot afterwards, including after interrupted motion. +`press`, `click`, and `longpress` take `readinessTimeoutMs`. With it, the command waits up to that many milliseconds for a target that is not on screen yet, then taps it once. Without it, the command looks once and fails at once, which is the right choice for an agent that most often misses because the selector is wrong. Use it in scripted flows, where a step can land a render early: + +```ts +await client.interactions.press({ + selector: 'label="Continue"', + readinessTimeoutMs: 2_000, +}); +``` + +The wait is capped at 2 seconds. It covers only a target that has not appeared; a covered, off-screen, or ambiguous target fails at once. A failure after the wait carries `error.details.readiness` with `waitedMs`, `polls`, and `end` (`expired`, `stalled`, or `sparse`). `readinessTimeoutMs` is not an MCP tool argument and has no CLI flag. + Vega OS client support is currently VVD-only and covers device discovery, app open/close, `back`, `home`, and `tvRemote`. Physical Fire TV, capture, selector, install, logging, and performance methods report unsupported for Vega targets. Supported command methods: diff --git a/website/docs/docs/replay-e2e.md b/website/docs/docs/replay-e2e.md index 304ec141da..e96da2ee99 100644 --- a/website/docs/docs/replay-e2e.md +++ b/website/docs/docs/replay-e2e.md @@ -58,6 +58,15 @@ agent-device replay ~/.agent-device/sessions/e2e-2026-02-09T12-00-00-000Z.ad --s Interior `close` actions still run. The flag is intentionally unavailable to `test` because suite attempts own cleanup, and it is rejected for Maestro YAML because that runtime owns its lifecycle. +- `press`, `click`, and `longpress` steps wait up to 2 seconds for their target to appear before + they fail. A step recorded against a screen that was still loading passes on replay once the + target shows up. The wait covers only a target that is not on screen yet: a target that is + covered, off-screen, or matched by more than one element fails at once, as it does live. +- When the target never appears, the step fails with the same `selector_not_found` error as a live + command, and `error.details.readiness` says how long it waited and how many times it looked + (`waitedMs`, `polls`, `end`). If the app stops exposing an accessibility tree during the wait, + the step fails with `error.details.reason: capture_sparse` instead; take a snapshot to see + where the app is. ## Run Maestro compatibility flows @@ -384,5 +393,11 @@ Passing `--plan-digest` that no longer matches the current script — because yo - Leave the replay plan unchanged, repair app state so the reported failed step can be retried, then use its `--from`/`--plan-digest`. Resume starts at `--from`; it does not skip that step. - Replay file parse error: - Validate quoting in `.ad` lines (unclosed quotes are rejected). +- A `press` or `click` step fails with `selector_not_found`, but the element is on the screenshot: + - Check `error.details.readiness.end`. `expired` means the element was not in the accessibility + tree for the whole 2-second wait: the selector is wrong for this build, or the element is not + exposed to accessibility. `sparse` means the app showed an empty tree: the screen was + mid-transition or the app had left. Add a `wait` step for a landmark on the new screen before + the press. - Maestro compatibility flow fails on unsupported syntax: - Check [ADR 0015](https://github.com/callstack/agent-device/blob/main/docs/adr/0015-direct-maestro-engine.md). If the missing feature matters to your suite, open a focused issue with a small flow snippet. From 5b23c923aaa2260ded45b2978333484340b7deff Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 14:07:40 +0200 Subject: [PATCH 06/23] refactor(interaction): keep readiness comments to their constraints --- packages/capture-kit/src/observe-until.ts | 12 ++++++------ packages/contracts/src/client-gesture.ts | 6 ++---- packages/contracts/src/command-flags.ts | 5 ++--- .../contracts/src/interaction-guarantees.ts | 12 ++++-------- packages/contracts/src/request-envelope.ts | 6 +++--- .../session-replay-action-runtime.ts | 7 ++----- .../selectors/src/selector-pipeline-policy.ts | 9 ++++----- .../client-interactions-readiness.test.ts | 8 +++----- src/commands/common-input-fields.ts | 9 ++++----- src/commands/interaction/runtime/gestures.ts | 4 ++-- .../interaction/runtime/interactions.ts | 4 ++-- .../interaction/runtime/resolution.ts | 13 +++++-------- .../runtime/selector-readiness.test.ts | 9 ++++----- .../interaction/runtime/selector-readiness.ts | 19 +++++++++---------- .../session-replay-action-runtime.test.ts | 6 ++---- .../runtime-selector.contract.test.ts | 3 +-- .../press-target-readiness.test.ts | 13 +++++-------- .../scroll-movement-observation.test.ts | 4 ++-- 18 files changed, 62 insertions(+), 87 deletions(-) diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts index 5ce305518f..7ce5e2c2bb 100644 --- a/packages/capture-kit/src/observe-until.ts +++ b/packages/capture-kit/src/observe-until.ts @@ -4,10 +4,10 @@ import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; /** * The one observation loop: capture until a verdict accepts, on a schedule, under a budget that is - * enforced by cancellation. Owns exactly what every loop in this codebase used to re-implement — - * cadence, the deadline, abort-and-join of an in-flight capture, which capture errors are ridden - * out, and the per-poll timeline. Owns nothing about WHAT is observed: the verdict is the caller's, - * and so is the equality rule behind it (digest, signature, pixel). + * enforced by cancellation. Owns cadence, the deadline, abort-and-join of an in-flight capture, + * which capture errors are ridden out, and the per-poll timeline. Owns nothing about WHAT is + * observed: the verdict is the caller's, and so is the equality rule behind it (digest, signature, + * pixel). */ export type ObservationSchedule = Readonly<{ @@ -23,8 +23,8 @@ export type ObservationSchedule = Readonly<{ minPolls?: number; /** * `'start'` (default) bounds every capture, the first included. `'first-capture'` leaves the first - * capture unbounded and spends the budget only on retries, so a caller whose first attempt is the - * pre-existing one-shot behaviour pays nothing on its success path. + * capture unbounded and spends the budget only on retries, so a caller whose first attempt is a + * one-shot pays nothing on its success path. */ budgetFrom?: 'start' | 'first-capture'; }>; diff --git a/packages/contracts/src/client-gesture.ts b/packages/contracts/src/client-gesture.ts index bc3675723d..efa8556e64 100644 --- a/packages/contracts/src/client-gesture.ts +++ b/packages/contracts/src/client-gesture.ts @@ -36,10 +36,8 @@ export type SettleCommandOptions = { }; /** - * Operator-only readiness budget (#1656 promotedTarget row): how long a tap-shaped interaction may - * poll for a target that does not exist yet, capped at the row's maxTimeoutMs. Never model- or - * CLI-writable; omitted means the one-attempt resolution path unchanged from before this option - * existed. + * How long a tap-shaped interaction may poll for a target that does not exist yet, capped at the + * promotedTarget row's maxTimeoutMs. Never model- or CLI-writable; omitted means one attempt. */ export type ReadinessBudgetOptions = { readinessTimeoutMs?: number; diff --git a/packages/contracts/src/command-flags.ts b/packages/contracts/src/command-flags.ts index c9805621da..a5ab9ab429 100644 --- a/packages/contracts/src/command-flags.ts +++ b/packages/contracts/src/command-flags.ts @@ -36,9 +36,8 @@ export type CommandFlags = Omit & { maestro?: MaestroRuntimeFlags; postGestureStabilization?: boolean; /** - * Operator-only readiness budget for press/click/longpress (#1656 promotedTarget row), capped at - * its maxTimeoutMs. No CliFlags counterpart: never CLI- or model-writable, so it is a daemon-side - * addition here rather than an `Omit` member. + * Readiness budget for press/click/longpress, capped at the promotedTarget row's maxTimeoutMs. + * No CliFlags counterpart: never CLI- or model-writable. */ readinessTimeoutMs?: number; snapshotIncludeHiddenContentHints?: boolean; diff --git a/packages/contracts/src/interaction-guarantees.ts b/packages/contracts/src/interaction-guarantees.ts index e816c1a295..316b34bbae 100644 --- a/packages/contracts/src/interaction-guarantees.ts +++ b/packages/contracts/src/interaction-guarantees.ts @@ -175,10 +175,7 @@ const TAP_OUTCOME_NOT_OBSERVED_GAP: GuaranteeEnforcement = { // dispatch ONE fused XCTest runner request (`command: 'tap'` with a selectorKey/selectorValue) built // in packages/platform-apple/src/interactions.ts#tapElementSelector — not the separate // packages/maestro/ flow-script engine, which drives standalone `.yaml` Maestro flows and never -// participates in an ordinary daemon click. Verified by reading the dispatch call graph -// (interaction-touch-direct-ios.ts -> BoundTouchExecutor.tapElementSelector -> -// platform-apple/src/interactions.ts) before writing this cell (AGENTS.md: verify each claim -// against the native implementation). +// participates in an ordinary daemon click. const DIRECT_IOS_SINGLE_QUERY_READINESS: GuaranteeEnforcement = { kind: 'waived', reason: @@ -288,9 +285,8 @@ export const INTERACTION_DISPATCH_PATHS: Record & holdMs?: number; jitterPx?: number; /** - * Operator-only: the readiness budget a tap-shaped interaction (press/click/longpress) may - * spend polling for a target that does not exist yet, capped at the promotedTarget row's - * maxTimeoutMs. Never model- or CLI-writable; absent means the one-attempt resolution path. + * The readiness budget a tap-shaped interaction (press/click/longpress) may spend polling for a + * target that does not exist yet, capped at the promotedTarget row's maxTimeoutMs. Never model- + * or CLI-writable; absent means one attempt. */ readinessTimeoutMs?: number; pixels?: number; diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index c1f0fca6d9..23e63f64e3 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -216,11 +216,8 @@ function readResponseTiming(data: unknown): Record | undefined } /** - * A press/click/longpress step replays against a UI that may still be a render or two behind where - * it was recorded — the same render race #1656's operator-only readiness budget absorbs live. A - * replayed step carries no recorded budget of its own (there is no CLI flag for it), so every such - * step defaults to one here unless the merged flags already name one — future-proofing against a - * flag source this PR does not add, rather than a live possibility today. + * A replayed press/click/longpress step carries no readiness budget of its own; replay supplies one + * so a step recorded against a loading screen can land. */ const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; const READINESS_BUDGETED_REPLAY_COMMANDS: ReadonlySet = new Set([ diff --git a/packages/selectors/src/selector-pipeline-policy.ts b/packages/selectors/src/selector-pipeline-policy.ts index e44ac24d06..fcde9cf2fe 100644 --- a/packages/selectors/src/selector-pipeline-policy.ts +++ b/packages/selectors/src/selector-pipeline-policy.ts @@ -78,7 +78,7 @@ export type WaitPollBudget = { /** * `promotedTarget`: the row states only a ceiling and a cadence, never a default — a miss with no * caller-supplied budget takes the one-attempt path instead of polling (`resolveSelectorInteractionTarget`). - * When the caller does supply one (an operator-only `readinessTimeoutMs`, never model- or + * When the caller does supply one (`readinessTimeoutMs`, never model- or * CLI-writable), it is capped at `maxTimeoutMs` before it bounds the readiness loop. */ export type ReadinessPollBudget = { @@ -132,12 +132,11 @@ const WAIT_POLL_BUDGET: WaitPollBudget = { defaultTimeoutMs: 10_000, intervalMs: export const SELECTOR_PIPELINE_POLICIES = { /** * `click`/`press`/`longpress`: the tap lands on the actionable owner of the - * match. When the caller supplies an operator-only readiness budget, polls for the target to + * match. When the caller supplies a readiness budget, polls for the target to * appear and become resolvable (a rect), capped at this row's `maxTimeoutMs`, before refusing — * resolution only; occlusion, off-screen, and promotion still run once, against the winning - * capture, after the loop ends (#1656). A miss with no caller-supplied budget takes the - * one-attempt path instead (`resolveSelectorInteractionTarget`), unchanged from before this - * budget existed. + * capture, after the loop ends. A miss with no caller-supplied budget takes the one-attempt path + * (`resolveSelectorInteractionTarget`). */ promotedTarget: { resolution: SELECTOR_RESOLUTION_POLICIES.act, diff --git a/src/__tests__/client-interactions-readiness.test.ts b/src/__tests__/client-interactions-readiness.test.ts index 6fe45fec78..5e5c04fe72 100644 --- a/src/__tests__/client-interactions-readiness.test.ts +++ b/src/__tests__/client-interactions-readiness.test.ts @@ -3,11 +3,9 @@ import { test } from 'vitest'; import { createAgentDeviceClient } from '../agent-device-client.ts'; import { createTransport } from './client-transport-fixture.ts'; -// #1656 follow-up: the operator-only readiness budget rides the same common-input seam as every -// other option (`commonToClientOptions`), so it must reach `req.flags` for press/click/longpress -// exactly like `button`/`durationMs` do (see client.test.ts). Never CLI- or model-writable — this -// is the SDK-only route the option exists for. Split out of client.test.ts, which is already over -// the test-file-size-ratchet tripwire and may not grow (docs/agents/testing.md). +// The readiness budget rides the common-input seam (`commonToClientOptions`), so it must reach +// `req.flags` for press/click/longpress exactly like `button`/`durationMs` do. It is never CLI- or +// model-writable; the SDK client option is its route. test('readinessTimeoutMs on press/click/longpress reaches the request flags', async () => { const setup = createTransport(async () => ({ ok: true, data: {} })); const client = createAgentDeviceClient(setup.config, { transport: setup.transport }); diff --git a/src/commands/common-input-fields.ts b/src/commands/common-input-fields.ts index bf266c8827..2726471c96 100644 --- a/src/commands/common-input-fields.ts +++ b/src/commands/common-input-fields.ts @@ -32,8 +32,8 @@ export type CommonCommandInput = Pick< /** `--no-record`: common to every recordable command (see `commonInputFromFlags`). */ noRecord?: boolean; /** - * Operator-only readiness budget for a tap-shaped interaction (press/click/longpress), capped at - * the promotedTarget row's maxTimeoutMs. No `flagIn`/`flagKey`: it never rides a CLI flag or the + * Readiness budget for a tap-shaped interaction (press/click/longpress), capped at the + * promotedTarget row's maxTimeoutMs. No `flagIn`/`flagKey`: it never rides a CLI flag or the * client "selection" projection, only the CLI/Node structured-input and SDK client option seams. */ readinessTimeoutMs?: number; @@ -227,9 +227,8 @@ const COMMON_INPUT_FIELDS = { "Operator-only: how long press/click/longpress may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", }, read: (record) => readOptionalInteger(record, 'readinessTimeoutMs', { min: 1 }), - // Not accepted as a tool argument, and no CLI flag exists for it (deliberately: agents mostly - // miss on a wrong selector, where fast feedback matters more than absorbing a render race) — an - // operator supplies it directly through the CLI/Node structured input or the SDK client option. + // No CLI flag: agents mostly miss on a wrong selector, where fast feedback beats absorbing a + // render race. audience: operatorAudience({ operatorPath: 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', diff --git a/src/commands/interaction/runtime/gestures.ts b/src/commands/interaction/runtime/gestures.ts index eee9585532..bba637fcc8 100644 --- a/src/commands/interaction/runtime/gestures.ts +++ b/src/commands/interaction/runtime/gestures.ts @@ -96,8 +96,8 @@ export type LongPressCommandOptions = CommandContext & { target: InteractionTarget; durationMs?: number; /** - * Operator-only readiness budget (#1656 promotedTarget row): polls for the target to exist and - * become actionable, capped at the row's maxTimeoutMs. Absent takes the one-attempt path. + * Polls for the target to exist and become actionable, capped at the promotedTarget row's + * maxTimeoutMs. Absent takes one attempt. */ readinessTimeoutMs?: number; /** ADR 0012 step 4: replay-only post-resolution guard; see resolution.ts. */ diff --git a/src/commands/interaction/runtime/interactions.ts b/src/commands/interaction/runtime/interactions.ts index ac8656eb2f..7df256ffd6 100644 --- a/src/commands/interaction/runtime/interactions.ts +++ b/src/commands/interaction/runtime/interactions.ts @@ -42,8 +42,8 @@ export type PressCommandOptions = CommandContext & target: InteractionTarget; button?: ClickButton; /** - * Operator-only readiness budget (#1656 promotedTarget row): polls for the target to exist and - * become actionable, capped at the row's maxTimeoutMs. Absent takes the one-attempt path. + * Polls for the target to exist and become actionable, capped at the promotedTarget row's + * maxTimeoutMs. Absent takes one attempt. */ readinessTimeoutMs?: number; /** ADR 0012 step 4: replay-only post-resolution guard; see resolution.ts. */ diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index 8a7fc041fb..7b6ec25e49 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -166,11 +166,9 @@ export type ResolveInteractionTargetParams = { */ pipeline: ActingPipelinePolicy; /** - * Operator-only readiness budget (never model- or CLI-writable): how long a `promotedTarget` row - * may poll for a target that does not exist yet, before `resolveSelectorInteractionTarget` caps it - * at the row's `maxTimeoutMs`. Anything other than a positive integer — including `resolvedTarget` - * rows, which declare no poll budget at all — takes the one-attempt path unchanged from before - * this budget existed. + * How long a `promotedTarget` row may poll for a target that does not exist yet (never model- or + * CLI-writable); `resolveSelectorInteractionTarget` caps it at the row's `maxTimeoutMs`. Anything + * other than a positive integer, and every `resolvedTarget` row, takes one attempt. */ readinessTimeoutMs?: number; /** @@ -392,9 +390,8 @@ function isPositiveInteger(value: number | undefined): value is number { /** * The schedule THIS call may poll `promotedTarget` under: `undefined` when the row resolves * against one capture (`resolvedTarget`'s `poll: 'none'`), or when the caller supplied no - * `readinessTimeoutMs` — both take the one-attempt path exactly as before this budget existed - * (MCP and CLI never supply one, so neither surface changes). Otherwise the row's own cadence, and - * a budget capped at the row's `maxTimeoutMs` so an operator-supplied value can shrink the wait but + * `readinessTimeoutMs` — both take the one-attempt path. Otherwise the row's own cadence, and a + * budget capped at the row's `maxTimeoutMs`, so a caller-supplied value can shrink the wait but * never stretch past what the row allows. */ function readinessScheduleFor( diff --git a/src/commands/interaction/runtime/selector-readiness.test.ts b/src/commands/interaction/runtime/selector-readiness.test.ts index fad786fab5..9467163c39 100644 --- a/src/commands/interaction/runtime/selector-readiness.test.ts +++ b/src/commands/interaction/runtime/selector-readiness.test.ts @@ -5,9 +5,8 @@ import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fix import { selector } from './selector-read-utils.ts'; import { createFakeClock, createInteractionDevice } from './__tests__/test-utils/index.ts'; -// #1656 follow-up: promotedTarget's readiness poll now runs only when the caller supplies an -// operator-only `readinessTimeoutMs`, capped at the row's `maxTimeoutMs` (2_000). These pin the -// gating/capping decision at the runtime layer; the end-to-end poll mechanics (interactive/fresh +// promotedTarget's readiness poll runs only when the caller supplies `readinessTimeoutMs`, capped +// at the row's `maxTimeoutMs` (2_000). These pin the gating/capping decision at the runtime layer; the end-to-end poll mechanics (interactive/fresh // capture sequencing, covered-target diagnosis, ridden-out capture errors) are unit-tested at the // daemon level in test/integration/provider-scenarios/press-target-readiness.test.ts. @@ -37,8 +36,8 @@ test('runtime press without readinessTimeoutMs takes the one-attempt path and re return true; }, ); - // The pre-existing interactive-capture-then-full-capture-fallback pair (unchanged from before - // this budget existed) — not the readiness loop's repeated polling. + // The interactive-capture-then-full-capture-fallback pair, not the readiness loop's repeated + // polling. assert.equal(captures, 2); }); diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts index 3f3d235f82..bffa9a2a88 100644 --- a/src/commands/interaction/runtime/selector-readiness.ts +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -22,11 +22,10 @@ import { resolveActionSelector } from './selector-action-resolution.ts'; import type { InteractionAction, ResolveInteractionTargetParams } from './resolution.ts'; /** - * `promotedTarget`'s readiness poll (#1656 / operator-only `readinessTimeoutMs`): the one-attempt - * capture-and-resolve, the covered-target diagnosis it shares with the exhausted-budget failure, and - * the `observeUntil`-driven loop itself. Split out of `resolution.ts` (#1656 follow-up) once it grew - * past 1,300 lines; `resolution.ts#resolveSelectorInteractionTarget` is this module's one caller, - * deciding whether a call polls at all and, if so, under what capped budget. + * `promotedTarget`'s readiness poll: the one-attempt capture-and-resolve, the covered-target + * diagnosis it shares with the exhausted-budget failure, and the `observeUntil`-driven loop itself. + * `resolution.ts#resolveSelectorInteractionTarget` is this module's one caller, deciding whether a + * call polls at all and under what capped budget. */ /** One poll's outcome: the capture it read the tree from, and what it resolved (if anything). */ @@ -106,7 +105,7 @@ export async function attemptSelectorResolution( * covered node is a different failure with a different recovery, and the * acting row — rect-required, candidates rejected — cannot tell the caller * that. Both probes name a policy row, so the two contracts stay visible side - * by side instead of as two sets of engine knobs (#1630). + * by side instead of as two sets of engine knobs. * * The diagnosis row keeps covered nodes as candidates precisely so its occlusion stage can report * them: "matched but covered" is a different failure than "did not match", produced by re-probing @@ -164,7 +163,7 @@ export async function selectorInteractionFailure(params: { } /** - * One readiness poll: today's capture-and-resolve attempt, the previous poll's post-gesture outcome + * One readiness poll: a capture-and-resolve attempt, the previous poll's post-gesture outcome * carried forward, and the covered-target probe on a miss. A covered candidate is excluded by * promotedTarget's own occlusion stage before it reaches `resolved`, so it looks identical to * "not found yet"; detecting it here ends the loop on this poll instead of spending the budget on a @@ -252,10 +251,10 @@ async function readinessExhaustedFailure( * would produce it — an ambiguity throw from `resolveActionSelector`, or a capture error the loop * does not ride out, ends the loop on this same poll, and is rethrown unchanged. Occlusion, * off-screen, non-hittable, and keyboard refusals are not judged here: they run once, after this - * loop returns a rect-bearing resolution, exactly as `runInteractionPipelineStages` always has. + * loop returns a rect-bearing resolution, as `runInteractionPipelineStages` does. * - * The first poll is unbounded and reuses today's snapshot cache (`forceFresh: false`), so a caller - * whose first capture already matches pays the same one-or-two-capture cost as before. Only a poll + * The first poll is unbounded and reuses the snapshot cache (`forceFresh: false`), so a caller + * whose first capture already matches pays the one-or-two-capture cost of a single attempt. Only a poll * that follows a miss (poll 2+) forces a fresh capture, mirroring how `wait` bypasses the cache. * * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites this as the diff --git a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts index 5358cdfa5b..d2d4cc97d0 100644 --- a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts +++ b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts @@ -58,10 +58,8 @@ test.each(['', ' '])( }, ); -// #1656 follow-up: a replayed press/click/longpress step races the same render gap the operator- -// only readiness budget exists for, but a replay step has no CLI flag to carry one — so the one -// place replay builds a step's dispatched flags (`invokeResolvedReplayAction`'s -// `buildReplayActionFlags`) defaults it in for those three commands. +// A replay step has no CLI flag to carry a readiness budget, so `buildReplayActionFlags` defaults +// one in for press/click/longpress. test.each(['press', 'click', 'longpress'])( 'replay defaults readinessTimeoutMs onto a dispatched %s step', async (command) => { diff --git a/test/integration/interaction-contract/runtime-selector.contract.test.ts b/test/integration/interaction-contract/runtime-selector.contract.test.ts index fd2d103891..495de88fa2 100644 --- a/test/integration/interaction-contract/runtime-selector.contract.test.ts +++ b/test/integration/interaction-contract/runtime-selector.contract.test.ts @@ -218,8 +218,7 @@ test(scenario('targetReadiness'), async () => { const result = await device.interactions.press(selector('label=Continue'), { session: 'default', - // Operator-only readiness budget (#1656): never CLI- or model-writable, so this contract - // scenario supplies it directly, the one route it reaches a call through in this PR. + // Never CLI- or model-writable, so the scenario supplies it directly. readinessTimeoutMs: 2_000, }); diff --git a/test/integration/provider-scenarios/press-target-readiness.test.ts b/test/integration/provider-scenarios/press-target-readiness.test.ts index dca8e84c4e..a9f6b6e4f8 100644 --- a/test/integration/provider-scenarios/press-target-readiness.test.ts +++ b/test/integration/provider-scenarios/press-target-readiness.test.ts @@ -111,8 +111,7 @@ test('press waits for a selector missing on the first two captures, then taps on tapEntry(200, 322), ], async (daemon, transcript) => { - // Operator-only readiness budget: never CLI- or model-writable, so this scenario supplies it - // directly as a request flag, the one route it reaches a call through in this PR. + // Never CLI- or model-writable, so the scenario supplies it directly as a request flag. const press = await daemon.callCommand('press', ['label=Continue'], { readinessTimeoutMs: 2_000, }); @@ -176,12 +175,10 @@ test('press resolving on the first capture costs exactly one snapshot call (zero ); }); -// (d) #1656 follow-up: reviewers found the row-wide 2s wait wrong for agents, whose misses are -// usually a wrong selector — fast feedback matters more than absorbing a render race. Without an -// explicit readinessTimeoutMs (never CLI- or model-writable, so MCP and CLI never supply one), a -// miss takes the one-attempt path: exactly one capture-and-resolve attempt (the pre-existing -// interactive-then-full-capture fallback, unchanged from main — not the readiness loop's repeated -// polling), and no readiness evidence to attach. +// (d) Agent misses are usually a wrong selector, so fast feedback beats absorbing a render race. +// Without an explicit readinessTimeoutMs (never CLI- or model-writable), a miss takes the +// one-attempt path: exactly one capture-and-resolve attempt (the interactive-then-full-capture +// fallback, not the readiness loop's repeated polling), and no readiness evidence to attach. test('press without a readinessTimeoutMs flag fails on the first capture attempt, with no readiness poll', async () => { await withPressReadinessDaemon( // Interactive capture, then the interactive->full fallback — both miss, exactly like poll 1 of diff --git a/test/integration/provider-scenarios/scroll-movement-observation.test.ts b/test/integration/provider-scenarios/scroll-movement-observation.test.ts index 90c4cf203f..e56e2808a6 100644 --- a/test/integration/provider-scenarios/scroll-movement-observation.test.ts +++ b/test/integration/provider-scenarios/scroll-movement-observation.test.ts @@ -15,7 +15,7 @@ import { createProviderTranscript, type ProviderScenarioProviderEntry } from './ const APP = 'com.example.app'; const DEVICE_ID = PROVIDER_SCENARIO_IOS_SIMULATOR.id; -// #2714 directional-scroll observation, ported onto `observeUntil` (packages/capture-kit/src/observe-until.ts): +// Directional-scroll observation on `observeUntil` (packages/capture-kit/src/observe-until.ts): // `scroll down` gates its own `movement` claim on the tree it held before the gesture against one // (or more, while the surface still looks untouched) post-gesture capture. The unit-level coverage // in src/daemon/__tests__/scroll-movement.test.ts pins the verdict against a scripted capture @@ -116,7 +116,7 @@ function iosSimulatorTool(): { provider: AppleToolProvider } { args: string[], options?: ExecOptions, ): Promise => { - // #2438's system-surface presence probe runs before target resolution: report every + // The system-surface presence probe runs before target resolution: report every // registered host absent so the route proceeds to resolve the (scripted) app target below, // instead of reading the probe as 'unknown' and falling back with a fresh random identity. if (cmd === 'pgrep') return { stdout: '', stderr: '', exitCode: 1 }; From c5bd3fd7c4bb1ac74ebb29cbf92171c6e9a18f42 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 15:00:21 +0200 Subject: [PATCH 07/23] refactor(move): split interaction target resolution by domain question resolution.ts (1,131 lines) keeps the entry and the point/selector target kinds (249 lines). The rest moves, function bodies unchanged: - ref-target-resolution.ts (262): how an @ref becomes a node, and when it is refused (adopt/read, frame reconciliation, tryResolveRefNode, refMissRefusal). - resolution-disclosure.ts (179): what a resolution records and reports (EXACT_REF_RESOLUTION, buildRefResolution and its ResolvedRefNode type, selector disclosure, non-hittable hint, press recording retarget). - target-visibility-stages.ts (219): is the resolved target reachable on screen: the node-stage runner, the covered refusal (covered-interaction- error.ts folded in and deleted) and the off-screen stage. - native-ref-interaction.ts (135): the native-ref preflight and dispatch. - resolution-touch-point.ts (48): node to touch point, shared by the ref and selector kinds, so neither kind module owns the other. - replay-target-guard.ts (65): assertExpectedResolvedTarget and assertReplayTargetResolution, used by the ref and selector kinds and by selector-is/selector-read. - interaction-resolution-request.ts (73): InteractionAction, ResolveInteractionTargetParams and ExpectedResolvedTarget. The layering scan (R9/R10) refused the split while sibling modules type-imported resolution.ts: the largest type cycle grew from 6 to 8 files. The request types now sit below every module that reads them. Not a pure move: `export` added to symbols now crossing a module boundary; two comments that pointed at "above" now name the file; the stale fallow-ignore on EXACT_REF_RESOLUTION (now statically imported) and on pollForSelectorReadiness (already statically imported by resolution.ts) are removed. interaction-guarantees.ts `via` paths and three cross-package comments follow the symbols. resolution.test.ts (706) splits along the same seams: resolution.test.ts 350, ref-target-resolution.test.ts 126, resolution-disclosure.test.ts 19, target-visibility-stages.test.ts 227. Test bodies unchanged. git diff -M90% --stat: 24 files changed, 1399 insertions(+), 1312 deletions(-). No file passes the 90% rename threshold (it is a split); resolution.ts +18/-900, resolution.test.ts +1/-357. A line-multiset comparison of old against new bodies (imports and blank lines excluded) differs only in the export keywords and comments listed above. No fallow baseline entry was keyed on the moved paths. --- .../internal/target-annotation-identity.ts | 2 +- .../contracts/src/interaction-guarantees.ts | 12 +- packages/contracts/src/replay.ts | 2 +- .../selectors/src/selector-pipeline-policy.ts | 2 +- .../runtime/covered-interaction-error.ts | 28 - src/commands/interaction/runtime/gestures.ts | 4 +- .../runtime/interaction-resolution-request.ts | 73 ++ .../interaction/runtime/interaction-verb.ts | 2 +- .../interaction/runtime/interactions.ts | 9 +- .../interaction/runtime/keyboard-occlusion.ts | 2 +- .../runtime/native-ref-interaction.ts | 135 +++ .../runtime/ref-target-resolution.test.ts | 126 +++ .../runtime/ref-target-resolution.ts | 262 +++++ .../runtime/replay-target-guard.ts | 65 ++ .../runtime/resolution-disclosure.test.ts | 19 + .../runtime/resolution-disclosure.ts | 179 ++++ .../runtime/resolution-touch-point.ts | 48 + .../interaction/runtime/resolution.test.ts | 358 +------ .../interaction/runtime/resolution.ts | 918 +----------------- .../interaction/runtime/selector-is.ts | 3 +- .../interaction/runtime/selector-read.ts | 3 +- .../interaction/runtime/selector-readiness.ts | 13 +- .../runtime/target-visibility-stages.test.ts | 227 +++++ .../runtime/target-visibility-stages.ts | 219 +++++ 24 files changed, 1399 insertions(+), 1312 deletions(-) delete mode 100644 src/commands/interaction/runtime/covered-interaction-error.ts create mode 100644 src/commands/interaction/runtime/interaction-resolution-request.ts create mode 100644 src/commands/interaction/runtime/native-ref-interaction.ts create mode 100644 src/commands/interaction/runtime/ref-target-resolution.test.ts create mode 100644 src/commands/interaction/runtime/ref-target-resolution.ts create mode 100644 src/commands/interaction/runtime/replay-target-guard.ts create mode 100644 src/commands/interaction/runtime/resolution-disclosure.test.ts create mode 100644 src/commands/interaction/runtime/resolution-disclosure.ts create mode 100644 src/commands/interaction/runtime/resolution-touch-point.ts create mode 100644 src/commands/interaction/runtime/target-visibility-stages.test.ts create mode 100644 src/commands/interaction/runtime/target-visibility-stages.ts diff --git a/packages/ad-script/src/internal/target-annotation-identity.ts b/packages/ad-script/src/internal/target-annotation-identity.ts index 80ba39047d..9126e32444 100644 --- a/packages/ad-script/src/internal/target-annotation-identity.ts +++ b/packages/ad-script/src/internal/target-annotation-identity.ts @@ -52,7 +52,7 @@ type IdentityTreeNode = Pick; * (`@agent-device/selectors/target-evidence`), replay-time verification * (`packages/replay-port/src/daemon-port/session-replay-target-verification.ts`), and the * dispatch-side post-resolution guard - * (`src/commands/interaction/runtime/resolution.ts`), so all three compute + * (`src/commands/interaction/runtime/replay-target-guard.ts`), so all three compute * a node's identity with byte-identical semantics. */ export function readNodeLocalIdentity( diff --git a/packages/contracts/src/interaction-guarantees.ts b/packages/contracts/src/interaction-guarantees.ts index 316b34bbae..0410b9eae4 100644 --- a/packages/contracts/src/interaction-guarantees.ts +++ b/packages/contracts/src/interaction-guarantees.ts @@ -240,7 +240,7 @@ const RUNTIME_TREE_SHARED_GUARANTEES = { // them is undetected. offscreen: { kind: 'runtime', - via: 'src/commands/interaction/runtime/resolution.ts#throwIfOffscreenInteractionTarget', + via: 'src/commands/interaction/runtime/target-visibility-stages.ts#throwIfOffscreenInteractionTarget', }, // Promotion runs only for rows that declare it (#1656); the retarget itself // is still resolveActionableTouchResolution. @@ -316,7 +316,7 @@ export const INTERACTION_DISPATCH_PATHS: Record, + action: InteractionAction, +): Promise<{ + targetHittable?: boolean; + hint?: string; + node?: SnapshotNode; + preAction?: SurfaceScopedNodes; +}> { + const session = await runtime.sessions.get(options.session ?? 'default'); + const storedSnapshot = session?.snapshot; + const nodes = storedSnapshot?.nodes; + if (!storedSnapshot || !nodes || normalizeRef(target.ref) === null) return {}; + const outcome = tryResolveRefNode(nodes, target.ref, { + fallbackLabel: target.fallbackLabel ?? '', + }); + if (outcome.kind !== 'resolved') return {}; + const { resolved } = outcome; + // `resolvedTarget` whatever the command: its `none` promotion is what holds + // ADR 0011's "the preflight never changes which element the backend acts on". + const pipeline = SELECTOR_PIPELINE_POLICIES.resolvedTarget; + // #1542: dispatches by REF, not coordinate, so no point is re-derived for the dispatch — but + // evidence/annotation below still describes the returned (visible) node. + const { node: visibleNode } = await runInteractionPipelineStages({ + policy: pipeline, + nodes, + node: resolved.node, + action, + label: `Ref ${target.ref}`, + hooks: { + offscreen: async (node, tree) => + await assertVisibleRefTarget(runtime, options, node, tree, target.ref, { + action, + pipeline, + }), + }, + resolveTapPoint: (node) => resolveRectCenter(node.rect), + }); + return { + ...describeNonHittableTarget(visibleNode, action), + // ADR 0012 decision 3: the guard lookup above doubles as the record-time + // evidence source for the fast path, at zero extra capture cost. + node: visibleNode, + preAction: surfaceScopedNodes(storedSnapshot), + }; +} + +/** + * ADR 0011 native-ref dispatch, shared by click/fill/hover @ref: run the + * preflight guards against the stored node, hand the ref to the backend as + * its own element handle, and return the exact-ref result envelope. Callers + * decide WHEN the path applies (backend capability, no non-default options, + * no replay guard, no settle baseline); this owns only the dispatch itself so + * the three commands cannot drift on preflight or disclosure. + */ +export async function dispatchNativeRefInteraction( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, + action: InteractionAction, + dispatch: ( + context: BackendCommandContext, + refTarget: BackendRefTarget, + ) => Promise, +): Promise< + Extract & { backendResult?: Record } +> { + const preflight = await preflightNativeRefInteraction(runtime, options, target, action); + const backendResult = await dispatch(toBackendContext(runtime, options), { + kind: 'ref', + ref: target.ref, + ...(target.fallbackLabel ? { fallbackLabel: target.fallbackLabel } : {}), + }); + const formattedBackendResult = toBackendResult(backendResult); + return { + kind: 'ref', + target: { kind: 'ref', ref: target.ref }, + resolution: EXACT_REF_RESOLUTION, + ...preflight, + ...(formattedBackendResult ? { backendResult: formattedBackendResult } : {}), + }; +} diff --git a/src/commands/interaction/runtime/ref-target-resolution.test.ts b/src/commands/interaction/runtime/ref-target-resolution.test.ts new file mode 100644 index 0000000000..a8e942e409 --- /dev/null +++ b/src/commands/interaction/runtime/ref-target-resolution.test.ts @@ -0,0 +1,126 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { ref } from './selector-read-utils.ts'; +import { tryResolveRefNode } from './ref-target-resolution.ts'; +import { STALE_REF_HINT } from '@agent-device/selectors'; +import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; +import type { Point } from '@agent-device/kernel/snapshot'; +import { createInteractionDevice, selectorSnapshot } from './__tests__/test-utils/index.ts'; + +test('runtime ref interactions fail closed when the authorized ref has no usable bounds (ADR 0014)', async () => { + const staleSnapshot = makeSnapshotState([ + { + index: 0, + depth: 0, + type: 'Button', + label: 'Continue', + hittable: true, + }, + ]); + const calls: Point[] = []; + let captures = 0; + const device = createInteractionDevice(staleSnapshot, { + captureSnapshot: async () => { + captures += 1; + return { snapshot: selectorSnapshot() }; + }, + tap: async (_context, point) => { + calls.push(point); + }, + }); + + // ADR 0014: the authorized frame's @e1 has no usable rect, so it FAILS rather + // than recapturing and accepting the same index from a newer tree by + // positional coincidence. + await assert.rejects( + () => device.interactions.click(ref('@e1'), { session: 'default' }), + (error: unknown) => { + assert.match((error as Error).message, /Ref @e1 has no usable bounds/); + assert.deepEqual( + (error as { details?: Record }).details, + { reason: 'target_bounds_invalid', ref: 'e1', hint: STALE_REF_HINT }, + 'the frame lists @e1, so the refusal names the bounds, not a missing ref', + ); + return true; + }, + ); + assert.equal(captures, 0); + assert.deepEqual(calls, []); +}); + +test('runtime ref interactions refuse a ref the authorized frame does not list with ref_not_found', async () => { + const calls: Point[] = []; + let captures = 0; + const device = createInteractionDevice(selectorSnapshot(), { + captureSnapshot: async () => { + captures += 1; + return { snapshot: selectorSnapshot() }; + }, + tap: async (_context, point) => { + calls.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.click(ref('@e9'), { session: 'default' }), + (error: unknown) => { + assert.match((error as Error).message, /Ref @e9 not found/); + assert.deepEqual((error as { details?: Record }).details, { + reason: 'ref_not_found', + ref: 'e9', + hint: STALE_REF_HINT, + }); + return true; + }, + ); + assert.equal(captures, 0); + assert.deepEqual(calls, []); +}); + +test('tryResolveRefNode discloses exact for a resolved ref and label-fallback for label recovery', () => { + const nodes = selectorSnapshot().nodes; + + const exact = tryResolveRefNode(nodes, '@e1', { fallbackLabel: '' }); + assert.equal(exact.kind, 'resolved'); + if (exact.kind !== 'resolved') throw new Error('unreachable'); + assert.equal(exact.resolved.node.label, 'Continue'); + assert.deepEqual(exact.resolved.resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'exact', + }); + + const recovered = tryResolveRefNode(nodes, '@e9', { fallbackLabel: 'Continue' }); + assert.equal(recovered.kind, 'resolved'); + if (recovered.kind !== 'resolved') throw new Error('unreachable'); + assert.equal(recovered.resolved.node.label, 'Continue'); + assert.deepEqual(recovered.resolved.resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'label-fallback', + }); + + assert.deepEqual(tryResolveRefNode(nodes, '@e9', { fallbackLabel: '' }), { kind: 'missing' }); +}); + +test('tryResolveRefNode tells a listed node without a usable centre from a missing one', () => { + const unusable = makeSnapshotState([ + { index: 0, depth: 0, type: 'Button', label: 'Continue', hittable: true }, + ]).nodes; + + const byRef = tryResolveRefNode(unusable, '@e1', { fallbackLabel: '' }); + assert.equal(byRef.kind, 'unusable'); + if (byRef.kind !== 'unusable') throw new Error('unreachable'); + assert.equal(byRef.node.label, 'Continue'); + + const byLabel = tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Continue' }); + assert.equal( + byLabel.kind, + 'unusable', + 'the trailing-label recovery found the node, so it is not missing', + ); + + assert.deepEqual(tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Elsewhere' }), { + kind: 'missing', + }); +}); diff --git a/src/commands/interaction/runtime/ref-target-resolution.ts b/src/commands/interaction/runtime/ref-target-resolution.ts new file mode 100644 index 0000000000..cd4333d7fa --- /dev/null +++ b/src/commands/interaction/runtime/ref-target-resolution.ts @@ -0,0 +1,262 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { + SnapshotKeyboardBandFact, + SnapshotNode, + SnapshotState, +} from '@agent-device/kernel/snapshot'; +import { findNodeByRef, normalizeRef } from '@agent-device/kernel/snapshot'; +import { resolveRectCenter } from '@agent-device/kernel/rect-center'; +import type { + AgentDeviceRuntime, + CommandContext, + CommandSessionRecord, +} from '../../../runtime-contract.ts'; +import { STALE_REF_HINT } from '@agent-device/selectors'; +import type { InteractionSnapshot } from './interaction-snapshot-capture.ts'; +import { requireSnapshotSession } from './selector-read-shared.ts'; +import { findNodeByLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; +import { surfaceScopedNodes } from './post-action-surface.ts'; +import type { + InteractionTarget, + PreresolvedInteractionTarget, + ResolvedInteractionTarget, + SurfaceScopedNodes, +} from '@agent-device/contracts/interaction'; +import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { localIdentitiesEqual, readNodeLocalIdentity } from '@agent-device/ad-script'; +import type { ResolveInteractionTargetParams } from './interaction-resolution-request.ts'; +import { assertReplayTargetResolution } from './replay-target-guard.ts'; +import { + buildRefResolution, + describeResolvedInteractionNode, + type ResolvedRefNode, +} from './resolution-disclosure.ts'; +import { + assertVisibleRefTarget, + runInteractionPipelineStages, +} from './target-visibility-stages.ts'; +import { resolveNodeTouchPoint } from './resolution-touch-point.ts'; + +/** The node a ref target acts on, plus the tree the shared guards read it against. */ +type RefResolution = { + tree: SurfaceScopedNodes; + resolved: ResolvedRefNode; + /** The keyboard band the capture of `tree.nodes` measured, when it measured one (#2660). */ + keyboard?: SnapshotKeyboardBandFact; +}; + +/** + * #1654: adopt the node the caller already resolved instead of resolving the + * same `@ref` a second time. This replaces the LOOKUP only — every guard below + * still runs, against the caller's tree, at the symbols the ADR 0011 + * `runtime-ref` cells name. + * + * `exact` is truthful only when all three pieces of carried provenance agree: + * the positional ref, the payload ref, and the node's own ref. Fail closed if + * future internal plumbing lets them drift. + */ +function adoptPreresolvedRefTarget( + target: Extract, + preresolved: PreresolvedInteractionTarget, +): RefResolution { + const ref = normalizeRef(target.ref); + if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${target.ref}`); + const carriedRef = normalizeRef(preresolved.ref); + const nodeRef = preresolved.node.ref ? normalizeRef(preresolved.node.ref) : null; + if (carriedRef !== ref || nodeRef !== ref || !preresolved.nodes.includes(preresolved.node)) { + throw new AppError( + 'COMMAND_FAILED', + 'Internal find target provenance does not match the interaction ref', + ); + } + return { + tree: { + nodes: preresolved.nodes, + ...(preresolved.iosSystemSurfaceBundleId + ? { surfaceBundleId: preresolved.iosSystemSurfaceBundleId } + : {}), + }, + ...(preresolved.keyboard ? { keyboard: preresolved.keyboard } : {}), + resolved: buildRefResolution(ref, preresolved.node, 'exact'), + }; +} + +async function readRefResolution( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, +): Promise { + const capture = await resolveSnapshotForRef(runtime, options, target); + return { + tree: surfaceScopedNodes(capture.snapshot), + ...(capture.snapshot.keyboard ? { keyboard: capture.snapshot.keyboard } : {}), + resolved: capture.resolved, + }; +} + +export async function resolveRefInteractionTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, + params: ResolveInteractionTargetParams, +): Promise { + const { tree, keyboard, resolved } = params.preresolvedTarget + ? adoptPreresolvedRefTarget(target, params.preresolvedTarget) + : await readRefResolution(runtime, options, target); + const nodes = tree.nodes; + // #1542: point/response read from the returned (possibly rescue-patched) node. + const { node: visibleNode, tapPoint: point } = await runInteractionPipelineStages({ + policy: params.pipeline, + nodes, + ...(keyboard ? { keyboard } : {}), + node: resolved.node, + action: params.action, + label: `Ref ${target.ref}`, + hooks: { + onResolved: (node, tree) => assertReplayTargetResolution(node, tree, params), + offscreen: async (node, tree) => + await assertVisibleRefTarget(runtime, options, node, tree, target.ref, params), + }, + resolveTapPoint: (node) => + resolveNodeTouchPoint(node, nodes, { + invalidMessage: `Ref ${target.ref} has no usable bounds`, + blockedTargetLabel: `Ref ${target.ref}`, + blockedTargetDetails: { ref: `@${normalizeRef(target.ref) ?? node.ref}` }, + }), + }); + return { + kind: 'ref', + point, + target: { kind: 'ref', ref: `@${resolved.ref}` }, + ...describeResolvedInteractionNode( + runtime, + visibleNode, + tree, + params.action, + resolved.resolution, + ), + }; +} + +async function resolveSnapshotForRef( + runtime: AgentDeviceRuntime, + options: CommandContext, + target: Extract, +): Promise { + const { session, snapshot: frameTree } = await requireSnapshotSession(runtime, options.session); + + const fallbackLabel = target.fallbackLabel ?? ''; + const outcome = tryResolveRefNode(frameTree.nodes, target.ref, { + fallbackLabel, + }); + // ADR 0014: missing authorized-frame evidence FAILS. It must not fall through + // to a fresh capture and accept the same ref body from a newer tree — that is + // exactly the positional-coincidence retarget the frame model forbids. A stale + // read is observable and recoverable; a stale mutation can act on the wrong + // element. The caller re-observes (snapshot) or uses a selector. + if (outcome.kind !== 'resolved') throw refMissRefusal(outcome, target.ref); + return reconcileFreshObservation({ + session, + frameTree, + target, + fallbackLabel, + authorized: outcome.resolved, + }); +} + +/** + * ADR 0014 step 5: decouple Android freshness from ref authorization. The frame + * tree names WHICH node `@eN` authorizes. When a freshness (or other read-only) + * capture has advanced the operational observation past the frame, adopt the + * observation's node — its fresh on-screen coordinates — ONLY when its local + * identity still matches the authorized node. That covers the legitimate case of + * an element that merely moved. If the identity differs (a different element now + * sits at that index) or the ref is absent from the observation, keep the + * authorized frame node so a positional coincidence cannot retarget the action. + */ +function reconcileFreshObservation(params: { + session: CommandSessionRecord; + frameTree: SnapshotState; + target: Extract; + fallbackLabel: string; + authorized: ResolvedRefNode; +}): InteractionSnapshot & { resolved: ResolvedRefNode } { + const { session, frameTree, target, fallbackLabel, authorized } = params; + const observation = session.snapshot; + if (!observation || observation === frameTree) { + return { snapshot: frameTree, resolved: authorized }; + } + const observed = tryResolveRefNode(observation.nodes, target.ref, { fallbackLabel }); + if ( + observed.kind === 'resolved' && + localIdentitiesEqual( + readNodeLocalIdentity(authorized.node), + readNodeLocalIdentity(observed.resolved.node), + ) + ) { + return { snapshot: observation, resolved: observed.resolved }; + } + return { snapshot: frameTree, resolved: authorized }; +} + +/** The runtime-ref resolver: `exact` for a resolved `@ref`, `label-fallback` for trailing-label recovery. */ +/** + * What one tree makes of a ref: the node it authorizes (exact, or the trailing-label recovery), a + * node it lists (by ref or by that label) that has no usable centre, or no node at all. The two + * misses are distinct outcomes so a caller can name a stale ref and an unactionable target apart. + */ +export type RefResolutionOutcome = + | { kind: 'resolved'; resolved: ResolvedRefNode } + | { kind: 'unusable'; node: SnapshotNode } + | { kind: 'missing' }; + +export function tryResolveRefNode( + nodes: SnapshotState['nodes'], + refInput: string, + options: { + fallbackLabel: string; + }, +): RefResolutionOutcome { + const ref = normalizeRef(refInput); + if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${refInput}`); + const refNode = findNodeByRef(nodes, ref); + if (isUsableResolvedNode(refNode)) { + return { kind: 'resolved', resolved: buildRefResolution(ref, refNode, 'exact') }; + } + const fallbackNode = + options.fallbackLabel.length > 0 ? findNodeByLabel(nodes, options.fallbackLabel) : null; + if (isUsableResolvedNode(fallbackNode)) { + return { kind: 'resolved', resolved: buildRefResolution(ref, fallbackNode, 'label-fallback') }; + } + const found = refNode ?? fallbackNode; + return found ? { kind: 'unusable', node: found } : { kind: 'missing' }; +} + +/** + * The refusal for a ref the frame could not authorize: a ref no node carries is stale or was never + * issued (`ref_not_found`); a ref whose node is listed but has no usable centre is present and + * unactionable (`target_bounds_invalid`). Both recover the same way, a fresh observation, so both + * carry the stale-ref hint; `details.ref` is the bare ref body either way. + */ +function refMissRefusal( + miss: Exclude, + refInput: string, +): AppError { + const ref = normalizeRef(refInput) ?? refInput; + return miss.kind === 'unusable' + ? new AppError('COMMAND_FAILED', `Ref ${refInput} has no usable bounds`, { + reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, + ref, + hint: STALE_REF_HINT, + }) + : new AppError('COMMAND_FAILED', `Ref ${refInput} not found`, { + reason: INTERACTION_ERROR_REASONS.refNotFound, + ref, + hint: STALE_REF_HINT, + }); +} + +function isUsableResolvedNode(node: SnapshotNode | null | undefined): node is SnapshotNode { + if (!node) return false; + return resolveRectCenter(node.rect) !== null; +} diff --git a/src/commands/interaction/runtime/replay-target-guard.ts b/src/commands/interaction/runtime/replay-target-guard.ts new file mode 100644 index 0000000000..ba4cff4ece --- /dev/null +++ b/src/commands/interaction/runtime/replay-target-guard.ts @@ -0,0 +1,65 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { SnapshotNode, SnapshotState } from '@agent-device/kernel/snapshot'; +import { + localIdentitiesEqual, + readNodeLocalIdentity, + readNodeStructuralDenotation, + structuralDenotationsEqual, +} from '@agent-device/ad-script'; +import { REPLAY_TARGET_GUARD_MISMATCH_REASON } from '@agent-device/contracts/replay'; +import type { + ExpectedResolvedTarget, + ResolveInteractionTargetParams, +} from './interaction-resolution-request.ts'; + +/** + * Compares the resolution winner (pre-promotion: hittable-ancestor promotion + * deliberately retargets to the same LEAF's actionable container and must not + * trip the guard — duplicates are distinct leaves with distinct structural + * denotations, so comparing the leaf is exactly right) against the verified + * member's local identity AND structural denotation; throws pre-action when + * EITHER differs. + */ +export function assertExpectedResolvedTarget( + node: SnapshotNode, + nodes: SnapshotState['nodes'], + expected: ExpectedResolvedTarget | undefined, + action: string, + targetRole?: 'source' | 'destination', +): void { + if (!expected) return; + const observedIdentity = readNodeLocalIdentity(node); + const observedStructural = readNodeStructuralDenotation(node, nodes); + if ( + localIdentitiesEqual(observedIdentity, expected.identity) && + structuralDenotationsEqual(observedStructural, expected.structural) + ) { + return; + } + throw new AppError( + 'COMMAND_FAILED', + `${action} resolved to a different element than replay verification isolated; the action was not sent`, + { + reason: REPLAY_TARGET_GUARD_MISMATCH_REASON, + observed: observedIdentity, + observedStructural, + expected: expected.identity, + expectedStructural: expected.structural, + ...(targetRole ? { targetRole } : {}), + }, + ); +} + +export function assertReplayTargetResolution( + node: SnapshotNode, + nodes: SnapshotState['nodes'], + params: ResolveInteractionTargetParams, +): void { + assertExpectedResolvedTarget( + node, + nodes, + params.expectedResolvedTarget, + params.action, + params.replayTargetRole, + ); +} diff --git a/src/commands/interaction/runtime/resolution-disclosure.test.ts b/src/commands/interaction/runtime/resolution-disclosure.test.ts new file mode 100644 index 0000000000..bb3daff2d7 --- /dev/null +++ b/src/commands/interaction/runtime/resolution-disclosure.test.ts @@ -0,0 +1,19 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { buildRefResolution } from './resolution-disclosure.ts'; +import { selectorSnapshot } from './__tests__/test-utils/index.ts'; + +test('buildRefResolution is the shared exact and label-fallback disclosure constructor', () => { + const node = selectorSnapshot().nodes[0]!; + + assert.deepEqual(buildRefResolution('e1', node, 'exact').resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'exact', + }); + assert.deepEqual(buildRefResolution('e1', node, 'label-fallback').resolution, { + source: 'ref', + phase: 'pre-action', + kind: 'label-fallback', + }); +}); diff --git a/src/commands/interaction/runtime/resolution-disclosure.ts b/src/commands/interaction/runtime/resolution-disclosure.ts new file mode 100644 index 0000000000..34b25c298e --- /dev/null +++ b/src/commands/interaction/runtime/resolution-disclosure.ts @@ -0,0 +1,179 @@ +import type { SnapshotNode, SnapshotState } from '@agent-device/kernel/snapshot'; +import type { AgentDeviceRuntime } from '../../../runtime-contract.ts'; +import { type SelectorResolution, buildSelectorChainForNode } from '@agent-device/selectors'; +import { resolvePressRecordingTarget } from '@agent-device/selectors/press-retarget'; +import { resolveRefLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; +import { normalizeType } from '@agent-device/contracts/snapshot'; +import { truncateUtf8 } from './truncate-utf8.ts'; +import type { + RecordingTargetOverride, + ResolutionDiagnosticEntry, + ResolutionDisclosure, + SurfaceScopedNodes, +} from '@agent-device/contracts/interaction'; +import type { InteractionAction } from './interaction-resolution-request.ts'; + +export type ResolvedRefNode = { + ref: string; + node: SnapshotNode; + resolution: ResolutionDisclosure; +}; + +// ADR 0012 decision 2 bounds: diagnostic strings and losing alternatives. +const RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP = 256; +const MAX_RESOLUTION_ALTERNATIVES = 5; + +/** + * A successful `@ref` lookup names exactly one node; label recovery discloses label-fallback instead. + * ADR 0011 registry anchor: interaction-guarantees.ts cites it as a `via` symbol. + */ +export const EXACT_REF_RESOLUTION: ResolutionDisclosure = { + source: 'ref', + phase: 'pre-action', + kind: 'exact', +}; + +const LABEL_FALLBACK_REF_RESOLUTION: ResolutionDisclosure = { + source: 'ref', + phase: 'pre-action', + kind: 'label-fallback', +}; + +/** Shared construction site for every runtime-ref resolution disclosure. */ +export function buildRefResolution( + ref: string, + node: SnapshotNode, + kind: 'exact' | 'label-fallback', +): ResolvedRefNode { + return { + ref, + node, + resolution: kind === 'exact' ? EXACT_REF_RESOLUTION : LABEL_FALLBACK_REF_RESOLUTION, + }; +} + +const UNIQUE_RUNTIME_RESOLUTION: ResolutionDisclosure = { + source: 'runtime', + phase: 'pre-action', + kind: 'unique', +}; + +// Disclosure only: the winner stays resolveSelectorChain's pick (ADR 0012). +export function buildSelectorResolutionDisclosure( + resolved: SelectorResolution, + nodes: SnapshotState['nodes'], +): ResolutionDisclosure { + if (!resolved.disambiguation) return UNIQUE_RUNTIME_RESOLUTION; + return { + source: 'runtime', + phase: 'pre-action', + kind: 'disambiguated', + matchCount: resolved.disambiguation.matchCount, + winnerDiagnostic: buildResolutionDiagnosticEntry(resolved.node, nodes), + tiebreak: resolved.disambiguation.tiebreak, + alternatives: resolved.disambiguation.alternatives + .slice(0, MAX_RESOLUTION_ALTERNATIVES) + .map((node) => buildResolutionDiagnosticEntry(node, nodes)), + }; +} + +function buildResolutionDiagnosticEntry( + node: SnapshotNode, + nodes: SnapshotState['nodes'], +): ResolutionDiagnosticEntry { + const role = normalizeType(node.type ?? ''); + const label = resolveRefLabel(node, nodes); + return { + diagnosticRef: `diag-${node.ref}`, + ...(role ? { role: truncateUtf8(role, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } : {}), + ...(label !== undefined + ? { label: truncateUtf8(label, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } + : {}), + }; +} + +// Shared tail of a resolved ref/selector interaction target: the node itself +// plus everything derived from it for the response. Every response field +// describes the DISPATCHED node — the #1280 retarget rides only on the +// `recordingTarget` side channel below. `tree` is the capture the node was +// resolved from, and becomes the pre-action baseline this publishes. +export function describeResolvedInteractionNode( + runtime: AgentDeviceRuntime, + node: SnapshotNode, + tree: SurfaceScopedNodes, + action: InteractionAction, + resolution: ResolutionDisclosure, +): { + node: SnapshotNode; + selectorChain: string[]; + refLabel: string | undefined; + targetHittable?: boolean; + hint?: string; + preAction: SurfaceScopedNodes; + resolution: ResolutionDisclosure; + recordingTarget?: RecordingTargetOverride; +} { + const nodes = tree.nodes; + return { + node, + selectorChain: buildSelectorChainForNode(node, runtime.backend.platform, { + action: action === 'fill' ? 'fill' : 'click', + nodes, + }), + refLabel: resolveRefLabel(node, nodes), + ...describeNonHittableTarget(node, action), + preAction: tree, + resolution, + ...pressRecordingTargetOverride(runtime, node, nodes, action), + }; +} + +/** + * #1280 (ADR 0012 decision 3 amendment): the recording-only side channel. + * When a click/press resolves to an identity-empty container, the RECORDED + * step retargets to its first labeled descendant — node, chain, and + * ref-label computed together here so the recorded action entry and its + * `target-v1` evidence can never half-retarget. The response payloads never + * consume this (see `interaction-touch-response.ts`). `fill` is deliberately + * excluded: its chain carries `editable=true` constraints a label descendant + * cannot satisfy, which would record an unreplayable selector. + */ +function pressRecordingTargetOverride( + runtime: AgentDeviceRuntime, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + action: InteractionAction, +): { recordingTarget?: RecordingTargetOverride } { + if (action !== 'click' && action !== 'press') return {}; + const recordingNode = resolvePressRecordingTarget(node, nodes); + if (recordingNode === node) return {}; + return { + recordingTarget: { + node: recordingNode, + selectorChain: buildSelectorChainForNode(recordingNode, runtime.backend.platform, { + action: 'click', + nodes, + }), + refLabel: resolveRefLabel(recordingNode, nodes), + }, + }; +} + +/** + * iOS AX `hittable` flags are unreliable on deep React Native trees (see #1037: + * a map-pin annotation exact-matched a longer recents row label and reported tap + * success while doing nothing visible). We deliberately do NOT fail or filter on + * this signal — that would break selectors that only ever resolve to nodes the + * platform marks non-hittable. Instead, surface it so the caller can notice a + * likely no-op tap and re-target with a ref or a more specific selector/longer text. + */ +export function describeNonHittableTarget( + node: SnapshotNode, + action: InteractionAction, +): { targetHittable?: boolean; hint?: string } { + if (node.hittable !== false) return {}; + return { + targetHittable: false, + hint: `The resolved element reports hittable: false, so this ${action} may have had no visible effect. Verify with a snapshot, or prefer a @ref or a longer/more specific selector to target the intended element.`, + }; +} diff --git a/src/commands/interaction/runtime/resolution-touch-point.ts b/src/commands/interaction/runtime/resolution-touch-point.ts new file mode 100644 index 0000000000..6e4ab7bd6f --- /dev/null +++ b/src/commands/interaction/runtime/resolution-touch-point.ts @@ -0,0 +1,48 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { Point, SnapshotNode, SnapshotState } from '@agent-device/kernel/snapshot'; +import { normalizeRef } from '@agent-device/kernel/snapshot'; +import { createSnapshotVisibility } from '@agent-device/contracts/snapshot'; +import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { resolveInteractionTouchPoint } from '@agent-device/selectors/interaction-touch-point'; + +export function resolveNodeTouchPoint( + node: SnapshotNode, + nodes: SnapshotState['nodes'], + failure: { + invalidMessage: string; + blockedTargetLabel: string; + blockedTargetDetails: { ref: string } | { selector: string }; + }, +): Point { + const visibility = createSnapshotVisibility(nodes); + const effectiveViewport = visibility.resolveEffectiveViewport(node); + const rootViewport = node.rect ? visibility.resolveViewport(node.rect) : null; + const resolution = resolveInteractionTouchPoint(nodes, node, { + bounds: [effectiveViewport, rootViewport].filter((rect) => rect !== null), + }); + if (resolution.kind === 'resolved') return resolution.point; + if (resolution.kind === 'invalid') { + throw new AppError('COMMAND_FAILED', failure.invalidMessage, { + reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, + ...bareTargetDetails(failure.blockedTargetDetails), + }); + } + throw new AppError( + 'COMMAND_FAILED', + `${failure.blockedTargetLabel} has no parent-owned touch point outside its interactive descendants`, + { + reason: 'covered_by_interactive_descendants', + ...failure.blockedTargetDetails, + competitorRefs: resolution.competitorRefs.slice(0, 5).map((ref) => `@${ref}`), + competitorCount: resolution.competitorRefs.length, + hint: 'Tap the specific interactive child you intend, or use a more specific selector. Every safely tappable region of the parent belongs to one of its child controls.', + }, + ); +} + +/** `details.ref` is the bare ref body on every reason; the blocked-target label keeps its `@`. */ +function bareTargetDetails( + details: { ref: string } | { selector: string }, +): { ref: string } | { selector: string } { + return 'ref' in details ? { ref: normalizeRef(details.ref) ?? details.ref } : details; +} diff --git a/src/commands/interaction/runtime/resolution.test.ts b/src/commands/interaction/runtime/resolution.test.ts index 70d30fe144..166ac26f96 100644 --- a/src/commands/interaction/runtime/resolution.test.ts +++ b/src/commands/interaction/runtime/resolution.test.ts @@ -2,20 +2,13 @@ import assert from 'node:assert/strict'; import { test } from 'vitest'; import type { BackendSnapshotOptions } from '../../../backend.ts'; import { ref, selector } from './selector-read-utils.ts'; -import { - buildRefResolution, - throwIfOffscreenInteractionTarget, - tryResolveRefNode, -} from './resolution.ts'; -import { resolveRecordedTarget, STALE_REF_HINT } from '@agent-device/selectors'; +import { resolveRecordedTarget } from '@agent-device/selectors'; import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; import type { Point } from '@agent-device/kernel/snapshot'; import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; import { clickRefE2, - coveredByTabBarSnapshot, createInteractionDevice, - duplicateCoveredLabelSnapshot, fillableSnapshot, iosTabBarSnapshot, mapPinAnnotationSnapshot, @@ -103,100 +96,6 @@ test('runtime selector misses carry a structured reason for retrying adapters', ); }); -test('runtime press refuses a selector that resolves to an off-screen element', async () => { - // Closed-drawer shape: the only match sits fully left of the viewport. The - // @ref path already refuses this; the selector path must not silently tap - // out-of-viewport coordinates. - const offscreenSnapshot = makeSnapshotState([ - { - index: 0, - depth: 0, - type: 'Application', - rect: { x: 0, y: 0, width: 400, height: 800 }, - hittable: true, - }, - { - index: 1, - depth: 2, - parentIndex: 0, - type: 'Button', - label: 'Explore', - rect: { x: -320, y: 240, width: 300, height: 50 }, - hittable: true, - }, - ]); - const taps: unknown[] = []; - const device = createInteractionDevice(offscreenSnapshot, { - tap: async (_context, point) => { - taps.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.press(selector('label=Explore'), { session: 'default' }), - (error: unknown) => { - assert.ok(error instanceof Error); - assert.match(error.message, /off-screen element and is not safe to press/); - const details = (error as { details?: Record }).details; - assert.equal(details?.reason, 'offscreen_selector'); - // #1366: the closed-drawer shape sits fully left of the viewport, so the - // hint names `scroll left` and steers back through the same selector. - assert.equal(details?.scrollDirection, 'left'); - assert.match(String(details?.hint), /scroll left/i); - assert.match(String(details?.hint), /selector/i); - return true; - }, - ); - assert.equal(taps.length, 0); -}); - -test('runtime press names a direction for a partial clip whose center is off-screen', async () => { - // #1366 regression: the row still OVERLAPS the viewport (top edge inside), so - // the rect-vs-viewport form yields no direction — but its tap-point center is - // below the bottom edge, which is what the visibility guard rejects. The hint - // must still name `scroll down` rather than falling back to the generic phrasing. - const partialClipSnapshot = makeSnapshotState([ - { - index: 0, - depth: 0, - type: 'Application', - rect: { x: 0, y: 0, width: 400, height: 800 }, - hittable: true, - }, - { - index: 1, - depth: 2, - parentIndex: 0, - type: 'Button', - label: 'Cash', - rect: { x: 20, y: 790, width: 200, height: 44 }, - hittable: true, - }, - ]); - const taps: unknown[] = []; - const device = createInteractionDevice(partialClipSnapshot, { - tap: async (_context, point) => { - taps.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.press(selector('label=Cash'), { session: 'default' }), - (error: unknown) => { - const details = (error as { details?: Record }).details; - assert.equal(details?.reason, 'offscreen_selector'); - assert.equal(details?.scrollDirection, 'down'); - assert.match(String(details?.hint), /scroll down/i); - // #1366 recovery must be bounded. `--until` is what bounds it now: it checks the same - // selector between passes, so the hint names one command rather than a manual step loop. - assert.match(String(details?.hint), /scroll down --until 'label=Cash'/); - assert.match(String(details?.hint), /stops on the target/i); - return true; - }, - ); - assert.equal(taps.length, 0); -}); - test('runtime click keeps distinct tab button centers when iOS reports the tab bar as hittable', async () => { const calls: Point[] = []; const device = createInteractionDevice(iosTabBarSnapshot(), { @@ -222,38 +121,6 @@ test('runtime click keeps distinct tab button centers when iOS reports the tab b assert.equal(selectorResult.node?.label, 'Settings'); }); -test('runtime click rejects refs covered by floating overlays', async () => { - const calls: Point[] = []; - const device = createInteractionDevice(coveredByTabBarSnapshot(), { - tap: async (_context, point) => { - calls.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.click(ref('@e2'), { session: 'default' }), - /Ref @e2 is covered by another visible element/, - ); - assert.deepEqual(calls, []); -}); - -test('runtime selector interactions skip covered matches when an uncovered duplicate exists', async () => { - const calls: Point[] = []; - const device = createInteractionDevice(duplicateCoveredLabelSnapshot(), { - tap: async (_context, point) => { - calls.push(point); - }, - }); - - const result = await device.interactions.click(selector('label="Save draft"'), { - session: 'default', - }); - - assert.equal(result.kind, 'selector'); - assert.equal(result.node?.ref, 'e2'); - assert.deepEqual(calls, [{ x: 86, y: 142 }]); -}); - test('runtime click keeps non-button semantic targets at their own center', async () => { const calls: Point[] = []; const device = createInteractionDevice(nonHittableCellSnapshot(), { @@ -481,226 +348,3 @@ test('runtime interactions reject unsupported macOS desktop and menubar surfaces assert.equal(pressed, true); }); - -test('runtime ref interactions fail closed when the authorized ref has no usable bounds (ADR 0014)', async () => { - const staleSnapshot = makeSnapshotState([ - { - index: 0, - depth: 0, - type: 'Button', - label: 'Continue', - hittable: true, - }, - ]); - const calls: Point[] = []; - let captures = 0; - const device = createInteractionDevice(staleSnapshot, { - captureSnapshot: async () => { - captures += 1; - return { snapshot: selectorSnapshot() }; - }, - tap: async (_context, point) => { - calls.push(point); - }, - }); - - // ADR 0014: the authorized frame's @e1 has no usable rect, so it FAILS rather - // than recapturing and accepting the same index from a newer tree by - // positional coincidence. - await assert.rejects( - () => device.interactions.click(ref('@e1'), { session: 'default' }), - (error: unknown) => { - assert.match((error as Error).message, /Ref @e1 has no usable bounds/); - assert.deepEqual( - (error as { details?: Record }).details, - { reason: 'target_bounds_invalid', ref: 'e1', hint: STALE_REF_HINT }, - 'the frame lists @e1, so the refusal names the bounds, not a missing ref', - ); - return true; - }, - ); - assert.equal(captures, 0); - assert.deepEqual(calls, []); -}); - -test('runtime ref interactions refuse a ref the authorized frame does not list with ref_not_found', async () => { - const calls: Point[] = []; - let captures = 0; - const device = createInteractionDevice(selectorSnapshot(), { - captureSnapshot: async () => { - captures += 1; - return { snapshot: selectorSnapshot() }; - }, - tap: async (_context, point) => { - calls.push(point); - }, - }); - - await assert.rejects( - () => device.interactions.click(ref('@e9'), { session: 'default' }), - (error: unknown) => { - assert.match((error as Error).message, /Ref @e9 not found/); - assert.deepEqual((error as { details?: Record }).details, { - reason: 'ref_not_found', - ref: 'e9', - hint: STALE_REF_HINT, - }); - return true; - }, - ); - assert.equal(captures, 0); - assert.deepEqual(calls, []); -}); - -test('tryResolveRefNode discloses exact for a resolved ref and label-fallback for label recovery', () => { - const nodes = selectorSnapshot().nodes; - - const exact = tryResolveRefNode(nodes, '@e1', { fallbackLabel: '' }); - assert.equal(exact.kind, 'resolved'); - if (exact.kind !== 'resolved') throw new Error('unreachable'); - assert.equal(exact.resolved.node.label, 'Continue'); - assert.deepEqual(exact.resolved.resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'exact', - }); - - const recovered = tryResolveRefNode(nodes, '@e9', { fallbackLabel: 'Continue' }); - assert.equal(recovered.kind, 'resolved'); - if (recovered.kind !== 'resolved') throw new Error('unreachable'); - assert.equal(recovered.resolved.node.label, 'Continue'); - assert.deepEqual(recovered.resolved.resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'label-fallback', - }); - - assert.deepEqual(tryResolveRefNode(nodes, '@e9', { fallbackLabel: '' }), { kind: 'missing' }); -}); - -test('tryResolveRefNode tells a listed node without a usable centre from a missing one', () => { - const unusable = makeSnapshotState([ - { index: 0, depth: 0, type: 'Button', label: 'Continue', hittable: true }, - ]).nodes; - - const byRef = tryResolveRefNode(unusable, '@e1', { fallbackLabel: '' }); - assert.equal(byRef.kind, 'unusable'); - if (byRef.kind !== 'unusable') throw new Error('unreachable'); - assert.equal(byRef.node.label, 'Continue'); - - const byLabel = tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Continue' }); - assert.equal( - byLabel.kind, - 'unusable', - 'the trailing-label recovery found the node, so it is not missing', - ); - - assert.deepEqual(tryResolveRefNode(unusable, '@e9', { fallbackLabel: 'Elsewhere' }), { - kind: 'missing', - }); -}); - -test('buildRefResolution is the shared exact and label-fallback disclosure constructor', () => { - const node = selectorSnapshot().nodes[0]!; - - assert.deepEqual(buildRefResolution('e1', node, 'exact').resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'exact', - }); - assert.deepEqual(buildRefResolution('e1', node, 'label-fallback').resolution, { - source: 'ref', - phase: 'pre-action', - kind: 'label-fallback', - }); -}); - -// #1542: throwIfOffscreenInteractionTarget is exported for ADR 0011 registry -// honesty (interaction-guarantees.ts's `offscreen` cells point their `via` -// here); this direct-import test is its real consumer, mirroring -// tryResolveRefNode above. End-to-end rescue/refuse coverage through the -// public click/press surface lives in offscreen-double-check.test.ts. -function fakeOffscreenFailure() { - return { - message: 'off-screen', - details: { reason: 'test' }, - hint: () => 'scroll toward it', - }; -} - -test('throwIfOffscreenInteractionTarget: an on-screen node passes through unchanged', async () => { - const device = createInteractionDevice(makeSnapshotState([])); - const nodes = makeSnapshotState([ - { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, - { - index: 1, - depth: 1, - parentIndex: 0, - type: 'Button', - rect: { x: 20, y: 20, width: 40, height: 40 }, - }, - ]).nodes; - - const result = await throwIfOffscreenInteractionTarget( - device, - { session: 'default' }, - nodes[1]!, - nodes, - fakeOffscreenFailure(), - ); - - assert.equal(result, nodes[1]); -}); - -test('throwIfOffscreenInteractionTarget: off-screen + backend confirms -> returns the node patched with the LIVE rect', async () => { - const device = createInteractionDevice(makeSnapshotState([]), { - confirmOffscreenTargetVisible: async () => ({ x: 30, y: 30, width: 40, height: 40 }), - }); - const nodes = makeSnapshotState([ - { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, - { - index: 1, - depth: 1, - parentIndex: 0, - type: 'Button', - rect: { x: 20, y: 2000, width: 40, height: 40 }, - }, - ]).nodes; - - const result = await throwIfOffscreenInteractionTarget( - device, - { session: 'default' }, - nodes[1]!, - nodes, - fakeOffscreenFailure(), - ); - - assert.deepEqual(result.rect, { x: 30, y: 30, width: 40, height: 40 }); - assert.equal(result.index, nodes[1]!.index); -}); - -test('throwIfOffscreenInteractionTarget: off-screen + no rescue -> throws with the supplied failure shape', async () => { - const device = createInteractionDevice(makeSnapshotState([])); - const nodes = makeSnapshotState([ - { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, - { - index: 1, - depth: 1, - parentIndex: 0, - type: 'Button', - rect: { x: 20, y: 2000, width: 40, height: 40 }, - }, - ]).nodes; - - await assert.rejects( - () => - throwIfOffscreenInteractionTarget( - device, - { session: 'default' }, - nodes[1]!, - nodes, - fakeOffscreenFailure(), - ), - /off-screen/, - ); -}); diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index 7b6ec25e49..551ede1a8c 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -1,32 +1,7 @@ import { AppError } from '@agent-device/kernel/errors'; -import type { - Point, - SnapshotKeyboardBandFact, - SnapshotNode, - SnapshotState, -} from '@agent-device/kernel/snapshot'; -import { findNodeByRef, normalizeRef } from '@agent-device/kernel/snapshot'; -import { resolveRectCenter } from '@agent-device/kernel/rect-center'; -import type { - AgentDeviceRuntime, - CommandContext, - CommandSessionRecord, -} from '../../../runtime-contract.ts'; -import { - STALE_REF_HINT, - type SelectorResolution, - buildSelectorChainForNode, -} from '@agent-device/selectors'; -import { - runNodePipelineStages, - type SelectorPipelineHooks, -} from '@agent-device/selectors/selector-pipeline'; -import { - SELECTOR_PIPELINE_POLICIES, - type ActingPipelinePolicy, - type SelectorPipelinePolicy, -} from '@agent-device/selectors/selector-pipeline-policy'; -import { resolvePressRecordingTarget } from '@agent-device/selectors/press-retarget'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import type { SelectorResolution } from '@agent-device/selectors'; +import type { ActingPipelinePolicy } from '@agent-device/selectors/selector-pipeline-policy'; import { attemptSelectorResolution, pollForSelectorReadiness, @@ -37,161 +12,34 @@ import { captureInteractionSnapshot, type InteractionSnapshot, } from './interaction-snapshot-capture.ts'; -import { buildCoveredInteractionError } from './covered-interaction-error.ts'; -import { requireSnapshotSession } from './selector-read-shared.ts'; -import { findNodeByLabel, resolveRefLabel } from '@agent-device/capture-kit/snapshot-node-lookup'; import { containsPoint } from '@agent-device/kernel/rect'; -import { createSnapshotVisibility, normalizeType } from '@agent-device/contracts/snapshot'; -import { - classifyOffscreenScrollDirection, - type OffscreenScrollDirection, -} from '@agent-device/capture-kit/mobile-snapshot-semantics'; -import { truncateUtf8 } from './truncate-utf8.ts'; +import { createSnapshotVisibility } from '@agent-device/contracts/snapshot'; import { surfaceScopedNodes } from './post-action-surface.ts'; import type { InteractionTarget, PointTarget, - PreresolvedInteractionTarget, - RecordingTargetOverride, - ResolutionDiagnosticEntry, - ResolutionDisclosure, ResolvedInteractionTarget, SurfaceScopedNodes, } from '@agent-device/contracts/interaction'; -import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; +import { describeKeyboardOccludedPointWarning } from './keyboard-occlusion.ts'; +import { assertReplayTargetResolution } from './replay-target-guard.ts'; import type { - BackendActionResult, - BackendCommandContext, - BackendRefTarget, -} from '../../../backend.ts'; -import { toBackendContext } from '../../runtime-common.ts'; -import { toBackendResult } from '../../runtime-types.ts'; -import { resolveInteractionTouchPoint } from '@agent-device/selectors/interaction-touch-point'; -import { - localIdentitiesEqual, - readNodeLocalIdentity, - readNodeStructuralDenotation, - structuralDenotationsEqual, -} from '@agent-device/ad-script'; + InteractionAction, + ResolveInteractionTargetParams, +} from './interaction-resolution-request.ts'; +import { resolveRefInteractionTarget } from './ref-target-resolution.ts'; import { - REPLAY_TARGET_GUARD_MISMATCH_REASON, - type ReplayTargetGuardDenotation, -} from '@agent-device/contracts/replay'; + buildSelectorResolutionDisclosure, + describeResolvedInteractionNode, +} from './resolution-disclosure.ts'; import { - assertTapTargetClearOfVisibleKeyboard, - describeKeyboardOccludedPointWarning, -} from './keyboard-occlusion.ts'; + assertVisibleSelectorTarget, + runInteractionPipelineStages, +} from './target-visibility-stages.ts'; +import { resolveNodeTouchPoint } from './resolution-touch-point.ts'; export type { InteractionTarget, ResolvedInteractionTarget }; -/** - * ADR 0012 migration step 4, post-resolution guard: the LOCAL identity AND - * the STRUCTURAL denotation (pre-order document index + same-parent sibling - * ordinal) of the element replay's pre-action verification isolated. Set ONLY - * by the replay step loop (via `DaemonRequest.internal.replayTargetGuard`) for - * annotated verified actions — never on live interactive commands. - * - * Local identity alone is insufficient: ADR path 6 isolates ONE member among - * several nodes that share the same `{id, role, label}` using sibling / - * region-scoped viewportOrder. If verification isolates duplicate A but - * dispatch's occlusion/visibility filtering selects duplicate B with the same - * local identity, a local-identity-only guard would pass and tap the wrong - * element. The structural denotation is the discriminator that catches that - * split BEFORE the device action. - */ -export type ExpectedResolvedTarget = ReplayTargetGuardDenotation; - -/** - * Compares the resolution winner (pre-promotion: hittable-ancestor promotion - * deliberately retargets to the same LEAF's actionable container and must not - * trip the guard — duplicates are distinct leaves with distinct structural - * denotations, so comparing the leaf is exactly right) against the verified - * member's local identity AND structural denotation; throws pre-action when - * EITHER differs. - */ -export function assertExpectedResolvedTarget( - node: SnapshotNode, - nodes: SnapshotState['nodes'], - expected: ExpectedResolvedTarget | undefined, - action: string, - targetRole?: 'source' | 'destination', -): void { - if (!expected) return; - const observedIdentity = readNodeLocalIdentity(node); - const observedStructural = readNodeStructuralDenotation(node, nodes); - if ( - localIdentitiesEqual(observedIdentity, expected.identity) && - structuralDenotationsEqual(observedStructural, expected.structural) - ) { - return; - } - throw new AppError( - 'COMMAND_FAILED', - `${action} resolved to a different element than replay verification isolated; the action was not sent`, - { - reason: REPLAY_TARGET_GUARD_MISMATCH_REASON, - observed: observedIdentity, - observedStructural, - expected: expected.identity, - expectedStructural: expected.structural, - ...(targetRole ? { targetRole } : {}), - }, - ); -} - -export type InteractionAction = - | 'click' - | 'press' - | 'fill' - | 'focus' - | 'longPress' - | 'hover' - | 'scroll' - | 'swipe' - | 'pinch' - | 'pan' - | 'drag' - | 'fling' - | 'rotate' - | 'transform'; - -export type ResolveInteractionTargetParams = { - action: InteractionAction; - requireInteractive: boolean; - /** - * The structural pipeline this action runs (#1656): occlusion, off-screen, - * and hittable-ancestor promotion are the row's decisions. `promotedTarget` - * for tap-shaped actions, `resolvedTarget` for the actions that must keep - * the element they resolved. - */ - pipeline: ActingPipelinePolicy; - /** - * How long a `promotedTarget` row may poll for a target that does not exist yet (never model- or - * CLI-writable); `resolveSelectorInteractionTarget` caps it at the row's `maxTimeoutMs`. Anything - * other than a positive integer, and every `resolvedTarget` row, takes one attempt. - */ - readinessTimeoutMs?: number; - /** - * `--verify` (#1047): also capture the pre-action node set for a `point` target - * so `changedFromBefore` evidence has a baseline. Ref/selector targets already - * capture a snapshot to resolve the target, so this is a no-op cost for them — - * their nodes are attached below regardless of this flag. For point targets, - * which normally skip capture entirely, this opts into one extra capture, only - * when the caller explicitly asked for verify evidence. Defaults to false. - */ - captureEvidenceBaseline?: boolean; - /** ADR 0012 step 4 post-resolution guard; see `ExpectedResolvedTarget`. */ - expectedResolvedTarget?: ExpectedResolvedTarget; - /** Identifies one endpoint when a multi-target replay guard refuses. */ - replayTargetRole?: 'source' | 'destination'; - /** - * #1654: the caller already resolved this `@ref` against its own capture, so - * the ref branch adopts that node instead of looking the ref up again. Ref - * targets only — a selector target has nothing pre-resolved to adopt. - */ - preresolvedTarget?: PreresolvedInteractionTarget; -}; - export async function resolveInteractionTarget( runtime: AgentDeviceRuntime, options: CommandContext & { target: InteractionTarget }, @@ -282,107 +130,6 @@ async function tryCaptureEvidenceBaseline( } } -/** The node a ref target acts on, plus the tree the shared guards read it against. */ -type RefResolution = { - tree: SurfaceScopedNodes; - resolved: ResolvedRefNode; - /** The keyboard band the capture of `tree.nodes` measured, when it measured one (#2660). */ - keyboard?: SnapshotKeyboardBandFact; -}; - -/** - * #1654: adopt the node the caller already resolved instead of resolving the - * same `@ref` a second time. This replaces the LOOKUP only — every guard below - * still runs, against the caller's tree, at the symbols the ADR 0011 - * `runtime-ref` cells name. - * - * `exact` is truthful only when all three pieces of carried provenance agree: - * the positional ref, the payload ref, and the node's own ref. Fail closed if - * future internal plumbing lets them drift. - */ -function adoptPreresolvedRefTarget( - target: Extract, - preresolved: PreresolvedInteractionTarget, -): RefResolution { - const ref = normalizeRef(target.ref); - if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${target.ref}`); - const carriedRef = normalizeRef(preresolved.ref); - const nodeRef = preresolved.node.ref ? normalizeRef(preresolved.node.ref) : null; - if (carriedRef !== ref || nodeRef !== ref || !preresolved.nodes.includes(preresolved.node)) { - throw new AppError( - 'COMMAND_FAILED', - 'Internal find target provenance does not match the interaction ref', - ); - } - return { - tree: { - nodes: preresolved.nodes, - ...(preresolved.iosSystemSurfaceBundleId - ? { surfaceBundleId: preresolved.iosSystemSurfaceBundleId } - : {}), - }, - ...(preresolved.keyboard ? { keyboard: preresolved.keyboard } : {}), - resolved: buildRefResolution(ref, preresolved.node, 'exact'), - }; -} - -async function readRefResolution( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, -): Promise { - const capture = await resolveSnapshotForRef(runtime, options, target); - return { - tree: surfaceScopedNodes(capture.snapshot), - ...(capture.snapshot.keyboard ? { keyboard: capture.snapshot.keyboard } : {}), - resolved: capture.resolved, - }; -} - -async function resolveRefInteractionTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, - params: ResolveInteractionTargetParams, -): Promise { - const { tree, keyboard, resolved } = params.preresolvedTarget - ? adoptPreresolvedRefTarget(target, params.preresolvedTarget) - : await readRefResolution(runtime, options, target); - const nodes = tree.nodes; - // #1542: point/response read from the returned (possibly rescue-patched) node. - const { node: visibleNode, tapPoint: point } = await runInteractionPipelineStages({ - policy: params.pipeline, - nodes, - ...(keyboard ? { keyboard } : {}), - node: resolved.node, - action: params.action, - label: `Ref ${target.ref}`, - hooks: { - onResolved: (node, tree) => assertReplayTargetResolution(node, tree, params), - offscreen: async (node, tree) => - await assertVisibleRefTarget(runtime, options, node, tree, target.ref, params), - }, - resolveTapPoint: (node) => - resolveNodeTouchPoint(node, nodes, { - invalidMessage: `Ref ${target.ref} has no usable bounds`, - blockedTargetLabel: `Ref ${target.ref}`, - blockedTargetDetails: { ref: `@${normalizeRef(target.ref) ?? node.ref}` }, - }), - }); - return { - kind: 'ref', - point, - target: { kind: 'ref', ref: `@${resolved.ref}` }, - ...describeResolvedInteractionNode( - runtime, - visibleNode, - tree, - params.action, - resolved.resolution, - ), - }; -} - function isPositiveInteger(value: number | undefined): value is number { return typeof value === 'number' && Number.isInteger(value) && value > 0; } @@ -442,7 +189,7 @@ async function resolveSelectorInteractionTarget( capture = ready.capture; resolved = ready.resolved; } - // #1542: see the ref-target twin above. + // #1542: see the ref-target twin in ref-target-resolution.ts. const selected = resolved; const { node: visibleNode, tapPoint: point } = await runInteractionPipelineStages({ policy: params.pipeline, @@ -477,228 +224,6 @@ async function resolveSelectorInteractionTarget( }; } -function assertReplayTargetResolution( - node: SnapshotNode, - nodes: SnapshotState['nodes'], - params: ResolveInteractionTargetParams, -): void { - assertExpectedResolvedTarget( - node, - nodes, - params.expectedResolvedTarget, - params.action, - params.replayTargetRole, - ); -} - -// ADR 0012 decision 2 bounds: diagnostic strings and losing alternatives. -const RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP = 256; -const MAX_RESOLUTION_ALTERNATIVES = 5; - -/** - * A successful `@ref` lookup names exactly one node; label recovery discloses label-fallback instead. - * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites it as a `via` - * symbol and the gate test imports it dynamically, which fallow cannot trace statically. - */ -// fallow-ignore-next-line unused-export -export const EXACT_REF_RESOLUTION: ResolutionDisclosure = { - source: 'ref', - phase: 'pre-action', - kind: 'exact', -}; - -const LABEL_FALLBACK_REF_RESOLUTION: ResolutionDisclosure = { - source: 'ref', - phase: 'pre-action', - kind: 'label-fallback', -}; - -/** Shared construction site for every runtime-ref resolution disclosure. */ -export function buildRefResolution( - ref: string, - node: SnapshotNode, - kind: 'exact' | 'label-fallback', -): ResolvedRefNode { - return { - ref, - node, - resolution: kind === 'exact' ? EXACT_REF_RESOLUTION : LABEL_FALLBACK_REF_RESOLUTION, - }; -} - -const UNIQUE_RUNTIME_RESOLUTION: ResolutionDisclosure = { - source: 'runtime', - phase: 'pre-action', - kind: 'unique', -}; - -// Disclosure only: the winner stays resolveSelectorChain's pick (ADR 0012). -function buildSelectorResolutionDisclosure( - resolved: SelectorResolution, - nodes: SnapshotState['nodes'], -): ResolutionDisclosure { - if (!resolved.disambiguation) return UNIQUE_RUNTIME_RESOLUTION; - return { - source: 'runtime', - phase: 'pre-action', - kind: 'disambiguated', - matchCount: resolved.disambiguation.matchCount, - winnerDiagnostic: buildResolutionDiagnosticEntry(resolved.node, nodes), - tiebreak: resolved.disambiguation.tiebreak, - alternatives: resolved.disambiguation.alternatives - .slice(0, MAX_RESOLUTION_ALTERNATIVES) - .map((node) => buildResolutionDiagnosticEntry(node, nodes)), - }; -} - -function buildResolutionDiagnosticEntry( - node: SnapshotNode, - nodes: SnapshotState['nodes'], -): ResolutionDiagnosticEntry { - const role = normalizeType(node.type ?? ''); - const label = resolveRefLabel(node, nodes); - return { - diagnosticRef: `diag-${node.ref}`, - ...(role ? { role: truncateUtf8(role, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } : {}), - ...(label !== undefined - ? { label: truncateUtf8(label, RESOLUTION_DIAGNOSTIC_STRING_BYTE_CAP) } - : {}), - }; -} - -// Shared tail of a resolved ref/selector interaction target: the node itself -// plus everything derived from it for the response. Every response field -// describes the DISPATCHED node — the #1280 retarget rides only on the -// `recordingTarget` side channel below. `tree` is the capture the node was -// resolved from, and becomes the pre-action baseline this publishes. -function describeResolvedInteractionNode( - runtime: AgentDeviceRuntime, - node: SnapshotNode, - tree: SurfaceScopedNodes, - action: InteractionAction, - resolution: ResolutionDisclosure, -): { - node: SnapshotNode; - selectorChain: string[]; - refLabel: string | undefined; - targetHittable?: boolean; - hint?: string; - preAction: SurfaceScopedNodes; - resolution: ResolutionDisclosure; - recordingTarget?: RecordingTargetOverride; -} { - const nodes = tree.nodes; - return { - node, - selectorChain: buildSelectorChainForNode(node, runtime.backend.platform, { - action: action === 'fill' ? 'fill' : 'click', - nodes, - }), - refLabel: resolveRefLabel(node, nodes), - ...describeNonHittableTarget(node, action), - preAction: tree, - resolution, - ...pressRecordingTargetOverride(runtime, node, nodes, action), - }; -} - -/** - * #1280 (ADR 0012 decision 3 amendment): the recording-only side channel. - * When a click/press resolves to an identity-empty container, the RECORDED - * step retargets to its first labeled descendant — node, chain, and - * ref-label computed together here so the recorded action entry and its - * `target-v1` evidence can never half-retarget. The response payloads never - * consume this (see `interaction-touch-response.ts`). `fill` is deliberately - * excluded: its chain carries `editable=true` constraints a label descendant - * cannot satisfy, which would record an unreplayable selector. - */ -function pressRecordingTargetOverride( - runtime: AgentDeviceRuntime, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - action: InteractionAction, -): { recordingTarget?: RecordingTargetOverride } { - if (action !== 'click' && action !== 'press') return {}; - const recordingNode = resolvePressRecordingTarget(node, nodes); - if (recordingNode === node) return {}; - return { - recordingTarget: { - node: recordingNode, - selectorChain: buildSelectorChainForNode(recordingNode, runtime.backend.platform, { - action: 'click', - nodes, - }), - refLabel: resolveRefLabel(recordingNode, nodes), - }, - }; -} - -/** - * iOS AX `hittable` flags are unreliable on deep React Native trees (see #1037: - * a map-pin annotation exact-matched a longer recents row label and reported tap - * success while doing nothing visible). We deliberately do NOT fail or filter on - * this signal — that would break selectors that only ever resolve to nodes the - * platform marks non-hittable. Instead, surface it so the caller can notice a - * likely no-op tap and re-target with a ref or a more specific selector/longer text. - */ -function describeNonHittableTarget( - node: SnapshotNode, - action: InteractionAction, -): { targetHittable?: boolean; hint?: string } { - if (node.hittable !== false) return {}; - return { - targetHittable: false, - hint: `The resolved element reports hittable: false, so this ${action} may have had no visible effect. Verify with a snapshot, or prefer a @ref or a longer/more specific selector to target the intended element.`, - }; -} - -/** - * Every node stage this action's row declares, plus the covered and keyboard refusals the - * interaction runtime owns. Which stages run is the row's decision; every acting path — selector, - * ref, and the native-ref preflight — enters them here, which is what keeps the native-ref fast - * path from succeeding on a target the shared rules would refuse. - * - * Each path hands in the resolver that produces the point it taps with, and taps the point that - * comes back, so the keyboard guard measures the coordinate the interaction is actually made of and no - * path derives a second one. A path whose point can fail to exist — the native-ref fast path taps by - * ref, reading the rect center the platform aims at — says so in its resolver's return type. - */ -async function runInteractionPipelineStages(params: { - policy: SelectorPipelinePolicy; - nodes: SnapshotState['nodes']; - /** The keyboard band `nodes`' capture measured, when it measured one (#2660). */ - keyboard?: SnapshotKeyboardBandFact; - node: SnapshotNode; - action: InteractionAction; - label: string; - hooks: SelectorPipelineHooks; - resolveTapPoint: (node: SnapshotNode) => TPoint; -}): Promise<{ node: SnapshotNode; tapPoint: TPoint }> { - const target = await runNodePipelineStages( - params.policy, - params.nodes, - params.node, - params.hooks, - ); - if (target.kind === 'occluded') { - throw buildCoveredInteractionError({ - label: params.label, - node: target.node, - action: params.action, - }); - } - const tapPoint = params.resolveTapPoint(target.node); - assertTapTargetClearOfVisibleKeyboard({ - nodes: params.nodes, - node: target.node, - action: params.action, - label: params.label, - ...(params.keyboard ? { keyboard: params.keyboard } : {}), - tapPoint, - }); - return { node: target.node, tapPoint }; -} - export async function assertSupportedInteractionSurface( runtime: AgentDeviceRuntime, options: CommandContext, @@ -722,410 +247,3 @@ async function resolveInteractionSurface( const session = await runtime.sessions.get(options.session ?? 'default'); return session?.metadata?.surface; } - -async function resolveSnapshotForRef( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, -): Promise { - const { session, snapshot: frameTree } = await requireSnapshotSession(runtime, options.session); - - const fallbackLabel = target.fallbackLabel ?? ''; - const outcome = tryResolveRefNode(frameTree.nodes, target.ref, { - fallbackLabel, - }); - // ADR 0014: missing authorized-frame evidence FAILS. It must not fall through - // to a fresh capture and accept the same ref body from a newer tree — that is - // exactly the positional-coincidence retarget the frame model forbids. A stale - // read is observable and recoverable; a stale mutation can act on the wrong - // element. The caller re-observes (snapshot) or uses a selector. - if (outcome.kind !== 'resolved') throw refMissRefusal(outcome, target.ref); - return reconcileFreshObservation({ - session, - frameTree, - target, - fallbackLabel, - authorized: outcome.resolved, - }); -} - -/** - * ADR 0014 step 5: decouple Android freshness from ref authorization. The frame - * tree names WHICH node `@eN` authorizes. When a freshness (or other read-only) - * capture has advanced the operational observation past the frame, adopt the - * observation's node — its fresh on-screen coordinates — ONLY when its local - * identity still matches the authorized node. That covers the legitimate case of - * an element that merely moved. If the identity differs (a different element now - * sits at that index) or the ref is absent from the observation, keep the - * authorized frame node so a positional coincidence cannot retarget the action. - */ -function reconcileFreshObservation(params: { - session: CommandSessionRecord; - frameTree: SnapshotState; - target: Extract; - fallbackLabel: string; - authorized: ResolvedRefNode; -}): InteractionSnapshot & { resolved: ResolvedRefNode } { - const { session, frameTree, target, fallbackLabel, authorized } = params; - const observation = session.snapshot; - if (!observation || observation === frameTree) { - return { snapshot: frameTree, resolved: authorized }; - } - const observed = tryResolveRefNode(observation.nodes, target.ref, { fallbackLabel }); - if ( - observed.kind === 'resolved' && - localIdentitiesEqual( - readNodeLocalIdentity(authorized.node), - readNodeLocalIdentity(observed.resolved.node), - ) - ) { - return { snapshot: observation, resolved: observed.resolved }; - } - return { snapshot: frameTree, resolved: authorized }; -} - -/** The runtime-ref resolver: `exact` for a resolved `@ref`, `label-fallback` for trailing-label recovery. */ -/** - * What one tree makes of a ref: the node it authorizes (exact, or the trailing-label recovery), a - * node it lists (by ref or by that label) that has no usable centre, or no node at all. The two - * misses are distinct outcomes so a caller can name a stale ref and an unactionable target apart. - */ -export type RefResolutionOutcome = - | { kind: 'resolved'; resolved: ResolvedRefNode } - | { kind: 'unusable'; node: SnapshotNode } - | { kind: 'missing' }; - -export function tryResolveRefNode( - nodes: SnapshotState['nodes'], - refInput: string, - options: { - fallbackLabel: string; - }, -): RefResolutionOutcome { - const ref = normalizeRef(refInput); - if (!ref) throw new AppError('INVALID_ARGS', `Invalid ref: ${refInput}`); - const refNode = findNodeByRef(nodes, ref); - if (isUsableResolvedNode(refNode)) { - return { kind: 'resolved', resolved: buildRefResolution(ref, refNode, 'exact') }; - } - const fallbackNode = - options.fallbackLabel.length > 0 ? findNodeByLabel(nodes, options.fallbackLabel) : null; - if (isUsableResolvedNode(fallbackNode)) { - return { kind: 'resolved', resolved: buildRefResolution(ref, fallbackNode, 'label-fallback') }; - } - const found = refNode ?? fallbackNode; - return found ? { kind: 'unusable', node: found } : { kind: 'missing' }; -} - -type ResolvedRefNode = { - ref: string; - node: SnapshotNode; - resolution: ResolutionDisclosure; -}; - -/** - * The refusal for a ref the frame could not authorize: a ref no node carries is stale or was never - * issued (`ref_not_found`); a ref whose node is listed but has no usable centre is present and - * unactionable (`target_bounds_invalid`). Both recover the same way, a fresh observation, so both - * carry the stale-ref hint; `details.ref` is the bare ref body either way. - */ -function refMissRefusal( - miss: Exclude, - refInput: string, -): AppError { - const ref = normalizeRef(refInput) ?? refInput; - return miss.kind === 'unusable' - ? new AppError('COMMAND_FAILED', `Ref ${refInput} has no usable bounds`, { - reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, - ref, - hint: STALE_REF_HINT, - }) - : new AppError('COMMAND_FAILED', `Ref ${refInput} not found`, { - reason: INTERACTION_ERROR_REASONS.refNotFound, - ref, - hint: STALE_REF_HINT, - }); -} - -function resolveNodeTouchPoint( - node: SnapshotNode, - nodes: SnapshotState['nodes'], - failure: { - invalidMessage: string; - blockedTargetLabel: string; - blockedTargetDetails: { ref: string } | { selector: string }; - }, -): Point { - const visibility = createSnapshotVisibility(nodes); - const effectiveViewport = visibility.resolveEffectiveViewport(node); - const rootViewport = node.rect ? visibility.resolveViewport(node.rect) : null; - const resolution = resolveInteractionTouchPoint(nodes, node, { - bounds: [effectiveViewport, rootViewport].filter((rect) => rect !== null), - }); - if (resolution.kind === 'resolved') return resolution.point; - if (resolution.kind === 'invalid') { - throw new AppError('COMMAND_FAILED', failure.invalidMessage, { - reason: INTERACTION_ERROR_REASONS.targetBoundsInvalid, - ...bareTargetDetails(failure.blockedTargetDetails), - }); - } - throw new AppError( - 'COMMAND_FAILED', - `${failure.blockedTargetLabel} has no parent-owned touch point outside its interactive descendants`, - { - reason: 'covered_by_interactive_descendants', - ...failure.blockedTargetDetails, - competitorRefs: resolution.competitorRefs.slice(0, 5).map((ref) => `@${ref}`), - competitorCount: resolution.competitorRefs.length, - hint: 'Tap the specific interactive child you intend, or use a more specific selector. Every safely tappable region of the parent belongs to one of its child controls.', - }, - ); -} - -/** `details.ref` is the bare ref body on every reason; the blocked-target label keeps its `@`. */ -function bareTargetDetails( - details: { ref: string } | { selector: string }, -): { ref: string } | { selector: string } { - return 'ref' in details ? { ref: normalizeRef(details.ref) ?? details.ref } : details; -} - -function isUsableResolvedNode(node: SnapshotNode | null | undefined): node is SnapshotNode { - if (!node) return false; - return resolveRectCenter(node.rect) !== null; -} - -/** - * The off-screen stage's refusal shape. Reached only through the pipeline - * owner, and only for rows whose off-screen stage refuses — the row's decision - * is made there, so this builds the message and never re-decides. - */ -type OffscreenStageParams = { action: InteractionAction; pipeline: SelectorPipelinePolicy }; - -// Selector parity for the @ref off-screen guard: without it, a selector -// resolving to a closed drawer/carousel item "succeeds" by tapping coordinates -// outside the viewport (observed as `Tapped (-161, 265)` against Bluesky's -// closed drawer) while the same node via @ref is refused. -async function assertVisibleSelectorTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - selector: string, - { action }: OffscreenStageParams, -): Promise { - return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { - message: `Selector ${selector} resolved to an off-screen element and is not safe to ${action}`, - details: { reason: 'offscreen_selector', selector }, - // A selector re-resolves against a fresh snapshot on every attempt, so the - // recovery is: move the named direction, then retry THIS selector — no - // separate snapshot step, and no @ref (a scroll expires the ref frame, - // #1366). `--until` is that whole loop as one command: it checks the same - // selector between passes, which is also what keeps a large step from - // overshooting, so the hint no longer has to trade distance for accuracy. - hint: (direction) => - `${scrollRevealClause(direction, selector)} then retry ${action} with the same selector. --until checks the selector between passes, so it stops on the target rather than sailing past it. If it is inside a closed drawer or another tab, open that container first.`, - }); -} - -async function assertVisibleRefTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - refInput: string, - { action }: OffscreenStageParams, -): Promise { - return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { - message: `Ref ${refInput} is off-screen and not safe to ${action}`, - details: { reason: 'offscreen_ref', ref: normalizeRef(refInput) }, - // The scroll that reveals the target expires the ref frame (#1366, ADR - // 0014), so retrying this @ref would be rejected next. Steer to a selector, - // which re-resolves against a fresh snapshot and bypasses the ref-frame guard - // — and which `--until` can then check between passes. - hint: (direction) => - `${scrollRevealClause(direction, null)} then retry ${action} with a selector (e.g. text=/id=) rather than this @ref — the scroll expires the ref frame, so re-run snapshot -i before reusing any @ref.`, - }); -} - -/** - * Shared lead-in for both off-screen hints: the one command that reveals the target. - * - * When the geometry names a direction AND the caller has a selector to check, this is a complete - * `scroll --until ` — one request that stops on the target instead of the - * scroll-then-look-again loop the hint used to prescribe. Without a selector to check (an @ref - * refusal) or without a single reveal direction (off more than one edge), it degrades to naming - * the move and leaves the stop condition to the caller's own next step. - */ -function scrollRevealClause( - direction: OffscreenScrollDirection | null, - selector: string | null, -): string { - if (!direction) return 'Scroll toward it,'; - if (!selector) return `Scroll ${direction} toward it,`; - return `Run scroll ${direction} --until '${selector}' to bring it on screen,`; -} - -/** - * ADR 0011 native-ref preflight: `click @ref` / `fill @ref` fast paths - * dispatch straight to `backend.tapTarget`/`fillTarget`, and a backend fast - * path can silently "succeed" — delegation-on-error never triggers there. The - * ref came from the stored session snapshot, so the node is already in hand: - * run the SAME shared guards the runtime path uses against it before the - * backend call — occlusion (`isSnapshotNodeInteractionBlocked` via - * `assertInteractionNotBlocked`) and offscreen (the snapshot visibility resolver via - * `assertVisibleRefTarget`) ERROR with the runtime path's exact shapes, and - * the non-hittable annotation is returned for the fast-path result. - * - * Zero extra round trips by construction on the accept path: no session, no - * stored snapshot, an unresolvable/invalid ref, or a node without a usable - * rect all make the preflight a no-op and the fast path proceeds exactly as - * before. Promotion to a hittable ancestor stays a runtime-path behavior — - * the preflight never changes which element the backend acts on. Exception: - * a would-be off-screen refusal may spend one extra iOS runner round trip - * (#1542's double-check) before erroring — cost only on the path that was - * about to fail anyway. - * - * Exported as an ADR 0011 registry anchor (interaction-guarantees.ts `via` - * symbol, imported dynamically by the gate test); production callers reach - * it through `dispatchNativeRefInteraction`. - */ -// fallow-ignore-next-line unused-export -export async function preflightNativeRefInteraction( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, - action: InteractionAction, -): Promise<{ - targetHittable?: boolean; - hint?: string; - node?: SnapshotNode; - preAction?: SurfaceScopedNodes; -}> { - const session = await runtime.sessions.get(options.session ?? 'default'); - const storedSnapshot = session?.snapshot; - const nodes = storedSnapshot?.nodes; - if (!storedSnapshot || !nodes || normalizeRef(target.ref) === null) return {}; - const outcome = tryResolveRefNode(nodes, target.ref, { - fallbackLabel: target.fallbackLabel ?? '', - }); - if (outcome.kind !== 'resolved') return {}; - const { resolved } = outcome; - // `resolvedTarget` whatever the command: its `none` promotion is what holds - // ADR 0011's "the preflight never changes which element the backend acts on". - const pipeline = SELECTOR_PIPELINE_POLICIES.resolvedTarget; - // #1542: dispatches by REF, not coordinate, so no point is re-derived for the dispatch — but - // evidence/annotation below still describes the returned (visible) node. - const { node: visibleNode } = await runInteractionPipelineStages({ - policy: pipeline, - nodes, - node: resolved.node, - action, - label: `Ref ${target.ref}`, - hooks: { - offscreen: async (node, tree) => - await assertVisibleRefTarget(runtime, options, node, tree, target.ref, { - action, - pipeline, - }), - }, - resolveTapPoint: (node) => resolveRectCenter(node.rect), - }); - return { - ...describeNonHittableTarget(visibleNode, action), - // ADR 0012 decision 3: the guard lookup above doubles as the record-time - // evidence source for the fast path, at zero extra capture cost. - node: visibleNode, - preAction: surfaceScopedNodes(storedSnapshot), - }; -} - -/** - * ADR 0011 native-ref dispatch, shared by click/fill/hover @ref: run the - * preflight guards against the stored node, hand the ref to the backend as - * its own element handle, and return the exact-ref result envelope. Callers - * decide WHEN the path applies (backend capability, no non-default options, - * no replay guard, no settle baseline); this owns only the dispatch itself so - * the three commands cannot drift on preflight or disclosure. - */ -export async function dispatchNativeRefInteraction( - runtime: AgentDeviceRuntime, - options: CommandContext, - target: Extract, - action: InteractionAction, - dispatch: ( - context: BackendCommandContext, - refTarget: BackendRefTarget, - ) => Promise, -): Promise< - Extract & { backendResult?: Record } -> { - const preflight = await preflightNativeRefInteraction(runtime, options, target, action); - const backendResult = await dispatch(toBackendContext(runtime, options), { - kind: 'ref', - ref: target.ref, - ...(target.fallbackLabel ? { fallbackLabel: target.fallbackLabel } : {}), - }); - const formattedBackendResult = toBackendResult(backendResult); - return { - kind: 'ref', - target: { kind: 'ref', ref: target.ref }, - resolution: EXACT_REF_RESOLUTION, - ...preflight, - ...(formattedBackendResult ? { backendResult: formattedBackendResult } : {}), - }; -} - -// Full on-screen visibility (not only the effective-viewport form): items inside an -// off-screen scrollable container (closed drawer) must also count as -// off-screen, not just items scrolled out of an on-screen container. -// -// #1542: once the bulk tree says off-screen, the guard gives iOS one chance -// to rescue a FALSE refusal via the optional backend.confirmOffscreenTargetVisible -// hook — a stale/corrupted bulk tree can say off-screen while the app is -// visually fine (zero cost on the accept path; runs only here). A confirmed -// rescue returns the node PATCHED WITH THE LIVE RECT: the caller must act on -// that returned node, never the original, because in the frozen-bulk-tree -// manifestation the original rect can be stale even when the rescue verdict -// is correct — tapping it would silently land at the wrong coordinate. The -// hook fails closed (null) on anything short of a positive confirmation, so -// a genuine refusal, or any backend without the hook, is unchanged. -// -// Exported (not just for callers here) for ADR 0011 registry honesty: -// interaction-guarantees.ts's `offscreen` cells point their `via` at this -// function, not at the contracts predicate alone, since this is the actual -// end-to-end enforcement point. -export async function throwIfOffscreenInteractionTarget( - runtime: AgentDeviceRuntime, - options: CommandContext, - node: SnapshotNode, - nodes: SnapshotState['nodes'], - failure: { - message: string; - details: Record; - hint: (direction: OffscreenScrollDirection | null) => string; - }, -): Promise { - const visibility = createSnapshotVisibility(nodes); - const viewport = node.rect ? visibility.resolveEffectiveViewport(node) : null; - if (!node.rect || !viewport || visibility.isVisibleOnScreen(node)) return node; - const rootViewport = visibility.resolveViewport(node.rect); - const liveRect = await runtime.backend.confirmOffscreenTargetVisible?.( - toBackendContext(runtime, options), - node, - rootViewport, - ); - if (liveRect) return { ...node, rect: liveRect }; - // The direction that scrolls this off-screen target into view. Named in the - // hint (and surfaced as a machine-readable detail) so the recovery is a single - // deterministic move instead of a guess (#1366). Derived from the same - // boundary the rejection above used, so partial clips and off-screen - // containers get a direction too, not just fully-scrolled-out items. - const scrollDirection = classifyOffscreenScrollDirection(node, visibility); - throw new AppError('COMMAND_FAILED', failure.message, { - ...failure.details, - rect: node.rect, - viewport, - ...(scrollDirection ? { scrollDirection } : {}), - hint: failure.hint(scrollDirection), - }); -} diff --git a/src/commands/interaction/runtime/selector-is.ts b/src/commands/interaction/runtime/selector-is.ts index 498e83acbb..03fb1eaece 100644 --- a/src/commands/interaction/runtime/selector-is.ts +++ b/src/commands/interaction/runtime/selector-is.ts @@ -16,7 +16,8 @@ import { AppError, isRequestCanceledError } from '@agent-device/kernel/errors'; import type { SelectorTarget } from '@agent-device/contracts/interaction'; import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; import type { RuntimeCommand } from '../../runtime-types.ts'; -import { assertExpectedResolvedTarget, type ExpectedResolvedTarget } from './resolution.ts'; +import { assertExpectedResolvedTarget } from './replay-target-guard.ts'; +import type { ExpectedResolvedTarget } from './interaction-resolution-request.ts'; import { type CapturedSnapshot, type SelectorSnapshotOptions, diff --git a/src/commands/interaction/runtime/selector-read.ts b/src/commands/interaction/runtime/selector-read.ts index f69d8c956c..65d5df9579 100644 --- a/src/commands/interaction/runtime/selector-read.ts +++ b/src/commands/interaction/runtime/selector-read.ts @@ -24,7 +24,8 @@ import type { ResolvedTarget, } from '@agent-device/contracts/interaction'; import type { RuntimeCommand } from '../../runtime-types.ts'; -import { assertExpectedResolvedTarget, type ExpectedResolvedTarget } from './resolution.ts'; +import { assertExpectedResolvedTarget } from './replay-target-guard.ts'; +import type { ExpectedResolvedTarget } from './interaction-resolution-request.ts'; import { type CapturedSnapshot, type SelectorSnapshotOptions, diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts index bffa9a2a88..6b258020d2 100644 --- a/src/commands/interaction/runtime/selector-readiness.ts +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -17,9 +17,12 @@ import { captureInteractionSnapshot, type InteractionSnapshot, } from './interaction-snapshot-capture.ts'; -import { buildCoveredInteractionError } from './covered-interaction-error.ts'; +import { buildCoveredInteractionError } from './target-visibility-stages.ts'; import { resolveActionSelector } from './selector-action-resolution.ts'; -import type { InteractionAction, ResolveInteractionTargetParams } from './resolution.ts'; +import type { + InteractionAction, + ResolveInteractionTargetParams, +} from './interaction-resolution-request.ts'; /** * `promotedTarget`'s readiness poll: the one-attempt capture-and-resolve, the covered-target @@ -257,11 +260,9 @@ async function readinessExhaustedFailure( * whose first capture already matches pays the one-or-two-capture cost of a single attempt. Only a poll * that follows a miss (poll 2+) forces a fresh capture, mirroring how `wait` bypasses the cache. * - * Exported as an ADR 0011 registry anchor: interaction-guarantees.ts cites this as the - * runtime-selector `targetReadiness` `via` symbol, and the gate test imports it dynamically, which - * fallow cannot trace statically. + * ADR 0011 registry anchor: interaction-guarantees.ts cites this as the runtime-selector + * `targetReadiness` `via` symbol. */ -// fallow-ignore-next-line unused-export export async function pollForSelectorReadiness( runtime: AgentDeviceRuntime, options: CommandContext, diff --git a/src/commands/interaction/runtime/target-visibility-stages.test.ts b/src/commands/interaction/runtime/target-visibility-stages.test.ts new file mode 100644 index 0000000000..7b689f0334 --- /dev/null +++ b/src/commands/interaction/runtime/target-visibility-stages.test.ts @@ -0,0 +1,227 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { ref, selector } from './selector-read-utils.ts'; +import { throwIfOffscreenInteractionTarget } from './target-visibility-stages.ts'; +import { makeSnapshotState } from '@agent-device/selectors/snapshot-geometry-fixtures'; +import type { Point } from '@agent-device/kernel/snapshot'; +import { + coveredByTabBarSnapshot, + createInteractionDevice, + duplicateCoveredLabelSnapshot, +} from './__tests__/test-utils/index.ts'; + +test('runtime press refuses a selector that resolves to an off-screen element', async () => { + // Closed-drawer shape: the only match sits fully left of the viewport. The + // @ref path already refuses this; the selector path must not silently tap + // out-of-viewport coordinates. + const offscreenSnapshot = makeSnapshotState([ + { + index: 0, + depth: 0, + type: 'Application', + rect: { x: 0, y: 0, width: 400, height: 800 }, + hittable: true, + }, + { + index: 1, + depth: 2, + parentIndex: 0, + type: 'Button', + label: 'Explore', + rect: { x: -320, y: 240, width: 300, height: 50 }, + hittable: true, + }, + ]); + const taps: unknown[] = []; + const device = createInteractionDevice(offscreenSnapshot, { + tap: async (_context, point) => { + taps.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.press(selector('label=Explore'), { session: 'default' }), + (error: unknown) => { + assert.ok(error instanceof Error); + assert.match(error.message, /off-screen element and is not safe to press/); + const details = (error as { details?: Record }).details; + assert.equal(details?.reason, 'offscreen_selector'); + // #1366: the closed-drawer shape sits fully left of the viewport, so the + // hint names `scroll left` and steers back through the same selector. + assert.equal(details?.scrollDirection, 'left'); + assert.match(String(details?.hint), /scroll left/i); + assert.match(String(details?.hint), /selector/i); + return true; + }, + ); + assert.equal(taps.length, 0); +}); + +test('runtime press names a direction for a partial clip whose center is off-screen', async () => { + // #1366 regression: the row still OVERLAPS the viewport (top edge inside), so + // the rect-vs-viewport form yields no direction — but its tap-point center is + // below the bottom edge, which is what the visibility guard rejects. The hint + // must still name `scroll down` rather than falling back to the generic phrasing. + const partialClipSnapshot = makeSnapshotState([ + { + index: 0, + depth: 0, + type: 'Application', + rect: { x: 0, y: 0, width: 400, height: 800 }, + hittable: true, + }, + { + index: 1, + depth: 2, + parentIndex: 0, + type: 'Button', + label: 'Cash', + rect: { x: 20, y: 790, width: 200, height: 44 }, + hittable: true, + }, + ]); + const taps: unknown[] = []; + const device = createInteractionDevice(partialClipSnapshot, { + tap: async (_context, point) => { + taps.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.press(selector('label=Cash'), { session: 'default' }), + (error: unknown) => { + const details = (error as { details?: Record }).details; + assert.equal(details?.reason, 'offscreen_selector'); + assert.equal(details?.scrollDirection, 'down'); + assert.match(String(details?.hint), /scroll down/i); + // #1366 recovery must be bounded. `--until` is what bounds it now: it checks the same + // selector between passes, so the hint names one command rather than a manual step loop. + assert.match(String(details?.hint), /scroll down --until 'label=Cash'/); + assert.match(String(details?.hint), /stops on the target/i); + return true; + }, + ); + assert.equal(taps.length, 0); +}); + +test('runtime click rejects refs covered by floating overlays', async () => { + const calls: Point[] = []; + const device = createInteractionDevice(coveredByTabBarSnapshot(), { + tap: async (_context, point) => { + calls.push(point); + }, + }); + + await assert.rejects( + () => device.interactions.click(ref('@e2'), { session: 'default' }), + /Ref @e2 is covered by another visible element/, + ); + assert.deepEqual(calls, []); +}); + +test('runtime selector interactions skip covered matches when an uncovered duplicate exists', async () => { + const calls: Point[] = []; + const device = createInteractionDevice(duplicateCoveredLabelSnapshot(), { + tap: async (_context, point) => { + calls.push(point); + }, + }); + + const result = await device.interactions.click(selector('label="Save draft"'), { + session: 'default', + }); + + assert.equal(result.kind, 'selector'); + assert.equal(result.node?.ref, 'e2'); + assert.deepEqual(calls, [{ x: 86, y: 142 }]); +}); + +// #1542: throwIfOffscreenInteractionTarget is exported for ADR 0011 registry +// honesty (interaction-guarantees.ts's `offscreen` cells point their `via` +// here); this direct-import test is its real consumer, mirroring +// tryResolveRefNode in ref-target-resolution.test.ts. End-to-end rescue/refuse coverage through the +// public click/press surface lives in offscreen-double-check.test.ts. +function fakeOffscreenFailure() { + return { + message: 'off-screen', + details: { reason: 'test' }, + hint: () => 'scroll toward it', + }; +} + +test('throwIfOffscreenInteractionTarget: an on-screen node passes through unchanged', async () => { + const device = createInteractionDevice(makeSnapshotState([])); + const nodes = makeSnapshotState([ + { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, + { + index: 1, + depth: 1, + parentIndex: 0, + type: 'Button', + rect: { x: 20, y: 20, width: 40, height: 40 }, + }, + ]).nodes; + + const result = await throwIfOffscreenInteractionTarget( + device, + { session: 'default' }, + nodes[1]!, + nodes, + fakeOffscreenFailure(), + ); + + assert.equal(result, nodes[1]); +}); + +test('throwIfOffscreenInteractionTarget: off-screen + backend confirms -> returns the node patched with the LIVE rect', async () => { + const device = createInteractionDevice(makeSnapshotState([]), { + confirmOffscreenTargetVisible: async () => ({ x: 30, y: 30, width: 40, height: 40 }), + }); + const nodes = makeSnapshotState([ + { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, + { + index: 1, + depth: 1, + parentIndex: 0, + type: 'Button', + rect: { x: 20, y: 2000, width: 40, height: 40 }, + }, + ]).nodes; + + const result = await throwIfOffscreenInteractionTarget( + device, + { session: 'default' }, + nodes[1]!, + nodes, + fakeOffscreenFailure(), + ); + + assert.deepEqual(result.rect, { x: 30, y: 30, width: 40, height: 40 }); + assert.equal(result.index, nodes[1]!.index); +}); + +test('throwIfOffscreenInteractionTarget: off-screen + no rescue -> throws with the supplied failure shape', async () => { + const device = createInteractionDevice(makeSnapshotState([])); + const nodes = makeSnapshotState([ + { index: 0, depth: 0, type: 'Application', rect: { x: 0, y: 0, width: 400, height: 800 } }, + { + index: 1, + depth: 1, + parentIndex: 0, + type: 'Button', + rect: { x: 20, y: 2000, width: 40, height: 40 }, + }, + ]).nodes; + + await assert.rejects( + () => + throwIfOffscreenInteractionTarget( + device, + { session: 'default' }, + nodes[1]!, + nodes, + fakeOffscreenFailure(), + ), + /off-screen/, + ); +}); diff --git a/src/commands/interaction/runtime/target-visibility-stages.ts b/src/commands/interaction/runtime/target-visibility-stages.ts new file mode 100644 index 0000000000..5963dd1a42 --- /dev/null +++ b/src/commands/interaction/runtime/target-visibility-stages.ts @@ -0,0 +1,219 @@ +import { AppError } from '@agent-device/kernel/errors'; +import type { + Point, + SnapshotKeyboardBandFact, + SnapshotNode, + SnapshotState, +} from '@agent-device/kernel/snapshot'; +import { normalizeRef } from '@agent-device/kernel/snapshot'; +import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; +import { + runNodePipelineStages, + type SelectorPipelineHooks, +} from '@agent-device/selectors/selector-pipeline'; +import type { SelectorPipelinePolicy } from '@agent-device/selectors/selector-pipeline-policy'; +import { createSnapshotVisibility } from '@agent-device/contracts/snapshot'; +import { + classifyOffscreenScrollDirection, + type OffscreenScrollDirection, +} from '@agent-device/capture-kit/mobile-snapshot-semantics'; +import { toBackendContext } from '../../runtime-common.ts'; +import { assertTapTargetClearOfVisibleKeyboard } from './keyboard-occlusion.ts'; +import { interactionVerb } from './interaction-verb.ts'; +import type { InteractionAction } from './interaction-resolution-request.ts'; + +/** + * The one construction site for "covered by another visible element" refusals, shared by the + * node-stage runner below (ref, selector, native-ref preflight) and `selector-readiness.ts`'s + * covered-target diagnosis probe. + */ +export function buildCoveredInteractionError(params: { + label: string; + node: SnapshotNode; + action: InteractionAction; + selector?: string; +}): AppError { + return new AppError( + 'COMMAND_FAILED', + `${params.label} is covered by another visible element and cannot ${interactionVerb(params.action)} safely`, + { + hint: 'Use a different visible target, scroll it clear of the overlay, or inspect with snapshot/screenshot before retrying.', + ...(params.selector ? { selector: params.selector } : {}), + ref: `@${params.node.ref}`, + interactionBlocked: params.node.interactionBlocked, + }, + ); +} + +/** + * Every node stage this action's row declares, plus the covered and keyboard refusals the + * interaction runtime owns. Which stages run is the row's decision; every acting path — selector, + * ref, and the native-ref preflight — enters them here, which is what keeps the native-ref fast + * path from succeeding on a target the shared rules would refuse. + * + * Each path hands in the resolver that produces the point it taps with, and taps the point that + * comes back, so the keyboard guard measures the coordinate the interaction is actually made of and no + * path derives a second one. A path whose point can fail to exist — the native-ref fast path taps by + * ref, reading the rect center the platform aims at — says so in its resolver's return type. + */ +export async function runInteractionPipelineStages(params: { + policy: SelectorPipelinePolicy; + nodes: SnapshotState['nodes']; + /** The keyboard band `nodes`' capture measured, when it measured one (#2660). */ + keyboard?: SnapshotKeyboardBandFact; + node: SnapshotNode; + action: InteractionAction; + label: string; + hooks: SelectorPipelineHooks; + resolveTapPoint: (node: SnapshotNode) => TPoint; +}): Promise<{ node: SnapshotNode; tapPoint: TPoint }> { + const target = await runNodePipelineStages( + params.policy, + params.nodes, + params.node, + params.hooks, + ); + if (target.kind === 'occluded') { + throw buildCoveredInteractionError({ + label: params.label, + node: target.node, + action: params.action, + }); + } + const tapPoint = params.resolveTapPoint(target.node); + assertTapTargetClearOfVisibleKeyboard({ + nodes: params.nodes, + node: target.node, + action: params.action, + label: params.label, + ...(params.keyboard ? { keyboard: params.keyboard } : {}), + tapPoint, + }); + return { node: target.node, tapPoint }; +} + +/** + * The off-screen stage's refusal shape. Reached only through the pipeline + * owner, and only for rows whose off-screen stage refuses — the row's decision + * is made there, so this builds the message and never re-decides. + */ +type OffscreenStageParams = { action: InteractionAction; pipeline: SelectorPipelinePolicy }; + +// Selector parity for the @ref off-screen guard: without it, a selector +// resolving to a closed drawer/carousel item "succeeds" by tapping coordinates +// outside the viewport (observed as `Tapped (-161, 265)` against Bluesky's +// closed drawer) while the same node via @ref is refused. +export async function assertVisibleSelectorTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + selector: string, + { action }: OffscreenStageParams, +): Promise { + return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { + message: `Selector ${selector} resolved to an off-screen element and is not safe to ${action}`, + details: { reason: 'offscreen_selector', selector }, + // A selector re-resolves against a fresh snapshot on every attempt, so the + // recovery is: move the named direction, then retry THIS selector — no + // separate snapshot step, and no @ref (a scroll expires the ref frame, + // #1366). `--until` is that whole loop as one command: it checks the same + // selector between passes, which is also what keeps a large step from + // overshooting, so the hint no longer has to trade distance for accuracy. + hint: (direction) => + `${scrollRevealClause(direction, selector)} then retry ${action} with the same selector. --until checks the selector between passes, so it stops on the target rather than sailing past it. If it is inside a closed drawer or another tab, open that container first.`, + }); +} + +export async function assertVisibleRefTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + refInput: string, + { action }: OffscreenStageParams, +): Promise { + return await throwIfOffscreenInteractionTarget(runtime, options, node, nodes, { + message: `Ref ${refInput} is off-screen and not safe to ${action}`, + details: { reason: 'offscreen_ref', ref: normalizeRef(refInput) }, + // The scroll that reveals the target expires the ref frame (#1366, ADR + // 0014), so retrying this @ref would be rejected next. Steer to a selector, + // which re-resolves against a fresh snapshot and bypasses the ref-frame guard + // — and which `--until` can then check between passes. + hint: (direction) => + `${scrollRevealClause(direction, null)} then retry ${action} with a selector (e.g. text=/id=) rather than this @ref — the scroll expires the ref frame, so re-run snapshot -i before reusing any @ref.`, + }); +} + +/** + * Shared lead-in for both off-screen hints: the one command that reveals the target. + * + * When the geometry names a direction AND the caller has a selector to check, this is a complete + * `scroll --until ` — one request that stops on the target instead of the + * scroll-then-look-again loop the hint used to prescribe. Without a selector to check (an @ref + * refusal) or without a single reveal direction (off more than one edge), it degrades to naming + * the move and leaves the stop condition to the caller's own next step. + */ +function scrollRevealClause( + direction: OffscreenScrollDirection | null, + selector: string | null, +): string { + if (!direction) return 'Scroll toward it,'; + if (!selector) return `Scroll ${direction} toward it,`; + return `Run scroll ${direction} --until '${selector}' to bring it on screen,`; +} + +// Full on-screen visibility (not only the effective-viewport form): items inside an +// off-screen scrollable container (closed drawer) must also count as +// off-screen, not just items scrolled out of an on-screen container. +// +// #1542: once the bulk tree says off-screen, the guard gives iOS one chance +// to rescue a FALSE refusal via the optional backend.confirmOffscreenTargetVisible +// hook — a stale/corrupted bulk tree can say off-screen while the app is +// visually fine (zero cost on the accept path; runs only here). A confirmed +// rescue returns the node PATCHED WITH THE LIVE RECT: the caller must act on +// that returned node, never the original, because in the frozen-bulk-tree +// manifestation the original rect can be stale even when the rescue verdict +// is correct — tapping it would silently land at the wrong coordinate. The +// hook fails closed (null) on anything short of a positive confirmation, so +// a genuine refusal, or any backend without the hook, is unchanged. +// +// Exported (not just for callers here) for ADR 0011 registry honesty: +// interaction-guarantees.ts's `offscreen` cells point their `via` at this +// function, not at the contracts predicate alone, since this is the actual +// end-to-end enforcement point. +export async function throwIfOffscreenInteractionTarget( + runtime: AgentDeviceRuntime, + options: CommandContext, + node: SnapshotNode, + nodes: SnapshotState['nodes'], + failure: { + message: string; + details: Record; + hint: (direction: OffscreenScrollDirection | null) => string; + }, +): Promise { + const visibility = createSnapshotVisibility(nodes); + const viewport = node.rect ? visibility.resolveEffectiveViewport(node) : null; + if (!node.rect || !viewport || visibility.isVisibleOnScreen(node)) return node; + const rootViewport = visibility.resolveViewport(node.rect); + const liveRect = await runtime.backend.confirmOffscreenTargetVisible?.( + toBackendContext(runtime, options), + node, + rootViewport, + ); + if (liveRect) return { ...node, rect: liveRect }; + // The direction that scrolls this off-screen target into view. Named in the + // hint (and surfaced as a machine-readable detail) so the recovery is a single + // deterministic move instead of a guess (#1366). Derived from the same + // boundary the rejection above used, so partial clips and off-screen + // containers get a direction too, not just fully-scrolled-out items. + const scrollDirection = classifyOffscreenScrollDirection(node, visibility); + throw new AppError('COMMAND_FAILED', failure.message, { + ...failure.details, + rect: node.rect, + viewport, + ...(scrollDirection ? { scrollDirection } : {}), + hint: failure.hint(scrollDirection), + }); +} From 87cd2b9f8eb1172d70d569a2c1af50f4a0567d4c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 15:07:41 +0200 Subject: [PATCH 08/23] refactor(replay): read the readiness-budgeted commands from their declaration Replay hardcoded press/click/longpress as the commands that get a default readinessTimeoutMs. Nothing declared that set per command: readinessTimeoutMs is a common input field every command reads. press, click and longpress now declare `targetReadiness: 'budgeted'` in the command registry. Replay reads it through commandAcceptsReadinessBudget. resolveCommandTimeoutPolicy attaches the trait to the resolved timeout policy, and resolveCommandRequestTimeoutMs widens the envelope by the readiness budget only when that trait is present. Before, it widened for any command whose flags carried readinessTimeoutMs; only these three commands forward the field to request flags, so no request changes. A test asserts that the commands with the trait equal the commands the `runtime` targetReadiness guarantee cells in interaction-guarantees.ts apply to. The replay default test iterates the declared set. --- packages/command-registry/src/registry.ts | 15 +++++++++++- .../command-registry/src/timeout-policy.ts | 8 +++++-- packages/command-registry/src/types.ts | 17 +++++++++++++ .../session-replay-action-runtime.ts | 10 +++----- .../command-descriptor-timeout-policy.test.ts | 24 +++++++++++++++++++ .../session-replay-action-runtime.test.ts | 12 ++++++++-- 6 files changed, 74 insertions(+), 12 deletions(-) diff --git a/packages/command-registry/src/registry.ts b/packages/command-registry/src/registry.ts index 3dfa4738fe..7957180c08 100644 --- a/packages/command-registry/src/registry.ts +++ b/packages/command-registry/src/registry.ts @@ -1301,6 +1301,7 @@ export const RAW_COMMAND_DESCRIPTORS = [ }, timeoutPolicy: postActionObservationTimeoutPolicy('click', PRESERVE_DAEMON_TIMEOUT_POLICY), postActionObservation: postActionObservation('click'), + targetReadiness: 'budgeted', responseDataTransform: TOUCH_INTERACTION_RESPONSE_DATA_TRANSFORM, batchable: true, platformExecution: { kind: 'device-runtime', uses: clickRuntimeUses }, @@ -1330,6 +1331,7 @@ export const RAW_COMMAND_DESCRIPTORS = [ envelopeMs: 210_000, }, postActionObservation: postActionObservation('longpress'), + targetReadiness: 'budgeted', batchable: true, platformExecution: { kind: 'device-runtime', uses: longPressRuntimeUses }, }, @@ -1358,6 +1360,7 @@ export const RAW_COMMAND_DESCRIPTORS = [ frameworkTier: 'core', timeoutPolicy: postActionObservationTimeoutPolicy('press', PRESERVE_DAEMON_TIMEOUT_POLICY), postActionObservation: postActionObservation('press'), + targetReadiness: 'budgeted', responseDataTransform: TOUCH_INTERACTION_RESPONSE_DATA_TRANSFORM, batchable: true, platformExecution: { kind: 'device-runtime', uses: pressRuntimeUses }, @@ -1972,7 +1975,12 @@ function readCatalogKey(descriptor: { } const TIMEOUT_POLICY_BY_COMMAND: ReadonlyMap = new Map( - commandDescriptors.map((descriptor) => [descriptor.name, descriptor.timeoutPolicy]), + Array.from(COMMAND_DESCRIPTOR_BY_NAME.values(), (descriptor) => [ + descriptor.name, + descriptor.targetReadiness + ? { ...descriptor.timeoutPolicy, targetReadiness: descriptor.targetReadiness } + : descriptor.timeoutPolicy, + ]), ); const DEVICE_CLAIM_POLICY_BY_COMMAND: ReadonlyMap = new Map( @@ -1999,6 +2007,11 @@ export function commandSupportsSettleObservation(command: string | undefined): b return resolveCommandPostActionObservationSupport(command) !== undefined; } +export function commandAcceptsReadinessBudget(command: string | undefined): boolean { + if (command === undefined) return false; + return COMMAND_DESCRIPTOR_BY_NAME.get(command)?.targetReadiness === 'budgeted'; +} + export function commandSupportsVerifyEvidence(command: string | undefined): boolean { return resolveCommandPostActionObservationSupport(command) === 'settle-and-verify'; } diff --git a/packages/command-registry/src/timeout-policy.ts b/packages/command-registry/src/timeout-policy.ts index 3f09b54d44..d5c8754dba 100644 --- a/packages/command-registry/src/timeout-policy.ts +++ b/packages/command-registry/src/timeout-policy.ts @@ -83,14 +83,18 @@ export function resolveCommandRequestTimeoutMs( resolvePositionalBudgetTimeoutMs(boundedPolicy, input.positionals ?? []) ?? resolveFlagBudgetTimeoutMs(boundedPolicy, input.flags) ?? boundedPolicy.envelopeMs; - return envelopeMs + readinessBudgetMs(input.flags); + return envelopeMs + readinessBudgetMs(policy, input.flags); } /** * A readiness budget is target-poll time spent before the action itself, so it extends whatever * envelope the command otherwise has; the daemon's own poll must never outlive the client's clock. */ -function readinessBudgetMs(flags: RequestTimeoutInput['flags']): number { +function readinessBudgetMs( + policy: CommandTimeoutPolicy, + flags: RequestTimeoutInput['flags'], +): number { + if (policy.targetReadiness !== 'budgeted') return 0; const budgetMs = flags?.readinessTimeoutMs; return typeof budgetMs === 'number' && Number.isFinite(budgetMs) && budgetMs > 0 ? budgetMs : 0; } diff --git a/packages/command-registry/src/types.ts b/packages/command-registry/src/types.ts index 934bd0bc38..5bcafffb4e 100644 --- a/packages/command-registry/src/types.ts +++ b/packages/command-registry/src/types.ts @@ -75,8 +75,21 @@ export type CommandTimeoutPolicy = { budget: CommandTimeoutBudget; envelopeMs: number | 'unbounded'; onTimeout: 'preserve-daemon' | 'reset-daemon'; + /** + * The descriptor's {@link CommandTargetReadiness} trait, attached by `resolveCommandTimeoutPolicy` + * so the envelope can widen by the readiness budget. Descriptors declare it as `targetReadiness`, + * never on their `timeoutPolicy`. + */ + targetReadiness?: CommandTargetReadiness; }; +/** + * `budgeted`: the command's target resolution may poll for a target that does not exist yet, under + * a caller-supplied `readinessTimeoutMs`. Must match the commands the `targetReadiness` interaction + * guarantee cells mark `runtime`. + */ +export type CommandTargetReadiness = 'budgeted'; + /** * #1320 "Command descriptor policy": what a command may do with the host-global * device claim store. REQUIRED on every descriptor (no default), and read by the @@ -199,6 +212,9 @@ export type TargetIdentityVerification = 'pre-dispatch' | 'post-resolution'; * commands that support `--settle`/`--verify`; consumed by * command surfaces and timeout policy instead of repeated * command-name lists. + * - `targetReadiness` — optional; the commands whose target resolution accepts a + * `readinessTimeoutMs` budget. Read by replay (which supplies a + * default budget) and by the request envelope (which widens by it). * - `responseDataTransform` — optional public response data shaping rules for * command-owned fields in daemon responses. This keeps * response shaping on the same descriptor surface as other @@ -221,6 +237,7 @@ type CommandDescriptorBase = { */ deviceClaimPolicy: DeviceClaimPolicy; postActionObservation?: PostActionObservationSupport; + targetReadiness?: CommandTargetReadiness; responseDataTransform?: CommandResponseDataTransform; catalog: CommandCatalogFacet; /** Required iff `catalog.group === 'public'`; see {@link CommandFrameworkTier}. */ diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index 23e63f64e3..ce2d4c03cc 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -6,6 +6,7 @@ import type { ReplayInvoke, } from '@agent-device/replay-port/command-types'; import { mergeParentFlags } from '@agent-device/command-registry/batch'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; import { AppError, normalizeError } from '@agent-device/kernel/errors'; import { gesturePayloadFromPositionals, @@ -216,15 +217,10 @@ function readResponseTiming(data: unknown): Record | undefined } /** - * A replayed press/click/longpress step carries no readiness budget of its own; replay supplies one + * A replayed step of a readiness-budgeted command carries no budget of its own; replay supplies one * so a step recorded against a loading screen can land. */ const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; -const READINESS_BUDGETED_REPLAY_COMMANDS: ReadonlySet = new Set([ - 'press', - 'click', - 'longpress', -]); function buildReplayActionFlags( parentFlags: CommandFlags | undefined, @@ -232,7 +228,7 @@ function buildReplayActionFlags( command: string, ): CommandFlags { const flags = mergeParentFlags(parentFlags, { ...(actionFlags ?? {}) }); - if (READINESS_BUDGETED_REPLAY_COMMANDS.has(command) && flags.readinessTimeoutMs === undefined) { + if (commandAcceptsReadinessBudget(command) && flags.readinessTimeoutMs === undefined) { flags.readinessTimeoutMs = REPLAY_DEFAULT_READINESS_TIMEOUT_MS; } return flags; diff --git a/src/__tests__/command-descriptor-timeout-policy.test.ts b/src/__tests__/command-descriptor-timeout-policy.test.ts index 3bbb2b4498..0b4e4c58d7 100644 --- a/src/__tests__/command-descriptor-timeout-policy.test.ts +++ b/src/__tests__/command-descriptor-timeout-policy.test.ts @@ -2,10 +2,12 @@ import { test } from 'vitest'; import assert from 'node:assert/strict'; import { PUBLIC_COMMANDS } from '@agent-device/command-registry/catalog'; import { + commandAcceptsReadinessBudget, commandDescriptors, resolveCommandPostActionObservationSupport, resolveCommandTimeoutPolicy, } from '@agent-device/command-registry/registry'; +import { INTERACTION_DISPATCH_PATHS } from '@agent-device/contracts/interaction-guarantees'; import { DEFAULT_TIMEOUT_POLICY, resolveCommandRequestTimeoutMs, @@ -423,4 +425,26 @@ test('a readiness budget widens the request envelope on top of the settle envelo 212_000, ); assert.equal(resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 0 } }), 90_000); + assert.equal( + resolveCommandRequestTimeoutMs(resolveCommandTimeoutPolicy('fill'), { + flags: { readinessTimeoutMs: 2_000 }, + }), + 90_000, + ); +}); + +test('the readiness-budgeted commands are the ones a runtime targetReadiness cell enforces', () => { + const declared = commandDescriptors + .map((descriptor) => descriptor.name) + .filter((command) => commandAcceptsReadinessBudget(command)) + .sort(); + const enforced = [ + ...new Set( + Object.values(INTERACTION_DISPATCH_PATHS).flatMap((path) => { + const cell = path.guarantees.targetReadiness; + return cell.kind === 'runtime' ? (cell.appliesTo ?? path.commands) : []; + }), + ), + ].sort(); + assert.deepEqual(declared, enforced); }); diff --git a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts index d2d4cc97d0..4c4092464f 100644 --- a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts +++ b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts @@ -6,6 +6,10 @@ import type { DaemonRequest } from '../../daemon-request.ts'; import { invokeReplayAction } from '@agent-device/replay-port/session-replay-action-runtime'; import { replayDaemonDependencies } from '../../handlers/session-replay-command.ts'; import { resolveReplayAction } from '@agent-device/ad-script'; +import { + commandAcceptsReadinessBudget, + commandDescriptors, +} from '@agent-device/command-registry/registry'; const REPLAY_REQUEST: DaemonRequest = { token: 'token', @@ -58,9 +62,13 @@ test.each(['', ' '])( }, ); +const READINESS_BUDGETED_COMMANDS = commandDescriptors + .map((descriptor) => descriptor.name) + .filter((command) => commandAcceptsReadinessBudget(command)); + // A replay step has no CLI flag to carry a readiness budget, so `buildReplayActionFlags` defaults -// one in for press/click/longpress. -test.each(['press', 'click', 'longpress'])( +// one in for every readiness-budgeted command. +test.each(READINESS_BUDGETED_COMMANDS)( 'replay defaults readinessTimeoutMs onto a dispatched %s step', async (command) => { const action: SessionAction = { ts: 0, command, positionals: ['label="Continue"'], flags: {} }; From 78ba97ce5a0e6ce237b048aa0219a1d8e62ada0c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 15:36:15 +0200 Subject: [PATCH 09/23] refactor(interaction): keep the engine and the readiness wait; loop migrations move to a stacked PR The post-gesture stability and scroll-movement loops return to their main versions, together with the scroll-movement observation provider scenario and the outcomeObservation guarantee cell that described them. They move to a stacked PR: both loops run captures that may not honor a per-capture deadline signal, so the engine needs a declared per-capture deadline mode and slow-capture scenarios before they can migrate without turning a late capture into a stalled end. The shared ride-out classifier had no production reader (readiness rides out isUnreadableCaptureContentError), so it is deleted with the contracts observation subpath. ObservationEnd has one reader, the engine, and now lives in observe-until.ts. --- .../capture-kit/src/observe-until.test.ts | 15 +- packages/capture-kit/src/observe-until.ts | 12 +- .../capture-kit/src/post-gesture-stability.ts | 113 ++++----- packages/contracts/package.json | 4 - .../contracts/src/interaction-guarantees.ts | 51 ----- packages/contracts/src/observation.ts | 24 -- .../layering/contracts-exports.snapshot.json | 1 - .../contracts/interaction-guarantees.test.ts | 6 - src/daemon/scroll-movement.ts | 77 ++++--- .../scroll-movement-observation.test.ts | 215 ------------------ 10 files changed, 109 insertions(+), 409 deletions(-) delete mode 100644 packages/contracts/src/observation.ts delete mode 100644 test/integration/provider-scenarios/scroll-movement-observation.test.ts diff --git a/packages/capture-kit/src/observe-until.test.ts b/packages/capture-kit/src/observe-until.test.ts index bf00cb7222..3f432d7c1a 100644 --- a/packages/capture-kit/src/observe-until.test.ts +++ b/packages/capture-kit/src/observe-until.test.ts @@ -1,7 +1,7 @@ import assert from 'node:assert/strict'; import { describe, test } from 'vitest'; import { AppError } from '@agent-device/kernel/errors'; -import { isRideOutCaptureError } from '@agent-device/contracts/observation'; +import { isUnreadableCaptureContentError } from '@agent-device/contracts/android-snapshot-quality'; import { observeUntil, type ObservationClock } from './observe-until.ts'; /** A clock that advances only when the loop sleeps or a capture declares its own cost. */ @@ -23,6 +23,13 @@ function fakeClock(): ObservationClock & { advance(ms: number): void; slept: num const SCHEDULE = { intervalMs: 200, budgetMs: 1_000 }; +/** The Android helper's content verdict: the capture ran but held no readable app content. */ +function unreadableContent(): AppError { + return new AppError('COMMAND_FAILED', 'no readable content', { + androidSnapshotHelperFailureReason: 'system-window-only', + }); +} + describe('observeUntil', () => { test('answers done on the poll whose verdict accepts, with the timeline', async () => { const clock = fakeClock(); @@ -69,12 +76,12 @@ describe('observeUntil', () => { const observed = await observeUntil({ capture: async () => { captures += 1; - if (captures === 1) throw new AppError('COMMAND_FAILED', 'busy', { retriable: true }); + if (captures === 1) throw unreadableContent(); return captures; }, verdict: (latest) => ({ kind: 'done', result: latest }), schedule: SCHEDULE, - rideOut: isRideOutCaptureError, + rideOut: isUnreadableCaptureContentError, clock, }); assert.equal(observed.kind, 'done'); @@ -93,7 +100,7 @@ describe('observeUntil', () => { }, verdict: () => ({ kind: 'continue' }), schedule: SCHEDULE, - rideOut: isRideOutCaptureError, + rideOut: isUnreadableCaptureContentError, clock, }); assert.equal(observed.kind, 'failed'); diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts index 7ce5e2c2bb..0623bf3d2b 100644 --- a/packages/capture-kit/src/observe-until.ts +++ b/packages/capture-kit/src/observe-until.ts @@ -1,5 +1,4 @@ import { createRequestCanceledError } from '@agent-device/kernel/errors'; -import type { ObservationEnd } from '@agent-device/contracts/observation'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; /** @@ -10,6 +9,17 @@ import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; * pixel). */ +/** + * How an observation loop ended. + * + * - `done` — the verdict accepted a capture. + * - `expired` — the budget ran out while the verdict still said continue. + * - `stalled` — a capture was still in flight at its deadline; it was cancelled and joined, so no + * late result can mutate state after the loop returned. + * - `failed` — a capture error the loop does not ride out. + */ +export type ObservationEnd = 'done' | 'expired' | 'stalled' | 'failed'; + export type ObservationSchedule = Readonly<{ /** Delay between polls, clamped to the remaining budget. */ intervalMs: number; diff --git a/packages/capture-kit/src/post-gesture-stability.ts b/packages/capture-kit/src/post-gesture-stability.ts index 9610bf5d49..5ce237f1ce 100644 --- a/packages/capture-kit/src/post-gesture-stability.ts +++ b/packages/capture-kit/src/post-gesture-stability.ts @@ -1,29 +1,21 @@ import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; +import { sleep } from '@agent-device/host-kit/retry'; import type { PostGestureAction, PostGestureOutcome } from '@agent-device/kernel/snapshot'; -import { observeUntil, type ObservationSchedule } from './observe-until.ts'; /** - * Pure post-gesture stability mechanics: the quiet-window verdict over the shared - * `observeUntil` engine, plus the baseline-distrust decision, parameterized over - * the capture value and the signature comparators. Deliberately a leaf — it - * imports no cycle owners and no `SessionState`, so it stays outside the R9 - * type cycle while the deferred-interaction-outcome owner (which holds the - * pending record and the session mutation) stays the one seam callers see. The - * owner supplies the comparators from interaction-outcome-policy; their - * semantics (subset tolerance, identity keying, discriminating entries) are - * documented there. + * Pure post-gesture stability mechanics: the quiet-window polling loop and the + * baseline-distrust verdict, parameterized over the capture value and the + * signature comparators. Deliberately a leaf — it imports no cycle owners and + * no `SessionState`, so it stays outside the R9 type cycle while the + * deferred-interaction-outcome owner (which holds the pending record and the + * session mutation) stays the one seam callers see. The owner supplies the + * comparators from interaction-outcome-policy; their semantics (subset + * tolerance, identity keying, discriminating entries) are documented there. */ -/** - * Cadence and budget for the quiet-window loop: poll every 200ms, allow 1.5s - * before an unsettled surface times out, and always complete two observations - * (an `initial` counts) so a quiet pair can form even under a tight budget. - */ -const POST_GESTURE_STABILITY_SCHEDULE: ObservationSchedule = { - intervalMs: 200, - budgetMs: 1_500, - minPolls: 2, -}; +const STABILIZATION_DEADLINE_MS = 1_500; +const STABILIZATION_INTERVAL_MS = 200; +const STABILIZATION_MIN_ATTEMPTS = 2; /** * Defect 2 (#1542): a bounded extra budget used ONLY when a quiet signature @@ -38,7 +30,7 @@ const POST_GESTURE_STABILITY_SCHEDULE: ObservationSchedule = { * settle-zero-margin-flake, a week-long contention-flake root cause), so this * cap is sized to never come close to that trap. */ -const STABILIZATION_DISTRUST_DEADLINE_MS = POST_GESTURE_STABILITY_SCHEDULE.budgetMs + 2_000; +const STABILIZATION_DISTRUST_DEADLINE_MS = STABILIZATION_DEADLINE_MS + 2_000; export type BaselineSurfaceEvidence = 'changed' | 'unchanged' | 'ambiguous'; @@ -124,11 +116,10 @@ export function decidePostGestureStabilityVerdict( } /** - * The quiet-window stability verdict over the shared observation engine: keep - * polling until two consecutive captures agree and the decision accepts the - * agreement, or the (possibly distrust-extended) budget expires. Session - * state never enters here — the caller owns the pending record's lifecycle - * and clears it when this returns. + * The quiet-window stability loop: poll until two consecutive captures agree + * and the verdict accepts the agreement, or the (possibly distrust-extended) + * deadline expires. Session state never enters here — the caller owns the + * pending record's lifecycle and clears it when this returns. */ export async function runPostGestureStabilityLoop(params: { pending: PostGestureStabilityPending; @@ -138,34 +129,24 @@ export async function runPostGestureStabilityLoop> { const { pending, needsBaselineDistrust, hooks } = params; const startedAt = Date.now(); - let attempts = 0; + let attempts = 1; + let previous = await captureSurface(hooks, params.initial); let baselineSignature = pending.baselineSignature; let baselineBackend = pending.baselineBackend; let baselineRebased = false; + // Extended past STABILIZATION_DEADLINE_MS only when the distrust verdict + // fires below; the ordinary (non-distrust) timeout path is unaffected. + let effectiveDeadlineMs = STABILIZATION_DEADLINE_MS; // A rebase or a distrust verdict keeps polling on a pair that DID agree, so - // the budget can expire on a surface that is already at rest. + // the deadline can expire on a surface that is already at rest. let lastPairAgreed = false; - let surfaceCache: CapturedSurface | undefined; - - const surfaceOf = (value: T): CapturedSurface => { - if (surfaceCache?.value === value) return surfaceCache; - surfaceCache = { value, ...hooks.readSurface(value) }; - return surfaceCache; - }; - - const observed = await observeUntil>({ - ...(params.initial !== undefined ? { initial: params.initial } : {}), - capture: () => hooks.capture(), - schedule: POST_GESTURE_STABILITY_SCHEDULE, - verdict: (latest, previousValue) => { - attempts += 1; - const current = surfaceOf(latest); - if (previousValue === undefined) return { kind: 'continue' }; - - const previous = surfaceOf(previousValue); - lastPairAgreed = hooks.signaturesStable(previous.signature, current.signature); - if (!lastPairAgreed) return { kind: 'continue' }; + while (attempts < STABILIZATION_MIN_ATTEMPTS || Date.now() - startedAt < effectiveDeadlineMs) { + await sleep(STABILIZATION_INTERVAL_MS); + attempts += 1; + const current = await captureSurface(hooks); + lastPairAgreed = hooks.signaturesStable(previous.signature, current.signature); + if (lastPairAgreed) { const elapsedMs = Date.now() - startedAt; // A capture plan may fall back or be pre-empted by the XCTest-channel // penalty at any time, so the backend can change mid-poll. Backends do @@ -181,7 +162,8 @@ export async function runPostGestureStabilityLoop = { @@ -227,6 +204,14 @@ type CapturedSurface = { backend: string | undefined; }; +async function captureSurface( + hooks: PostGestureStabilityHooks, + initial?: T, +): Promise> { + const value = initial ?? (await hooks.capture()); + return { value, ...hooks.readSurface(value) }; +} + function emitSettleDiagnostic( verdict: 'trust' | 'accept-stale', action: string, diff --git a/packages/contracts/package.json b/packages/contracts/package.json index 8a3cfe9bad..9582f1ada8 100644 --- a/packages/contracts/package.json +++ b/packages/contracts/package.json @@ -308,10 +308,6 @@ "types": "./src/facades/observability.ts", "default": "./src/facades/observability.ts" }, - "./observation": { - "types": "./src/observation.ts", - "default": "./src/observation.ts" - }, "./orientation-runtime": { "types": "./src/orientation-runtime.ts", "default": "./src/orientation-runtime.ts" diff --git a/packages/contracts/src/interaction-guarantees.ts b/packages/contracts/src/interaction-guarantees.ts index 0410b9eae4..57b56a54a9 100644 --- a/packages/contracts/src/interaction-guarantees.ts +++ b/packages/contracts/src/interaction-guarantees.ts @@ -78,11 +78,6 @@ export const INTERACTION_GUARANTEES = [ // dispatch time. Distinct from occlusion/offscreen/nonHittable, which judge a target already // found; this is about whether one is found at all. 'targetReadiness', - // What the path observes after dispatch to back its success claim, and whether that observation - // is advisory (a hint the caller may ignore) or can change the response (a corroborated failure, - // a settled diff). Distinct from the opt-in verifyEvidence/settleObservation features; this is the - // path's baseline, always-on claim. - 'outcomeObservation', ] as const; export type InteractionGuarantee = (typeof INTERACTION_GUARANTEES)[number]; @@ -159,18 +154,6 @@ const SHARED_RESPONSE_CONSTRUCTION: GuaranteeEnforcement = { via: 'src/daemon/interaction/internal/interaction-touch-response.ts#buildInteractionResponseData', }; -// runtime-selector and runtime-ref share this waiver by construction: neither path observes the tap -// outcome by default. The deferred/corroborated-failure outcome mark (a recorded XCTest failure -// after a tap is an ambiguous outcome, per docs/agents/selector-capture.md) is scoped to -// navigation-sensitive Android actions and target-authored gestures, not the tap-shaped runtime -// paths; --settle/--verify are the opt-in ways to observe one. -const TAP_OUTCOME_NOT_OBSERVED_GAP: GuaranteeEnforcement = { - kind: 'waived', - reason: - 'gap: neither path observes the tap outcome by default; only --settle/--verify capture post-action evidence, and the deferred/corroborated-failure outcome mark applies only to navigation-sensitive Android actions and target-authored gestures.', - trackingIssue: GAPS_UMBRELLA_ISSUE, -}; - // Both Maestro-compatible fast paths (src/daemon/interaction/internal/interaction-touch-direct-ios.ts) // dispatch ONE fused XCTest runner request (`command: 'tap'` with a selectorKey/selectorValue) built // in packages/platform-apple/src/interactions.ts#tapElementSelector — not the separate @@ -182,19 +165,6 @@ const DIRECT_IOS_SINGLE_QUERY_READINESS: GuaranteeEnforcement = { "Intentional: the fused runner request performs a single XCTest selector/frame query per dispatch and does not itself retry for an as-yet-nonexistent target; a miss is a runner failure (see errorTaxonomy) rather than a wait, and delegation-on-error (ADR 0011) is a different path's guarantee, not a retry of this one.", }; -// Shares the SAME reasoning as TAP_OUTCOME_NOT_OBSERVED_GAP: the tap outcome is not observed on the -// success path here either. The shared ambiguous-XCTest-failure corroboration -// (src/daemon/interaction/internal/interaction-ios-tap-outcome.ts#corroborateIosTapFailure) is -// invoked identically from this fast path and from the tree-based runtime dispatch, but only ever -// reconsiders a THROWN error — it never validates a call that returned success, so it is not a -// baseline outcome claim for either path. -const DIRECT_IOS_OUTCOME_NOT_OBSERVED_GAP: GuaranteeEnforcement = { - kind: 'waived', - reason: - "gap: this replay-only route observes no outcome beyond the runner's own report; --verify/--settle do not apply here (see the inapplicable cells on this row), and the shared ambiguous-failure corroboration only reconsiders a thrown error, never a successful dispatch.", - trackingIssue: GAPS_UMBRELLA_ISSUE, -}; - // The two runtime tree paths (selector and ref resolution) run the SAME shared // guard/observation implementations; only how the target is found // (disambiguation) and how failures are described (errorTaxonomy) differ. @@ -292,7 +262,6 @@ export const INTERACTION_DISPATCH_PATHS: Record { // (umbrella: https://github.com/callstack/agent-device/issues/1081). assert.deepEqual(gaps.sort(), [ 'maestro-direct-selector/errorTaxonomy', - 'maestro-direct-selector/outcomeObservation', 'maestro-direct-selector/parentOwnedTouchPoint', 'maestro-direct-selector/responseIdentity', 'maestro-non-hittable-fallback/errorTaxonomy', - 'maestro-non-hittable-fallback/outcomeObservation', 'maestro-non-hittable-fallback/parentOwnedTouchPoint', - 'native-ref/outcomeObservation', - 'runtime-ref/outcomeObservation', - 'runtime-selector/outcomeObservation', - 'target-drag/outcomeObservation', ]); }); diff --git a/src/daemon/scroll-movement.ts b/src/daemon/scroll-movement.ts index c9e9383397..7bc96f6ca0 100644 --- a/src/daemon/scroll-movement.ts +++ b/src/daemon/scroll-movement.ts @@ -14,7 +14,7 @@ import { containsPoint } from '@agent-device/kernel/rect'; import { AppError } from '@agent-device/kernel/errors'; import type { Point, Rect, SnapshotState } from '@agent-device/kernel/snapshot'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; -import { observeUntil, type ObservationSchedule } from '@agent-device/capture-kit/observe-until'; +import { sleep } from '@agent-device/host-kit/retry'; import { areInteractionSurfaceSignaturesStable, buildInteractionSurfaceSignature, @@ -62,10 +62,8 @@ import type { SessionState } from './session-state.ts'; */ /** How long a scroll keeps asking whether an untouched surface is really untouched (#1542's window). */ -const SCROLL_MOVEMENT_SCHEDULE: ObservationSchedule = { - intervalMs: 200, - budgetMs: 1_500, -}; +const MOVEMENT_VERDICT_BUDGET_MS = 1_500; +const MOVEMENT_POLL_MS = 200; /** The pre-gesture surface, already in hand: no capture is spent to produce it. */ export type ScrollSurfaceBaseline = Readonly<{ @@ -219,12 +217,6 @@ type SurfaceVerdict = startedAt: number; }>; -/** What one capture's verdict decides, before the loop's own timing is stitched back on. */ -type SurfaceJudgement = - | Readonly<{ kind: 'blind'; reason: SurfaceBlindReason }> - | Readonly<{ kind: 'moved'; observed: ObservedSurface }> - | Readonly<{ kind: 'settled'; observed: ObservedSurface; evidence: InteractionSurfaceChange }>; - /** Why this capture cannot be compared against the pre-gesture tree at all, movement included. */ type SurfaceBlindReason = 'capture-unreadable' | 'surface-unsettled' | ScrollSurfacePairDrift; @@ -265,32 +257,30 @@ async function pollForSurfaceVerdict( swipe: ScrollSwipeEvidence; }, ): Promise { - // The edge question is asked only of the pre-gesture (baseline) tree, so it never depends on a - // poll's outcome and can be answered once, before the loop starts, rather than lazily on the - // first `changed` reading. - const changeNeedsRest = await baselineEndsInDirection(baseline, params.direction, params.swipe); + const startedAt = Date.now(); + const deadline = startedAt + (params.budgetMs ?? MOVEMENT_VERDICT_BUDGET_MS); let previous: InteractionSurfaceSignature | undefined; - const observed = await observeUntil({ - capture: () => readOneCapture(baseline, params.capture), - schedule: { - intervalMs: params.pollMs ?? SCROLL_MOVEMENT_SCHEDULE.intervalMs, - budgetMs: params.budgetMs ?? SCROLL_MOVEMENT_SCHEDULE.budgetMs, - }, - verdict: (latest) => { - if (latest.kind === 'blind') - return { kind: 'done', result: { kind: 'blind', reason: latest.reason } }; - const judged = settledVerdict(latest, { previous, changeNeedsRest }); - previous = latest.observed.signature; - return judged ? { kind: 'done', result: judged } : { kind: 'continue' }; - }, - }); - - const attempts = observed.polls.length; - const startedAt = Date.now() - observed.waitedMs; - if (observed.kind !== 'done') return budgetExpiredVerdict(params, attempts, startedAt); - return observed.result.kind === 'blind' - ? observed.result - : { ...observed.result, attempts, startedAt }; + let attempts = 0; + let changeNeedsRest: boolean | undefined; + + while (true) { + const reading = await readOneCapture(baseline, params.capture); + attempts += 1; + if (reading.kind === 'blind') return { kind: 'blind', reason: reading.reason }; + if (reading.kind === 'changed') { + changeNeedsRest ??= await baselineEndsInDirection(baseline, params.direction, params.swipe); + } + const verdict = settledVerdict(reading, { + previous, + changeNeedsRest: changeNeedsRest === true, + attempts, + startedAt, + }); + if (verdict) return verdict; + if (Date.now() >= deadline) return budgetExpiredVerdict(params, attempts, startedAt); + previous = reading.observed.signature; + await sleep(params.pollMs ?? MOVEMENT_POLL_MS); + } } /** @@ -303,17 +293,26 @@ function settledVerdict( poll: { previous: InteractionSurfaceSignature | undefined; changeNeedsRest: boolean; + attempts: number; + startedAt: number; }, -): SurfaceJudgement | undefined { +): SurfaceVerdict | undefined { const atRest = surfaceIsAtRest(poll.previous, reading.observed.signature); + const { attempts, startedAt } = poll; if (reading.kind === 'changed') { if (!poll.changeNeedsRest || atRest) { - return { kind: 'moved', observed: reading.observed }; + return { kind: 'moved', observed: reading.observed, attempts, startedAt }; } return undefined; } if (!atRest) return undefined; - return { kind: 'settled', observed: reading.observed, evidence: reading.evidence }; + return { + kind: 'settled', + observed: reading.observed, + evidence: reading.evidence, + attempts, + startedAt, + }; } /** diff --git a/test/integration/provider-scenarios/scroll-movement-observation.test.ts b/test/integration/provider-scenarios/scroll-movement-observation.test.ts deleted file mode 100644 index e56e2808a6..0000000000 --- a/test/integration/provider-scenarios/scroll-movement-observation.test.ts +++ /dev/null @@ -1,215 +0,0 @@ -import assert from 'node:assert/strict'; -import { test } from 'vitest'; -import type { AppleToolProvider } from '@agent-device/platform-apple/tool-provider'; -import type { ExecOptions, ExecResult } from '@agent-device/host-kit/command'; -import { assertRpcError, assertRpcOk } from './assertions.ts'; -import { PROVIDER_SCENARIO_IOS_SIMULATOR } from './fixtures.ts'; -import { createProviderScenarioHarness, withProviderScenarioResource } from './harness.ts'; -import { - createAppleRunnerProviderFromTranscript, - createRecordingAppleToolProvider, - simctlDeviceLifecycleHandler, -} from './providers.ts'; -import { createProviderTranscript, type ProviderScenarioProviderEntry } from './transcript.ts'; - -const APP = 'com.example.app'; -const DEVICE_ID = PROVIDER_SCENARIO_IOS_SIMULATOR.id; - -// Directional-scroll observation on `observeUntil` (packages/capture-kit/src/observe-until.ts): -// `scroll down` gates its own `movement` claim on the tree it held before the gesture against one -// (or more, while the surface still looks untouched) post-gesture capture. The unit-level coverage -// in src/daemon/__tests__/scroll-movement.test.ts pins the verdict against a scripted capture -// function directly; these two scenarios drive the SAME loop end to end through the daemon's real -// command routing and a scripted Apple runner, the way `settle-observation.test.ts` drives `--settle`. - -// Fixed tab-bar-free list: a ScrollView reporting hidden content below, holding two rows. `rowOffset` -// moves the rows to model a scroll that actually shifted content; `hiddenBelow` toggles whether the -// container still reports more to reveal, which is what the no-progress refusal keys on. -const CONTAINER = { x: 18, y: 178, width: 366, height: 662 }; - -function screen(rowOffset: number, hiddenBelow: boolean) { - return [ - { - index: 0, - type: 'Application', - label: 'Example', - rect: { x: 0, y: 0, width: 393, height: 852 }, - }, - { - index: 1, - parentIndex: 0, - type: 'ScrollView', - identifier: 'lab-list', - rect: CONTAINER, - ...(hiddenBelow ? { hiddenContentBelow: true } : {}), - }, - { - index: 2, - parentIndex: 1, - type: 'StaticText', - label: 'Row one', - // 300 (+ up to -100 offset) stays inside CONTAINER's clip (y 178..840): the presentation - // validator refuses a child frame that escapes its container's cumulative clip. - rect: { x: 24, y: 300 + rowOffset, width: 300, height: 20 }, - }, - { - index: 3, - parentIndex: 1, - type: 'StaticText', - label: 'Row two', - rect: { x: 24, y: 360 + rowOffset, width: 300, height: 20 }, - }, - ]; -} - -function snapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { - return { - command: 'ios.runner.snapshot', - deviceId: DEVICE_ID, - platform: 'apple', - result: { nodes, truncated: false }, - }; -} - -// A UI that never moves: every post-gesture capture returns the same tree, so the loop can never -// see a quiet-then-still-hidden pair on a fixed count. How many captures it takes to prove that is -// wall-clock (200ms poll floor), so the count is not scripted — a repeat entry serves every call. -function repeatSnapshotEntry(nodes: readonly unknown[]): ProviderScenarioProviderEntry { - return { ...snapshotEntry(nodes), repeat: true }; -} - -// Midpoint (201, 437) sits inside CONTAINER, matching the swipe fixture -// `src/daemon/__tests__/scroll-movement.test.ts` uses for the same geometry. -function scrollEntry(): ProviderScenarioProviderEntry { - return { - command: 'ios.runner.scroll', - deviceId: DEVICE_ID, - platform: 'apple', - result: { x: 201, y: 487, x2: 201, y2: 387 }, - }; -} - -/** - * The route ahead of a directional scroll's own capture resolves a live simulator target (a real - * host process lookup) before it ever reaches the scripted runner, so a bare `simctlDeviceLifecycleHandler` - * leaves every capture with a fresh, per-call "unknown generation" identity — comparable to nothing, - * which is `capture-lineage-drift` on the very first post-gesture read. Naming one fixed, stable - * "running app" job (a `launchctl list` line) and a fixed process start time (`ps -p -o lstart=`) - * lets the route resolve ONE identity and reuse it every capture, exactly like a real simulator whose - * app process never restarts mid-scroll — the AX bridge itself still isn't modeled, so every capture - * still falls back to the scripted runner, carrying that one stable identity instead of a random one. - */ -function iosSimulatorTool(): { provider: AppleToolProvider } { - const baseSimctl = simctlDeviceLifecycleHandler('com.apple.CoreSimulator.SimRuntime.iOS-18-0', [ - { name: PROVIDER_SCENARIO_IOS_SIMULATOR.name, udid: DEVICE_ID }, - ]); - const recorded = createRecordingAppleToolProvider({ - simctl: async (args, options) => { - if (args[0] === 'spawn' && args[1] === DEVICE_ID && args[2] === 'launchctl') { - return { stdout: `4242\t0\tUIKitApplication:${APP}[0x1]\n`, stderr: '', exitCode: 0 }; - } - return await baseSimctl(args, options); - }, - }); - const runCommand = async ( - cmd: string, - args: string[], - options?: ExecOptions, - ): Promise => { - // The system-surface presence probe runs before target resolution: report every - // registered host absent so the route proceeds to resolve the (scripted) app target below, - // instead of reading the probe as 'unknown' and falling back with a fresh random identity. - if (cmd === 'pgrep') return { stdout: '', stderr: '', exitCode: 1 }; - if (cmd === 'ps' && args[0] === '-p') { - return { stdout: 'Thu Jan 1 00:00:00 1970\n', stderr: '', exitCode: 0 }; - } - return await recorded.provider.runCommand(cmd, args, options); - }; - return { provider: { ...recorded.provider, runCommand } }; -} - -test('Provider-backed integration scroll down answers moved on the first post-gesture capture', async () => { - const runnerTranscript = createProviderTranscript([ - snapshotEntry(screen(0, true)), // pre-scroll baseline - scrollEntry(), - snapshotEntry(screen(-100, true)), // post-gesture: rows shifted into view on the first read - ]); - const appleRunnerProvider = createAppleRunnerProviderFromTranscript( - runnerTranscript, - 'ios.runner', - ); - const appleTool = iosSimulatorTool(); - - await withProviderScenarioResource( - async () => - await createProviderScenarioHarness({ - appleRunnerProvider: () => appleRunnerProvider, - appleToolProvider: () => appleTool.provider, - deviceInventoryProvider: async () => [PROVIDER_SCENARIO_IOS_SIMULATOR], - }), - async (daemon) => { - assertRpcOk(await daemon.callCommand('open', [APP], { platform: 'ios', udid: DEVICE_ID })); - assertRpcOk(await daemon.callCommand('snapshot')); - - const callsBeforeScroll = runnerTranscript.calls.length; - const scroll = await daemon.callCommand('scroll', ['down']); - const scrollData = assertRpcOk<{ movement?: string }>(scroll); - - assert.equal(scrollData.movement, 'moved'); - // Gesture first, then exactly one post-gesture snapshot: a scroll that worked is confirmed by - // the cheapest possible read, and the order proves the read followed the gesture rather than - // racing it. - assert.deepEqual( - runnerTranscript.calls.slice(callsBeforeScroll).map((call) => call.command), - ['ios.runner.scroll', 'ios.runner.snapshot'], - ); - - runnerTranscript.assertComplete(); - }, - ); -}); - -test('Provider-backed integration scroll down that never moves a hidden-content container refuses with scroll_no_progress', async () => { - const runnerTranscript = createProviderTranscript([ - snapshotEntry(screen(0, true)), // pre-scroll baseline - scrollEntry(), - // Every post-gesture read is byte-identical to the baseline and the container still reports - // hidden content below: the gesture never reached it. - repeatSnapshotEntry(screen(0, true)), - ]); - const appleRunnerProvider = createAppleRunnerProviderFromTranscript( - runnerTranscript, - 'ios.runner', - ); - const appleTool = iosSimulatorTool(); - - await withProviderScenarioResource( - async () => - await createProviderScenarioHarness({ - appleRunnerProvider: () => appleRunnerProvider, - appleToolProvider: () => appleTool.provider, - deviceInventoryProvider: async () => [PROVIDER_SCENARIO_IOS_SIMULATOR], - }), - async (daemon) => { - assertRpcOk(await daemon.callCommand('open', [APP], { platform: 'ios', udid: DEVICE_ID })); - assertRpcOk(await daemon.callCommand('snapshot')); - - const callsBeforeScroll = runnerTranscript.calls.length; - const scroll = await daemon.callCommand('scroll', ['down']); - const errorData = assertRpcError(scroll, 'COMMAND_FAILED', /scroll down moved nothing/); - assert.equal( - errorData.details && (errorData.details as Record).reason, - 'scroll_no_progress', - ); - - const callsDuringScroll = runnerTranscript.calls.slice(callsBeforeScroll); - // Gesture first, then at least the quiet-pair minimum of post-gesture snapshots — an exact - // count would assert the loop's polling speed rather than the refusal itself. - assert.equal(callsDuringScroll[0]?.command, 'ios.runner.scroll'); - assert.ok( - callsDuringScroll.filter((call) => call.command === 'ios.runner.snapshot').length >= 2, - `expected at least two post-gesture captures, saw ${callsDuringScroll.length - 1}`, - ); - }, - ); -}); From 5edfe953d9ef577cbb74139c3a70dcd277080ae0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 15:40:01 +0200 Subject: [PATCH 10/23] feat(interaction): declare a per-capture deadline mode and report ridden-out errors observeUntil schedules now declare captureDeadline. 'cancel' keeps the existing contract: each capture is armed with the remaining budget and a capture that ends at or past its deadline ends the loop stalled. 'none' hands the capture no deadline, so a capture that cannot be cancelled is still judged when it finishes late and the loop ends done or expired, never stalled. A ridden-out poll now records its error, so a loop whose every capture was ridden out ends expired with lastError set; the readiness wait raises that typed unreadable-content error instead of a selector miss against an empty tree. A terminal capture failure is labelled 'failed' in the poll timeline instead of 'observed'. --- .../capture-kit/src/observe-until.test.ts | 98 +++++++++++++++++-- packages/capture-kit/src/observe-until.ts | 66 +++++++++---- .../runtime/selector-readiness.test.ts | 28 ++++++ .../interaction/runtime/selector-readiness.ts | 11 +-- 4 files changed, 172 insertions(+), 31 deletions(-) diff --git a/packages/capture-kit/src/observe-until.test.ts b/packages/capture-kit/src/observe-until.test.ts index 3f432d7c1a..a1bdde49e2 100644 --- a/packages/capture-kit/src/observe-until.test.ts +++ b/packages/capture-kit/src/observe-until.test.ts @@ -21,7 +21,7 @@ function fakeClock(): ObservationClock & { advance(ms: number): void; slept: num }; } -const SCHEDULE = { intervalMs: 200, budgetMs: 1_000 }; +const SCHEDULE = { intervalMs: 200, budgetMs: 1_000, captureDeadline: 'cancel' } as const; /** The Android helper's content verdict: the capture ran but held no readable app content. */ function unreadableContent(): AppError { @@ -105,7 +105,30 @@ describe('observeUntil', () => { }); assert.equal(observed.kind, 'failed'); assert.equal(observed.kind === 'failed' && observed.error, failure); - assert.equal(observed.polls.length, 1); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['failed'], + ); + }); + + test('reports the last ridden-out error when every poll was ridden out', async () => { + const clock = fakeClock(); + const failures: AppError[] = []; + const observed = await observeUntil({ + capture: async () => { + const failure = unreadableContent(); + failures.push(failure); + throw failure; + }, + verdict: (latest) => ({ kind: 'done', result: latest }), + schedule: SCHEDULE, + rideOut: isUnreadableCaptureContentError, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.kind === 'expired' && observed.last, undefined); + assert.equal(observed.kind === 'expired' && observed.lastError, failures.at(-1)); + assert.ok(observed.polls.every((poll) => poll.outcome === 'rode-out')); }); test('cancels and joins a capture still in flight at the deadline', async () => { @@ -119,7 +142,7 @@ describe('observeUntil', () => { }); }), verdict: () => ({ kind: 'done', result: true }), - schedule: { intervalMs: 5, budgetMs: 20 }, + schedule: { intervalMs: 5, budgetMs: 20, captureDeadline: 'cancel' }, }); assert.equal(observed.kind, 'stalled'); assert.equal(aborted, true); @@ -135,14 +158,14 @@ describe('observeUntil', () => { const observed = await observeUntil({ capture: async () => { captures += 1; - clock.advance(900); + clock.advance(1_000); return captures; }, verdict: (latest, previous) => previous !== undefined ? { kind: 'done', result: [previous, latest] } : { kind: 'continue' }, - schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2 }, + schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2, captureDeadline: 'none' }, clock, }); assert.equal(observed.kind, 'done'); @@ -196,7 +219,12 @@ describe('observeUntil budgetFrom first-capture', () => { return captures; }, verdict: (latest) => (latest === 3 ? { kind: 'done', result: latest } : { kind: 'continue' }), - schedule: { intervalMs: 200, budgetMs: 1_000, budgetFrom: 'first-capture' }, + schedule: { + intervalMs: 200, + budgetMs: 1_000, + budgetFrom: 'first-capture', + captureDeadline: 'cancel', + }, clock, }); assert.equal(observed.kind, 'done'); @@ -213,7 +241,7 @@ describe('observeUntil budgetFrom first-capture', () => { previous === undefined ? { kind: 'continue' } : { kind: 'done', result: [previous, latest] }, - schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2 }, + schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2, captureDeadline: 'cancel' }, initial: 1, clock, }); @@ -222,3 +250,59 @@ describe('observeUntil budgetFrom first-capture', () => { assert.equal(observed.polls.length, 1); }); }); + +describe('observeUntil captureDeadline', () => { + /** A second capture that returns only after the remaining budget is spent, ignoring its signal. */ + function lateSecondCapture(clock: ReturnType) { + let captures = 0; + return async () => { + captures += 1; + clock.advance(captures === 1 ? 100 : 1_500); + return captures; + }; + } + + test("'none' judges a capture that finishes past the budget and can end done", async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: lateSecondCapture(clock), + verdict: (latest) => (latest === 2 ? { kind: 'done', result: latest } : { kind: 'continue' }), + schedule: { ...SCHEDULE, captureDeadline: 'none' }, + clock, + }); + assert.equal(observed.kind, 'done'); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['observed', 'observed'], + ); + }); + + test("'none' ends expired, never stalled, when the late capture's verdict continues", async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: lateSecondCapture(clock), + verdict: () => ({ kind: 'continue' }), + schedule: { ...SCHEDULE, captureDeadline: 'none' }, + clock, + }); + assert.equal(observed.kind, 'expired'); + assert.equal(observed.kind === 'expired' && observed.last, 2); + assert.equal(observed.polls.length, 2); + }); + + test("'cancel' ends stalled on a capture that finishes past its deadline", async () => { + const clock = fakeClock(); + const observed = await observeUntil({ + capture: lateSecondCapture(clock), + verdict: (latest) => (latest === 2 ? { kind: 'done', result: latest } : { kind: 'continue' }), + schedule: SCHEDULE, + clock, + }); + assert.equal(observed.kind, 'stalled'); + assert.equal(observed.kind === 'stalled' && observed.last, 1); + assert.deepEqual( + observed.polls.map((poll) => poll.outcome), + ['observed', 'stalled'], + ); + }); +}); diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts index 0623bf3d2b..b6138ba61f 100644 --- a/packages/capture-kit/src/observe-until.ts +++ b/packages/capture-kit/src/observe-until.ts @@ -2,8 +2,8 @@ import { createRequestCanceledError } from '@agent-device/kernel/errors'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; /** - * The one observation loop: capture until a verdict accepts, on a schedule, under a budget that is - * enforced by cancellation. Owns cadence, the deadline, abort-and-join of an in-flight capture, + * The one observation loop: capture until a verdict accepts, on a schedule, under a budget. Owns + * cadence, the deadline, abort-and-join of an in-flight capture under `captureDeadline: 'cancel'`, * which capture errors are ridden out, and the per-poll timeline. Owns nothing about WHAT is * observed: the verdict is the caller's, and so is the equality rule behind it (digest, signature, * pixel). @@ -24,9 +24,10 @@ export type ObservationSchedule = Readonly<{ /** Delay between polls, clamped to the remaining budget. */ intervalMs: number; /** - * Wall-clock budget from the first capture. It bounds when a poll may start, and a capture that - * starts inside it may overrun it by at most one interval: the sample after the last sleep is - * always taken, so a verdict that reads elapsed time against a cap sees a poll at or past it. + * Wall-clock budget from the first capture. It bounds when a poll may start; under + * `captureDeadline: 'cancel'` a capture that starts inside it may overrun it by at most one + * interval. The sample after the last sleep is always taken, so a verdict that reads elapsed time + * against a cap sees a poll at or past it. */ budgetMs: number; /** Observations (an `initial` counts) the loop always completes before the budget may end it. */ @@ -37,6 +38,14 @@ export type ObservationSchedule = Readonly<{ * one-shot pays nothing on its success path. */ budgetFrom?: 'start' | 'first-capture'; + /** + * Whether a capture is bounded by its own deadline. `'cancel'` arms each capture with the + * remaining budget (at least one interval) as an abort signal, and a capture that ends at or past + * that deadline ends the loop `stalled`: declare it only when the capture honors the signal. + * `'none'` hands the capture no deadline; a capture that finishes past the budget is still judged, + * so the loop ends `done` on an accepting verdict and `expired` otherwise. + */ + captureDeadline: 'cancel' | 'none'; }>; export type ObservationClock = Readonly<{ @@ -49,7 +58,8 @@ export type ObservationVerdict = /** Keep polling; `budgetMs` raises the budget measured from the first capture (never lowers it). */ | Readonly<{ kind: 'continue'; budgetMs?: number }>; -export type ObservationPollOutcome = 'observed' | 'rode-out' | 'stalled'; +/** `failed` is a capture error the loop does not ride out; it always ends the loop. */ +export type ObservationPollOutcome = 'observed' | 'rode-out' | 'stalled' | 'failed'; export type ObservationPoll = Readonly<{ startedMs: number; @@ -65,7 +75,7 @@ export type ObservationEvidence = Readonly<{ type ObservedEnd = | Readonly<{ kind: 'done'; result: R; value: T }> | Readonly<{ kind: 'expired'; last: T | undefined; lastError: unknown }> - | Readonly<{ kind: 'stalled'; last: T | undefined; error: unknown }> + | Readonly<{ kind: 'stalled'; last: T | undefined; lastError: unknown; error: unknown }> | Readonly<{ kind: 'failed'; last: T | undefined; error: unknown }>; export type Observed = ObservationEvidence & ObservedEnd; @@ -75,7 +85,10 @@ export type ObserveUntilParams = Readonly<{ /** Judges the latest capture; `previous` is the last observed value, for quiet-pair predicates. */ verdict: (latest: T, previous: T | undefined, polls: number) => ObservationVerdict; schedule: ObservationSchedule; - /** True keeps polling past this capture error. Default: no error is ridden out. */ + /** + * True keeps polling past this capture error. Default: no error is ridden out. The last + * ridden-out error is reported as `lastError` when the loop expires or stalls. + */ rideOut?: (error: unknown) => boolean; signal?: AbortSignal; clock?: ObservationClock; @@ -164,10 +177,12 @@ async function pollOnce(loop: Loop): Promise | undefi const { params, clock, state } = loop; const unbounded = params.schedule.budgetFrom === 'first-capture' && state.polls.length === 0; const pollStartedMs = clock.now(); + const bounded = !unbounded && params.schedule.captureDeadline === 'cancel'; const poll = await captureWithin( - unbounded ? undefined : Math.max(remainingMs(loop), params.schedule.intervalMs), + bounded ? Math.max(remainingMs(loop), params.schedule.intervalMs) : undefined, params.signal, params.capture, + clock, ); state.observations += 1; if (unbounded) state.budgetStartedMs = clock.now(); @@ -178,8 +193,19 @@ async function pollOnce(loop: Loop): Promise | undefi outcome, }); if (poll.kind === 'observed') return judge(loop, poll.value); - if (outcome === 'rode-out') return undefined; - return finish(loop, { kind: poll.kind, last: state.last, error: poll.error }); + if (outcome === 'rode-out') { + state.lastError = poll.error; + return undefined; + } + if (poll.kind === 'stalled') { + return finish(loop, { + kind: 'stalled', + last: state.last, + lastError: state.lastError, + error: poll.error, + }); + } + return finish(loop, { kind: 'failed', last: state.last, error: poll.error }); } function pollOutcome( @@ -187,7 +213,7 @@ function pollOutcome( poll: CaptureWithinOutcome, ): ObservationPollOutcome { if (poll.kind === 'stalled') return 'stalled'; - if (poll.kind === 'failed') return loop.params.rideOut?.(poll.error) ? 'rode-out' : 'observed'; + if (poll.kind === 'failed') return loop.params.rideOut?.(poll.error) ? 'rode-out' : 'failed'; return 'observed'; } @@ -236,8 +262,9 @@ async function captureWithin( remainingMs: number | undefined, parent: AbortSignal | undefined, capture: (signal: AbortSignal) => Promise, + clock: ObservationClock, ): Promise> { - const deadline = armDeadline(remainingMs); + const deadline = armDeadline(remainingMs, clock); const signal = parent ? AbortSignal.any([parent, deadline.signal]) : deadline.signal; try { const value = await capture(signal); @@ -259,19 +286,24 @@ function endOfCapture( return deadlineExpired ? { kind: 'stalled', error } : undefined; } -function armDeadline(remainingMs: number | undefined): { +/** The deadline has passed once its timer fired or the loop's clock reached it, whichever is first. */ +function armDeadline( + remainingMs: number | undefined, + clock: ObservationClock, +): { signal: AbortSignal; expired: () => boolean; dispose: () => void; } { const controller = new AbortController(); - let expired = false; + const deadlineAtMs = remainingMs === undefined ? undefined : clock.now() + remainingMs; + let fired = false; const timer = remainingMs === undefined ? undefined : setTimeout( () => { - expired = true; + fired = true; controller.abort(new DOMException('Observation deadline exceeded', 'TimeoutError')); }, Math.max(0, remainingMs), @@ -279,7 +311,7 @@ function armDeadline(remainingMs: number | undefined): { timer?.unref(); return { signal: controller.signal, - expired: () => expired, + expired: () => fired || (deadlineAtMs !== undefined && clock.now() >= deadlineAtMs), dispose: () => { if (timer !== undefined) clearTimeout(timer); }, diff --git a/src/commands/interaction/runtime/selector-readiness.test.ts b/src/commands/interaction/runtime/selector-readiness.test.ts index 9467163c39..a1493b2247 100644 --- a/src/commands/interaction/runtime/selector-readiness.test.ts +++ b/src/commands/interaction/runtime/selector-readiness.test.ts @@ -93,3 +93,31 @@ test('runtime press caps a readinessTimeoutMs larger than the row maxTimeoutMs a }, ); }); + +test('runtime press whose every capture is unreadable fails with the unreadable-content error, not a selector miss', async () => { + let captures = 0; + const device = createInteractionDevice(makeSnapshotState([]), { + clock: createFakeClock(), + captureSnapshot: async () => { + captures += 1; + throw new AppError('COMMAND_FAILED', 'Android snapshot has no readable app content', { + androidSnapshotHelperFailureReason: 'system-window-only', + }); + }, + }); + + await assert.rejects( + () => + device.interactions.press(selector('label=Continue'), { + session: 'default', + readinessTimeoutMs: 2_000, + }), + (error: unknown) => { + assert.ok(error instanceof AppError); + assert.equal(error.details?.androidSnapshotHelperFailureReason, 'system-window-only'); + assert.notEqual(error.details?.reason, 'selector_not_found'); + return true; + }, + ); + assert.ok(captures >= 2, `expected the unreadable captures to be ridden out, got ${captures}`); +}); diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts index 6b258020d2..184013ff89 100644 --- a/src/commands/interaction/runtime/selector-readiness.ts +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -224,13 +224,7 @@ async function readinessExhaustedFailure( { kind: 'expired' | 'stalled' } >, ): Promise { - if ( - observed.last === undefined && - observed.kind === 'expired' && - observed.lastError !== undefined - ) { - return observed.lastError; - } + if (observed.last === undefined && observed.lastError !== undefined) return observed.lastError; const readiness: SelectorReadinessDetails = { polls: observed.polls.length, waitedMs: observed.waitedMs, @@ -292,6 +286,9 @@ export async function pollForSelectorReadiness( intervalMs: schedule.intervalMs, budgetMs: schedule.budgetMs, budgetFrom: 'first-capture', + // The poll signal reaches the platform as CaptureSnapshotInput.signal, which the snapshot + // binding joins (captureSnapshotSignal): the same per-capture cancellation `wait` relies on. + captureDeadline: 'cancel', }, rideOut: isUnreadableCaptureContentError, ...(signal ? { signal } : {}), From 44f537b8dc60586894b6280cd3606f706e9e2a16 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:03:41 +0200 Subject: [PATCH 11/23] fix(interaction): scope readinessTimeoutMs to budgeted commands and cap its envelope readinessTimeoutMs was a common input every structured command accepted and ignored. Only commands whose descriptor declares targetReadiness: 'budgeted' now carry it, through targetReadinessFields; the common input reader refuses the key with INVALID_ARGS for every command whose fields do not declare it. The request envelope widened by the raw readinessTimeoutMs, so a caller value of 999000 stretched it by 999 s while the poll itself stops at the promotedTarget row's ceiling. The widening now uses the same capped budget. The press readiness provider scenario claimed to exercise forceFresh through the selector capture cache; the press route captures through the interaction backend, which keeps no such cache, so the comments now say so. --- .../command-registry/src/timeout-policy.ts | 5 +- .../client-interactions-readiness.test.ts | 5 +- .../command-descriptor-timeout-policy.test.ts | 6 +++ src/commands/__tests__/input-audience.test.ts | 1 - src/commands/command-input.ts | 1 + src/commands/common-input-fields.ts | 47 ++++++++++--------- src/commands/interaction/metadata.ts | 4 ++ src/commands/target-readiness-grammar.test.ts | 41 ++++++++++++++++ src/commands/target-readiness-grammar.ts | 26 ++++++++++ .../command-tools-operator-inputs.test.ts | 28 ++++++++++- .../press-target-readiness.test.ts | 10 ++-- 11 files changed, 141 insertions(+), 33 deletions(-) create mode 100644 src/commands/target-readiness-grammar.test.ts create mode 100644 src/commands/target-readiness-grammar.ts diff --git a/packages/command-registry/src/timeout-policy.ts b/packages/command-registry/src/timeout-policy.ts index d5c8754dba..42a3b4fb02 100644 --- a/packages/command-registry/src/timeout-policy.ts +++ b/packages/command-registry/src/timeout-policy.ts @@ -1,3 +1,4 @@ +import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import type { CommandTimeoutBudget, CommandTimeoutPolicy } from './types.ts'; // Request-envelope constants, relocated from src/daemon/request-timeouts.ts when @@ -89,6 +90,7 @@ export function resolveCommandRequestTimeoutMs( /** * A readiness budget is target-poll time spent before the action itself, so it extends whatever * envelope the command otherwise has; the daemon's own poll must never outlive the client's clock. + * The poll never runs past the promotedTarget row's ceiling, so neither does the widening. */ function readinessBudgetMs( policy: CommandTimeoutPolicy, @@ -96,7 +98,8 @@ function readinessBudgetMs( ): number { if (policy.targetReadiness !== 'budgeted') return 0; const budgetMs = flags?.readinessTimeoutMs; - return typeof budgetMs === 'number' && Number.isFinite(budgetMs) && budgetMs > 0 ? budgetMs : 0; + if (typeof budgetMs !== 'number' || !Number.isInteger(budgetMs) || budgetMs <= 0) return 0; + return Math.min(budgetMs, SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs); } function resolvePositionalBudgetTimeoutMs( diff --git a/src/__tests__/client-interactions-readiness.test.ts b/src/__tests__/client-interactions-readiness.test.ts index 5e5c04fe72..9b77ccf024 100644 --- a/src/__tests__/client-interactions-readiness.test.ts +++ b/src/__tests__/client-interactions-readiness.test.ts @@ -3,9 +3,8 @@ import { test } from 'vitest'; import { createAgentDeviceClient } from '../agent-device-client.ts'; import { createTransport } from './client-transport-fixture.ts'; -// The readiness budget rides the common-input seam (`commonToClientOptions`), so it must reach -// `req.flags` for press/click/longpress exactly like `button`/`durationMs` do. It is never CLI- or -// model-writable; the SDK client option is its route. +// The readiness budget must reach `req.flags` for press/click/longpress exactly like +// `button`/`durationMs` do. It is never CLI- or model-writable; the SDK client option is its route. test('readinessTimeoutMs on press/click/longpress reaches the request flags', async () => { const setup = createTransport(async () => ({ ok: true, data: {} })); const client = createAgentDeviceClient(setup.config, { transport: setup.transport }); diff --git a/src/__tests__/command-descriptor-timeout-policy.test.ts b/src/__tests__/command-descriptor-timeout-policy.test.ts index 0b4e4c58d7..97d25c8706 100644 --- a/src/__tests__/command-descriptor-timeout-policy.test.ts +++ b/src/__tests__/command-descriptor-timeout-policy.test.ts @@ -8,6 +8,7 @@ import { resolveCommandTimeoutPolicy, } from '@agent-device/command-registry/registry'; import { INTERACTION_DISPATCH_PATHS } from '@agent-device/contracts/interaction-guarantees'; +import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import { DEFAULT_TIMEOUT_POLICY, resolveCommandRequestTimeoutMs, @@ -425,6 +426,11 @@ test('a readiness budget widens the request envelope on top of the settle envelo 212_000, ); assert.equal(resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 0 } }), 90_000); + assert.equal( + resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 999_000 } }), + 90_000 + SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs, + ); + assert.equal(SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs, 2_000); assert.equal( resolveCommandRequestTimeoutMs(resolveCommandTimeoutPolicy('fill'), { flags: { readinessTimeoutMs: 2_000 }, diff --git a/src/commands/__tests__/input-audience.test.ts b/src/commands/__tests__/input-audience.test.ts index 18387d0203..eb4eb67562 100644 --- a/src/commands/__tests__/input-audience.test.ts +++ b/src/commands/__tests__/input-audience.test.ts @@ -85,6 +85,5 @@ function commonInputValueFor(key: string): unknown { if (key === 'platform') return 'ios'; if (key === 'deviceTarget' || key === 'target') return 'mobile'; if (key === 'debug') return true; - if (key === 'readinessTimeoutMs') return 2_000; return `value-${key}`; } diff --git a/src/commands/command-input.ts b/src/commands/command-input.ts index 1fa4b59ebb..675e1b9d84 100644 --- a/src/commands/command-input.ts +++ b/src/commands/command-input.ts @@ -412,6 +412,7 @@ export function readFieldInput( ); const commonInput = readCommonInput(record, { readTargetAlias: !Object.hasOwn(fields, 'target'), + readinessBudgetDeclared: Object.hasOwn(fields, 'readinessTimeoutMs'), }); return compactRecord({ ...commonInput, diff --git a/src/commands/common-input-fields.ts b/src/commands/common-input-fields.ts index 2726471c96..e28dd48773 100644 --- a/src/commands/common-input-fields.ts +++ b/src/commands/common-input-fields.ts @@ -1,5 +1,4 @@ import type { CliFlags } from '@agent-device/contracts/command'; -import { readOptionalInteger } from '@agent-device/contracts/command'; import type { AgentDeviceRequestOverrides, AgentDeviceSelectionOptions, @@ -31,15 +30,13 @@ export type CommonCommandInput = Pick< androidDeviceAllowlist?: string; /** `--no-record`: common to every recordable command (see `commonInputFromFlags`). */ noRecord?: boolean; - /** - * Readiness budget for a tap-shaped interaction (press/click/longpress), capped at the - * promotedTarget row's maxTimeoutMs. No `flagIn`/`flagKey`: it never rides a CLI flag or the - * client "selection" projection, only the CLI/Node structured-input and SDK client option seams. - */ - readinessTimeoutMs?: number; }; -export type CommonInputReadOptions = { readTargetAlias?: boolean }; +export type CommonInputReadOptions = { + readTargetAlias?: boolean; + /** The command's own fields declare `readinessTimeoutMs` (`target-readiness-grammar.ts`). */ + readinessBudgetDeclared?: boolean; +}; /** * `cli-grammar/common.ts`'s two flag-derived projections a row can join: @@ -220,19 +217,9 @@ const COMMON_INPUT_FIELDS = { flagIn: ['input', 'selection'], }, readinessTimeoutMs: { - schema: { - type: 'integer', - minimum: 1, - description: - "Operator-only: how long press/click/longpress may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", - }, - read: (record) => readOptionalInteger(record, 'readinessTimeoutMs', { min: 1 }), - // No CLI flag: agents mostly miss on a wrong selector, where fast feedback beats absorbing a - // render race. - audience: operatorAudience({ - operatorPath: - 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', - }), + // No schema and no value: a command whose descriptor declares `targetReadiness: 'budgeted'` + // carries and reads the key through its own fields; every other command refuses it here. + read: refuseUndeclaredReadinessBudget, }, daemonBaseUrl: { schema: { type: 'string', description: 'Remote daemon base URL.' }, @@ -268,7 +255,10 @@ const COMMON_INPUT_FIELDS = { schema: { type: 'boolean', description: 'Enable debug diagnostics.' }, read: (record) => optionalBoolean(record, 'debug'), }, -} as const satisfies Record; +} as const satisfies Record< + keyof CommonCommandInput | 'target' | 'readinessTimeoutMs', + CommonInputFieldSpec +>; const COMMON_INPUT_ROWS: ReadonlyArray = Object.entries(COMMON_INPUT_FIELDS); @@ -346,3 +336,16 @@ function readDeviceTarget( } return deviceTarget ?? targetAlias; } + +function refuseUndeclaredReadinessBudget( + record: Record, + options: CommonInputReadOptions, +): undefined { + if (options.readinessBudgetDeclared === true || !Object.hasOwn(record, 'readinessTimeoutMs')) { + return undefined; + } + throw new AppError( + 'INVALID_ARGS', + 'readinessTimeoutMs applies only to commands that wait for their target: press, click, and longpress.', + ); +} diff --git a/src/commands/interaction/metadata.ts b/src/commands/interaction/metadata.ts index 6faba1ad95..17659852b7 100644 --- a/src/commands/interaction/metadata.ts +++ b/src/commands/interaction/metadata.ts @@ -40,6 +40,7 @@ import { readCommonInput, type CommonCommandInput } from '../common-input-fields import { readInputRecord } from '../input-readers.ts'; import { defineFieldCommandMetadata } from '../field-command-contract.ts'; import { postActionObservationFields } from '../post-action-observation-grammar.ts'; +import { targetReadinessFields } from '../target-readiness-grammar.ts'; const FIND_ACTION_VALUES = [ 'click', @@ -84,6 +85,7 @@ const clickFields = { ...selectorSnapshotFields(), ...repeatedFields(), ...postActionObservationFields('click'), + ...targetReadinessFields('click'), }; const pressFields = { @@ -91,6 +93,7 @@ const pressFields = { ...selectorSnapshotFields(), ...repeatedFields(), ...postActionObservationFields('press'), + ...targetReadinessFields('press'), }; const fillFields = { @@ -113,6 +116,7 @@ const longPressFields = { durationMs: integerField('Long press duration in milliseconds.', { min: 0 }), ...selectorSnapshotFields(), ...postActionObservationFields('longpress'), + ...targetReadinessFields('longpress'), }; const hoverFields = { diff --git a/src/commands/target-readiness-grammar.test.ts b/src/commands/target-readiness-grammar.test.ts new file mode 100644 index 0000000000..1756340a26 --- /dev/null +++ b/src/commands/target-readiness-grammar.test.ts @@ -0,0 +1,41 @@ +import assert from 'node:assert/strict'; +import { test } from 'vitest'; +import { AppError } from '@agent-device/kernel/errors'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import { findCommandMetadata, listCommandMetadata } from './command-metadata.ts'; + +const SELECTOR_TARGET = { kind: 'selector', selector: 'label=Continue' }; + +function metadataFor(name: string) { + const metadata = findCommandMetadata(name); + assert.ok(metadata, `expected command metadata for ${name}`); + return metadata; +} + +test('a command advertises readinessTimeoutMs exactly when its descriptor declares the budget', () => { + for (const metadata of listCommandMetadata()) { + const properties = metadata.inputSchema.properties ?? {}; + assert.equal( + 'readinessTimeoutMs' in properties, + commandAcceptsReadinessBudget(metadata.name), + metadata.name, + ); + } +}); + +test('press reads readinessTimeoutMs', () => { + const input = metadataFor('press').readInput({ + target: SELECTOR_TARGET, + readinessTimeoutMs: 2_000, + }) as { readinessTimeoutMs?: number }; + assert.equal(input.readinessTimeoutMs, 2_000); +}); + +test('fill refuses readinessTimeoutMs with an input error, and reads without it', () => { + const fill = metadataFor('fill'); + assert.doesNotThrow(() => fill.readInput({ target: SELECTOR_TARGET, text: 'hi' })); + assert.throws( + () => fill.readInput({ target: SELECTOR_TARGET, text: 'hi', readinessTimeoutMs: 2_000 }), + (error: unknown) => error instanceof AppError && error.code === 'INVALID_ARGS', + ); +}); diff --git a/src/commands/target-readiness-grammar.ts b/src/commands/target-readiness-grammar.ts new file mode 100644 index 0000000000..90e6e70fbb --- /dev/null +++ b/src/commands/target-readiness-grammar.ts @@ -0,0 +1,26 @@ +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import { integerField, operatorField } from './command-input.ts'; + +/** + * The input field a command's `targetReadiness: 'budgeted'` descriptor trait entitles it to. Only + * those commands declare `readinessTimeoutMs`; the common input reader refuses the key for every + * command whose fields do not (`common-input-fields.ts`). Fails closed at module load for a command + * without the trait, so a field map cannot advertise a budget its runtime never polls under. + */ +export function targetReadinessFields(command: string) { + if (!commandAcceptsReadinessBudget(command)) { + throw new Error(`${command} does not declare targetReadiness: 'budgeted'`); + } + return { + readinessTimeoutMs: operatorField( + integerField( + "Operator-only: how long the command may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", + { min: 1 }, + ), + { + operatorPath: + 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', + }, + ), + }; +} diff --git a/src/mcp/__tests__/command-tools-operator-inputs.test.ts b/src/mcp/__tests__/command-tools-operator-inputs.test.ts index 01f0228b27..91b2633b0d 100644 --- a/src/mcp/__tests__/command-tools-operator-inputs.test.ts +++ b/src/mcp/__tests__/command-tools-operator-inputs.test.ts @@ -24,7 +24,6 @@ const OPERATOR_OWNED_KEYS = [ 'iosXctestrunFile', 'iosXctestDerivedDataPath', 'iosXctestEnvDir', - 'readinessTimeoutMs', ] as const; // The rest of the shared common input: keys a model may write, because they @@ -87,6 +86,33 @@ test('MCP refuses every explicit operator-owned argument with guidance', async ( assert.deepEqual(calls, [], 'a refused operator input must never reach the command route'); }); +test('MCP neither advertises nor admits the operator-only readiness budget', async () => { + for (const tool of listCommandTools()) { + const properties = tool.inputSchema.properties ?? {}; + assert.equal('readinessTimeoutMs' in properties, false, `${tool.name} advertises it`); + } + const calls: unknown[] = []; + const executor = createCommandToolExecutor({ + createClient: () => ({}) as AgentDeviceClient, + runCommand: async (_client, name, input) => { + calls.push({ name, input }); + return {}; + }, + }); + + const result = await executor.execute('press', { + target: { kind: 'selector', selector: 'label=Continue' }, + readinessTimeoutMs: 2_000, + }); + + assert.equal(result.isError, true); + assert.match( + result.content[0]?.text ?? '', + /readinessTimeoutMs is not accepted as a tool argument/, + ); + assert.deepEqual(calls, [], 'a refused operator input must never reach the command route'); +}); + test('MCP refuses an explicit daemonAuthToken argument with env guidance', async () => { const calls: unknown[] = []; const executor = createCommandToolExecutor({ diff --git a/test/integration/provider-scenarios/press-target-readiness.test.ts b/test/integration/provider-scenarios/press-target-readiness.test.ts index a9f6b6e4f8..d98e43766a 100644 --- a/test/integration/provider-scenarios/press-target-readiness.test.ts +++ b/test/integration/provider-scenarios/press-target-readiness.test.ts @@ -12,8 +12,8 @@ import { createProviderTranscript, type ProviderScenarioProviderEntry } from './ // promotedTarget readiness (press/click/longpress poll for the target to exist and become // actionable before refusing): end-to-end proof through the real daemon stack, not the plain -// runtime harness — this is the only suite that exercises captureSnapshot's `forceFresh` threading -// through the daemon's selector-runtime-backend cache decision. +// runtime harness. The press route captures through the interaction backend, which keeps no +// selector capture cache, so every poll reaches the runner transcript below. const APP = 'com.example.app'; const DEVICE_ID = PROVIDER_SCENARIO_IOS_SIMULATOR.id; @@ -102,11 +102,11 @@ async function withPressReadinessDaemon( test('press waits for a selector missing on the first two captures, then taps once it appears', async () => { await withPressReadinessDaemon( [ - // Poll 1 (not yet forceFresh): interactive capture, then the interactive->full fallback — - // both miss because the button is not in the tree yet. + // Poll 1: interactive capture, then the interactive->full fallback — both miss because the + // button is not in the tree yet. snapshotEntry(APPLICATION_ONLY_NODES), snapshotEntry(APPLICATION_ONLY_NODES), - // Poll 2 onward (forceFresh: bypasses the selector capture cache): the button has appeared. + // Poll 2 onward: the button has appeared. repeatSnapshotEntry(CONTINUE_BUTTON_NODES), tapEntry(200, 322), ], From 9890d3bd7087797d866954887683fe487ad181b9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:03:45 +0200 Subject: [PATCH 12/23] fix(replay): wait for a late target at the pre-dispatch gate and report readiness A recorded press/click/longpress step carries target evidence, and replay classified that target from one capture before dispatch. A target that had not rendered yet diverged as a selector miss before the dispatch's own readiness wait could run, so the replay default budget never applied to annotated steps. The engine now asks the port to observeTarget, which captures and classifies in one capability. For a step whose dispatch would wait (a readiness-budgeted command resolving a selector), the port re-captures under the same schedule while the recorded target is a selector miss; any other classification or an unavailable capture ends the wait. A target that never appears diverges as before. The divergence capture takes no per-capture signal, so the wait declares captureDeadline 'none'. ReplayDaemonDependencies gains an optional clock the wait paces itself by. REPLAY_DIVERGENCE now carries the cause's readiness detail at error.details.readiness. --- packages/ad-replay/src/index.ts | 3 +- .../src/internal/__tests__/step-loop.test.ts | 16 +-- .../src/internal/runtime-port-types.ts | 36 ++--- .../ad-replay/src/internal/verify-dispatch.ts | 32 ++--- ...on-replay-runtime-failure-response.test.ts | 24 ++++ .../src/daemon-port/command-types.ts | 3 + .../session-replay-action-runtime.ts | 16 +++ .../session-replay-runtime-engine-adapter.ts | 129 ++++++++++++------ ...session-replay-runtime-failure-response.ts | 1 + .../session-replay-target-verification.ts | 4 +- .../replay-runtime/replay-command-fixture.ts | 14 +- ...replay-target-verification-runtime.test.ts | 37 +++++ 12 files changed, 216 insertions(+), 99 deletions(-) diff --git a/packages/ad-replay/src/index.ts b/packages/ad-replay/src/index.ts index 4a4df28df6..2f533795f9 100644 --- a/packages/ad-replay/src/index.ts +++ b/packages/ad-replay/src/index.ts @@ -60,7 +60,7 @@ * and the classification/guard/binding-evidence/verification-routing shapes * `verifyAndDispatchStep` exchanges with the daemon's `AdReplayStepRuntime` * implementation (`AdReplayVerificationEntry`, `AdReplayTargetClassification`, - * `AdReplayTargetBindingEvidence`, `AdReplayDispatchGuard`, + * `AdReplayTargetObservation`, `AdReplayTargetBindingEvidence`, `AdReplayDispatchGuard`, * `AdReplayDispatchOutcome`) ARE named here: the * daemon builds/reads real values of these shapes directly now (routing in * `session-replay-target-verification.ts`, wire-narrowing in @@ -93,6 +93,7 @@ export type { AdReplayStepRuntime, AdReplayTargetBindingEvidence, AdReplayTargetClassification, + AdReplayTargetObservation, AdReplayVarSources, AdReplayVerificationEntry, } from './internal/runtime-port-types.ts'; diff --git a/packages/ad-replay/src/internal/__tests__/step-loop.test.ts b/packages/ad-replay/src/internal/__tests__/step-loop.test.ts index 0cd7ee6abf..bb2ce6a98f 100644 --- a/packages/ad-replay/src/internal/__tests__/step-loop.test.ts +++ b/packages/ad-replay/src/internal/__tests__/step-loop.test.ts @@ -63,11 +63,8 @@ function createFakeRuntime(params: { isRepairArmed?: () => boolean } = {}): { let armCount = 0; const runtime: AdReplayStepRuntime = { beginTargetVerification: () => ({ kind: 'inactive' }), - captureObservation: async () => { - throw new Error('captureObservation: not used by this fixture (no targetEvidence)'); - }, - classifyTarget: () => { - throw new Error('classifyTarget: not used by this fixture (no targetEvidence)'); + observeTarget: async () => { + throw new Error('observeTarget: not used by this fixture (no targetEvidence)'); }, async dispatchStep(dispatchedAction, _resolvedAction, _index, artifactPaths) { dispatched.push(dispatchedAction.command); @@ -175,13 +172,8 @@ test("a post-dispatch target-binding mismatch reports the pre-step artifact snap // called for it — routed to the #1349 deferred-landmark path, which // dispatches with a guard WITHOUT any capture/classify round trip. beginTargetVerification: () => ({ kind: 'post-resolution', isSelectorWait: true }), - captureObservation: async () => { - throw new Error( - 'captureObservation: not used — deferred-landmark skips straight to dispatch', - ); - }, - classifyTarget: () => { - throw new Error('classifyTarget: not used — deferred-landmark skips straight to dispatch'); + observeTarget: async () => { + throw new Error('observeTarget: not used — deferred-landmark skips straight to dispatch'); }, async dispatchStep(dispatchedAction, _resolvedAction, _index, artifactPaths, _guard) { if (dispatchedAction.command === 'open') { diff --git a/packages/ad-replay/src/internal/runtime-port-types.ts b/packages/ad-replay/src/internal/runtime-port-types.ts index 3d21be54bb..502c8ce457 100644 --- a/packages/ad-replay/src/internal/runtime-port-types.ts +++ b/packages/ad-replay/src/internal/runtime-port-types.ts @@ -97,9 +97,9 @@ export type AdReplayVerifiedTargetGuard = Readonly<{ matchCount: number; }>; -/** `captureObservation`'s neutral result: nodes for classification, or why a capture was not available. */ -export type AdReplayObservation = Readonly< - | { readonly state: 'available'; readonly nodes: readonly SnapshotNode[] } +/** `observeTarget`'s neutral result: the recorded target's classification, or why no capture was available. */ +export type AdReplayTargetObservation = Readonly< + | { readonly state: 'classified'; readonly classification: AdReplayTargetClassification } | { readonly state: 'unavailable'; readonly reason: string; readonly hint?: string } >; @@ -120,7 +120,7 @@ export type AdReplayVerificationEntry = Readonly< } >; -/** `classifyTarget`'s result: a verified guard, or the divergence evidence a target-binding failure reports. */ +/** The recorded target's classification: a verified guard, or the divergence evidence a target-binding failure reports. */ export type AdReplayTargetClassification = Readonly< | { readonly verified: true; readonly guard: AdReplayVerifiedTargetGuard } | Readonly<{ @@ -220,26 +220,20 @@ export type AdReplayStepRuntime = Readonly<{ targetRole?: 'source' | 'destination', ): AdReplayVerificationEntry; /** - * Captures a fresh snapshot for classification or for a divergence's - * `screen` — daemon authority (`SessionStore`, the capture pipeline, the - * #1385 launch-race retry). + * Captures a fresh snapshot (with the #1385 launch-race retry) and resolves + * the recorded target against it using the SAME lookup/matching a real + * dispatch would — daemon authority (`SessionStore`, the capture pipeline, + * tree helpers and the selectors package engine). A step whose dispatch + * would wait for its target under a readiness budget re-captures while the + * target does not match yet, under that same budget, so this gate never + * refuses a target the dispatch itself would have waited for. The last + * capture is the divergence's `screen`. */ - captureObservation( - action: SessionAction, - index: number, - options: { retryLaunchRace: boolean }, - ): Promise; - /** - * Resolves the recorded target against `nodes` using the SAME - * lookup/matching a real dispatch would — daemon authority (tree helpers and - * the selectors package engine). - */ - classifyTarget(params: { + observeTarget(params: { action: SessionAction; index: number; token: string; - nodes: readonly SnapshotNode[]; - }): AdReplayTargetClassification; + }): Promise; /** * Dispatches the action, optionally carrying a pre-action identity guard, * and detects the guard-mismatch / wait-landmark-mismatch post-resolution @@ -279,7 +273,7 @@ export type AdReplayStepRuntime = Readonly<{ ): Promise; /** * Builds a target-binding divergence from `evidence`, reusing the LAST - * `captureObservation` result for its `screen` (the pre-dispatch capture + * `observeTarget` capture for its `screen` (the pre-dispatch capture * and classification/capture-failure evidence share one capture) — * daemon authority. `artifactPaths` is the pre-step snapshot, as above; * `scrubVars` as above. diff --git a/packages/ad-replay/src/internal/verify-dispatch.ts b/packages/ad-replay/src/internal/verify-dispatch.ts index 99e6ad7b75..e82303bc5d 100644 --- a/packages/ad-replay/src/internal/verify-dispatch.ts +++ b/packages/ad-replay/src/internal/verify-dispatch.ts @@ -8,10 +8,10 @@ import { import type { AdReplayDispatchGuard, AdReplayDispatchOutcome, - AdReplayObservation, AdReplayScrubValue, AdReplayStepOutcome, AdReplayStepRuntime, + AdReplayTargetObservation, AdReplayVerifiedTargetGuard, } from './runtime-port-types.ts'; @@ -25,8 +25,8 @@ import type { * those four functions live here — engine-private, never re-exported by the * façade — and the daemon side is the narrow `AdReplayStepRuntime` * capabilities this function drives: routing (`beginTargetVerification`), - * capture (`captureObservation`), classification (`classifyTarget`), - * dispatch (`dispatchStep`), and wire-building the resulting divergence + * capture-and-classification (`observeTarget`), dispatch (`dispatchStep`), + * and wire-building the resulting divergence * (`buildRecordedUnverifiableFailure`, `buildTargetBindingFailure`, * `buildPostDispatchTargetBindingFailure`). `./step-loop.ts`'s `runAdReplay` * is this module's one caller. @@ -121,11 +121,8 @@ export async function verifyAndDispatchStep( } const token = preDispatchPlan.token; - // #1385: this is the pre-dispatch gate a step right after `open --relaunch` - // can race — the app may still be launching/mounting when this capture - // lands. Bounded retry rides out that transition (`retryLaunchRace`). - const observation = await runtime.captureObservation(action, index, { retryLaunchRace: true }); - if (observation.state !== 'available') { + const observation = await runtime.observeTarget({ action, index, token }); + if (observation.state !== 'classified') { return { status: 'failed', failure: await runtime.buildTargetBindingFailure( @@ -138,7 +135,7 @@ export async function verifyAndDispatchStep( }; } - const classification = runtime.classifyTarget({ action, index, token, nodes: observation.nodes }); + const { classification } = observation; if (classification.verified) { return dispatchWithGuard(runtime, scrubVars, action, resolvedAction, index, artifactPaths, { kind: 'target', @@ -213,10 +210,12 @@ async function verifyAndDispatchMultiTargetStep( }; } - const observation = await runtime.captureObservation(endpointAction, index, { - retryLaunchRace: true, + const observation = await runtime.observeTarget({ + action: endpointAction, + index, + token: plan.token, }); - if (observation.state !== 'available') { + if (observation.state !== 'classified') { return { status: 'failed', failure: await runtime.buildTargetBindingFailure( @@ -229,12 +228,7 @@ async function verifyAndDispatchMultiTargetStep( }; } - const classification = runtime.classifyTarget({ - action: endpointAction, - index, - token: plan.token, - nodes: observation.nodes, - }); + const { classification } = observation; if (!classification.verified) { return { status: 'failed', @@ -268,7 +262,7 @@ async function verifyAndDispatchMultiTargetStep( } function captureUnavailableEvidence( - observation: Extract, + observation: Extract, ) { return { kind: 'identity-unverifiable' as const, diff --git a/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts b/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts index 52651408fb..c6834d9bcb 100644 --- a/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts +++ b/packages/replay-port/src/daemon-port/__tests__/session-replay-runtime-failure-response.test.ts @@ -46,3 +46,27 @@ test('native replay failure metadata keeps machine fields and daemon-owned paths expect(response.error.details).toHaveProperty('reason'); expect(response.error.details).not.toHaveProperty('reas'); }); + +test('a replay divergence carries the readiness evidence of an exhausted target wait', () => { + const readiness = { polls: 11, waitedMs: 2_004, end: 'expired' }; + const response = buildReplayDivergenceFailureResponseFromDescriptor({ + error: { + code: 'COMMAND_FAILED', + message: 'Selector did not match', + details: { reason: 'selector_not_found', readiness }, + }, + actionLabel: 'press label=Continue', + action: 'press', + positionals: ['label=Continue'], + step: 3, + replayPath: '/tmp/flows/checkout.ad', + artifactPaths: [], + divergence: {}, + scrubVars: [], + }); + + expect(response.ok).toBe(false); + if (response.ok) return; + expect(response.error.code).toBe('REPLAY_DIVERGENCE'); + expect(response.error.details).toMatchObject({ reason: 'selector_not_found', readiness }); +}); diff --git a/packages/replay-port/src/daemon-port/command-types.ts b/packages/replay-port/src/daemon-port/command-types.ts index 5b765daa4f..9082fc1be5 100644 --- a/packages/replay-port/src/daemon-port/command-types.ts +++ b/packages/replay-port/src/daemon-port/command-types.ts @@ -10,6 +10,7 @@ import type { ReplayTestAttemptStepSink } from '@agent-device/replay-test'; import type { DaemonResponse, SessionRuntimeHints } from '@agent-device/kernel/contracts'; import type { DeviceInfo } from '@agent-device/kernel/device'; import type { SnapshotState } from '@agent-device/kernel/snapshot'; +import type { ObservationClock } from '@agent-device/capture-kit/observe-until'; /** * The slice of the daemon's live session record replay reads. The daemon passes its full @@ -127,6 +128,8 @@ export type ReplayInvoke = (request: ReplayDispatchRequest) => Promise SessionScope; + /** The time source the pre-dispatch target-readiness wait paces itself by. Absent: wall clock. */ + clock?: ObservationClock; }>; export type ReplayCommand = Readonly<{ diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index ce2d4c03cc..bbabaa0ee1 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -7,6 +7,7 @@ import type { } from '@agent-device/replay-port/command-types'; import { mergeParentFlags } from '@agent-device/command-registry/batch'; import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import { AppError, normalizeError } from '@agent-device/kernel/errors'; import { gesturePayloadFromPositionals, @@ -222,6 +223,21 @@ function readResponseTiming(data: unknown): Record | undefined */ const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; +/** + * The readiness schedule a replayed step's dispatch polls its target under, or undefined when the + * step's command does not wait for its target. The pre-dispatch target gate polls under the same + * schedule, so neither refuses a target the other would still wait for. + */ +export function replayStepReadinessSchedule( + parentFlags: CommandFlags | undefined, + action: SessionAction, +): { intervalMs: number; budgetMs: number } | undefined { + const { readinessTimeoutMs } = buildReplayActionFlags(parentFlags, action.flags, action.command); + if (readinessTimeoutMs === undefined || readinessTimeoutMs <= 0) return undefined; + const poll = SELECTOR_PIPELINE_POLICIES.promotedTarget.poll; + return { intervalMs: poll.intervalMs, budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs) }; +} + function buildReplayActionFlags( parentFlags: CommandFlags | undefined, actionFlags: SessionAction['flags'] | undefined, diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts index 0cfbf22d3b..59f43b838c 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts @@ -17,8 +17,16 @@ import { import type { SnapshotTimingSample } from '@agent-device/contracts/capture'; import { withReplayFailureDiagnostics } from './session-replay-runtime-failure.ts'; -import { invokeReplayAction } from './session-replay-action-runtime.ts'; -import type { AdReplayStepFailure, AdReplayStepRuntime } from '@agent-device/ad-replay'; +import { + invokeReplayAction, + replayStepReadinessSchedule, +} from './session-replay-action-runtime.ts'; +import type { + AdReplayStepFailure, + AdReplayStepRuntime, + AdReplayTargetObservation, +} from '@agent-device/ad-replay'; +import { observeUntil } from '@agent-device/capture-kit/observe-until'; import { collectReplayActionArtifactPaths } from '@agent-device/replay-port/session-replay-runtime-artifacts'; import { applyReplayDispatchGuard, @@ -79,7 +87,7 @@ import type { ReplayTestAttemptStepSink } from '@agent-device/replay-test'; * engine itself never touches it. * * `lastObservation` is the analogous side-map for `buildTargetBindingFailure` - * — it reuses the SAME capture `captureObservation` just took (for its + * — it reuses the SAME capture `observeTarget` last took (for its * `screen`), mirroring the pre-R3 code's single-capture-serves-both-paths * invariant instead of taking a second, possibly-different snapshot. * @@ -88,14 +96,14 @@ import type { ReplayTestAttemptStepSink } from '@agent-device/replay-test'; * `createAdReplayStepRuntime` call covers every step), so an un-reset * `lastObservation` would silently carry a PREVIOUS step's capture into a * step that somehow reached `buildTargetBindingFailure` without its own - * `captureObservation` call first — the `?? { reason: 'observation-missing' + * `observeTarget` call first — the `?? { reason: 'observation-missing' * }` fallback below exists to name that condition, but could never actually * fire for it; it would instead attach a stale, wrong-step screen. `armStep` * runs exactly once per step, before any of this step's capabilities do — * clearing `lastObservation` there makes the fallback message correct for * ANY future call ordering, not just the current one where every * `buildTargetBindingFailure` call site happens to be preceded by this same - * step's own `captureObservation`. + * step's own `observeTarget`. */ export function createAdReplayStepRuntime(params: { ctx: ReplayStepContext; @@ -157,6 +165,31 @@ export function createAdReplayStepRuntime(params: { ); }; + // #1385: the pre-dispatch gate a step right after `open --relaunch` can + // race — the app may still be launching/mounting when this capture lands, + // producing a transient `capture-failed` / `sparse-snapshot` verdict that is + // not a real divergence. Bounded retry (`retryLaunchRace`) rides out that + // transition instead of failing closed on the first unlucky capture. + const captureTargetObservation = async ( + action: SessionAction, + ): Promise => { + const session = ctx.observationStore.get(); + if (!session) { + return { + state: 'unavailable', + reason: 'no-session', + hint: 'The session closed before a screen could be captured to verify the recorded target.', + }; + } + return await captureDivergenceObservation({ + session, + observationStore: ctx.observationStore, + logPath: ctx.logPath, + action, + retryLaunchRace: true, + }); + }; + const runtime: AdReplayStepRuntime = { beginTargetVerification(action, resolvedAction, _index, targetRole) { return resolveTargetVerificationEntry({ @@ -167,46 +200,47 @@ export function createAdReplayStepRuntime(params: { }); }, - async captureObservation(action, _index, options) { - const session = ctx.observationStore.get(); - // #1385: this is the pre-dispatch gate a step right after `open - // --relaunch` can race — the app may still be launching/mounting when - // this capture lands, producing a transient `capture-failed` / - // `sparse-snapshot` verdict that is not a real divergence. Bounded - // retry (`retryLaunchRace`, engine-driven) rides out that transition - // instead of failing closed on the first unlucky capture. - const observation: DivergenceObservation = session - ? await captureDivergenceObservation({ - session, - observationStore: ctx.observationStore, - logPath: ctx.logPath, + async observeTarget({ action, token }) { + const observeOnce = async (): Promise => { + const observation = await captureTargetObservation(action); + lastObservation = observation; + if (observation.state !== 'available') { + return { state: 'unavailable', reason: observation.reason, hint: observation.hint }; + } + const session = ctx.sessionStore.get(); + return { + state: 'classified', + classification: classifyPreDispatchTarget({ + // An available capture implies an active session, and the engine + // only calls this for an action carrying `targetEvidence`. + recorded: action.targetEvidence!, + token, action, - retryLaunchRace: options.retryLaunchRace, - }) - : { - state: 'unavailable', - reason: 'no-session', - hint: 'The session closed before a screen could be captured to verify the recorded target.', - }; - lastObservation = observation; - return observation.state === 'available' - ? { state: 'available', nodes: observation.nodes } - : { state: 'unavailable', reason: observation.reason, hint: observation.hint }; - }, - - classifyTarget({ action, token, nodes }) { - const session = ctx.sessionStore.get(); - return classifyPreDispatchTarget({ - // Only ever called right after a successful `captureObservation`, - // which itself only reaches `state: 'available'` when a session is - // active — `action.targetEvidence`/`session` are always defined here - // in practice. - recorded: action.targetEvidence!, - token, - action, - nodes: [...nodes], - platform: session!.device.platform, + nodes: [...observation.nodes], + platform: session!.device.platform, + }), + }; + }; + // The gate waits only where the dispatch would: a readiness-budgeted + // command resolving a selector (a `@ref` resolves without a wait). + const readiness = token.startsWith('@') + ? undefined + : replayStepReadinessSchedule(ctx.replayReq.flags, action); + if (!readiness) return await observeOnce(); + const observed = await observeUntil({ + capture: observeOnce, + verdict: (latest) => + isTargetNotRenderedYet(latest) ? { kind: 'continue' } : { kind: 'done', result: latest }, + // The divergence capture takes no per-capture signal, so a late + // capture is judged rather than cancelled. + schedule: { ...readiness, captureDeadline: 'none' }, + ...(ctx.signal ? { signal: ctx.signal } : {}), + ...(ctx.dependencies.clock ? { clock: ctx.dependencies.clock } : {}), + phase: 'replay_target_readiness', }); + if (observed.kind === 'done') return observed.result; + if (observed.last !== undefined) return observed.last; + throw observed.kind === 'expired' ? observed.lastError : observed.error; }, // `_stepArtifactPaths` (the pre-step snapshot) is unused here — dispatch @@ -453,3 +487,12 @@ function readSessionSnapshotSamplesSince( ): SnapshotTimingSample[] { return sessionStore.get()?.snapshotDiagnostics?.samples.slice(start) ?? []; } + +/** A selector miss on a readiness-budgeted step: the target may still be rendering. */ +function isTargetNotRenderedYet(observation: AdReplayTargetObservation): boolean { + return ( + observation.state === 'classified' && + !observation.classification.verified && + observation.classification.kind === 'selector-miss' + ); +} diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts index fc705aebe2..d62d42f93d 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-failure-response.ts @@ -122,6 +122,7 @@ function readStringDetail( } const SAFE_CAUSE_DETAIL_KEYS = [ + 'readiness', 'reason', 'recovery', 'retriable', diff --git a/packages/replay-port/src/daemon-port/session-replay-target-verification.ts b/packages/replay-port/src/daemon-port/session-replay-target-verification.ts index 0bf3feb9a4..2da5832542 100644 --- a/packages/replay-port/src/daemon-port/session-replay-target-verification.ts +++ b/packages/replay-port/src/daemon-port/session-replay-target-verification.ts @@ -73,7 +73,7 @@ import { extractReplayTargetToken, readRefLabel } from '@agent-device/replay-por // // - `resolveTargetVerificationEntry` — routing (registry lookup, session // read, wait-form parse, token extraction) for `beginTargetVerification`. -// - `classifyPreDispatchTarget` — tree matching for `classifyTarget`. +// - `classifyPreDispatchTarget` — tree matching for `observeTarget`. // - `buildRecordedUnverifiableFailureResponse` / // `buildTargetBindingFailureResponse` / // `buildPostDispatchTargetBindingFailureResponse` — capture + wire-shaping @@ -409,7 +409,7 @@ export function resolveTargetVerificationEntry(params: { } // --------------------------------------------------------------------------- -// `classifyTarget`: resolves the recorded target against an already-captured +// `observeTarget`'s classification: resolves the recorded target against an already-captured // tree using the SAME lookup/matching a real dispatch would. // // #1555 structural-quality review ("unify on the engine's types"): returns diff --git a/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts b/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts index 453d3ccd02..5cddac764e 100644 --- a/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts +++ b/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts @@ -11,6 +11,7 @@ import { splitReplayCommandRequest, } from '@agent-device/replay-port/replay-dispatch-envelope'; import type { ReplayCommand } from '@agent-device/replay-port/command-types'; +import type { ObservationClock } from '@agent-device/capture-kit/observe-until'; export type ReplayCommandTestInput = Readonly<{ req: DaemonRequest; @@ -33,7 +34,7 @@ export function replayCommandForTest(params: ReplayCommandTestInput): ReplayComm ...splitReplayCommandRequest(req), session: createReplaySession(sessionName, logPath, sessionStore), invoke: replayInvokeOverDispatch(invoke, req), - dependencies: replayDaemonDependencies, + dependencies: { ...replayDaemonDependencies, clock: instantReplayClock() }, ...(tracePath === undefined ? {} : { tracePath }), ...(onStep === undefined ? {} : { onStep }), }; @@ -42,3 +43,14 @@ export function replayCommandForTest(params: ReplayCommandTestInput): ReplayComm export function runReplayForTest(params: ReplayCommandTestInput): Promise { return runReplayCommand(replayCommandForTest(params)); } + +/** Target-readiness waits advance this clock instead of sleeping, so a unit replay spends no wall time. */ +function instantReplayClock(): ObservationClock { + let nowMs = Date.now(); + return { + now: () => nowMs, + sleep: async (ms) => { + nowMs += ms; + }, + }; +} diff --git a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts index ca97616a8f..8a4fe8afe2 100644 --- a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts +++ b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts @@ -177,6 +177,43 @@ test('a selector-miss divergence blocks dispatch and never sends the action', as expect(targetBinding.matchCount).toBe(0); expect(targetBinding.observed).toBeUndefined(); expect(targetBinding.recorded).toEqual({ id: 'save', role: 'button', label: 'Save' }); + // click waits for its target, so the gate re-captured before refusing. + expect(mockDispatchCommand.mock.calls.length).toBeGreaterThan(1); +}); + +test('an annotated click whose target renders on the second capture waits for it and dispatches', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-late-render-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + + mockDispatchCommand.mockResolvedValueOnce(emptyCapture()).mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay(); + + expect(response.ok).toBe(true); + expect(scene.invoked.map((req) => req.command)).toEqual(['click']); + expect(scene.invoked[0]?.internal?.replayTargetGuard).toMatchObject({ + identity: { id: 'save', role: 'button', label: 'Save' }, + }); + expect(mockDispatchCommand).toHaveBeenCalledTimes(2); +}); + +test('an annotated step whose command does not wait for its target refuses a selector miss on one capture', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-no-wait-', [ + SAVE_ANNOTATION, + 'fill id="save" "hello"', + ]); + + mockDispatchCommand.mockResolvedValueOnce(emptyCapture()).mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay(); + + expect(scene.invoked.length).toBe(0); + expect(response.ok).toBe(false); + if (response.ok) return; + const divergence = response.error.details?.divergence as Record; + expect(divergence.kind).toBe('selector-miss'); }); test('an identity-mismatch divergence reports matchCount and an observed identity', async () => { From 661c3f8f22936cee2df590e32f231853469b99d1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:03:47 +0200 Subject: [PATCH 13/23] docs(interaction): describe where readiness evidence appears The client API page says the command performs the requested interaction after the wait, and names the failures that carry readiness evidence and those that do not. The replay page says how a missing target surfaces for annotated and unannotated steps. --- website/docs/docs/client-api.md | 4 ++-- website/docs/docs/replay-e2e.md | 25 ++++++++++++++----------- 2 files changed, 16 insertions(+), 13 deletions(-) diff --git a/website/docs/docs/client-api.md b/website/docs/docs/client-api.md index 75da1673a8..08cd4d31de 100644 --- a/website/docs/docs/client-api.md +++ b/website/docs/docs/client-api.md @@ -288,7 +288,7 @@ await client.command.fold({ `fold` accepts either `pose` or `keyframes`. Keyframes use linear interpolation at roughly 60 updates per second; repeat an angle to hold it. Timestamps must start at zero and increase strictly, with 2–64 frames and a final timestamp no greater than 60,000ms. Angles must be finite and between 0° and 180°. The final timestamp bounds motion, excluding helper preparation and final hinge verification. A custom final angle is verified within 0.5°; interior angles must also settle. Cancellation stops the motion at its current angle. Re-snapshot afterwards, including after interrupted motion. -`press`, `click`, and `longpress` take `readinessTimeoutMs`. With it, the command waits up to that many milliseconds for a target that is not on screen yet, then taps it once. Without it, the command looks once and fails at once, which is the right choice for an agent that most often misses because the selector is wrong. Use it in scripted flows, where a step can land a render early: +`press`, `click`, and `longpress` take `readinessTimeoutMs`. With it, the command waits up to that many milliseconds for a target that is not on screen yet, then performs the requested interaction. Without it, the command looks once and fails at once, which is the right choice for an agent that most often misses because the selector is wrong. Use it in scripted flows, where a step can land a render early: ```ts await client.interactions.press({ @@ -297,7 +297,7 @@ await client.interactions.press({ }); ``` -The wait is capped at 2 seconds. It covers only a target that has not appeared; a covered, off-screen, or ambiguous target fails at once. A failure after the wait carries `error.details.readiness` with `waitedMs`, `polls`, and `end` (`expired`, `stalled`, or `sparse`). `readinessTimeoutMs` is not an MCP tool argument and has no CLI flag. +The wait is capped at 2 seconds and covers only a target that has not appeared. When the target is still missing after the wait, the error carries `error.details.readiness` with `waitedMs`, `polls`, and `end` (`expired` or `stalled`). A capture that shows an empty accessibility tree ends the wait at once with `capture_sparse` and `readiness.end: sparse`. A covered, off-screen, or ambiguous target fails at once, and a screen that stays unreadable for the whole wait fails with its own error; neither carries `readiness`. `readinessTimeoutMs` is not an MCP tool argument and has no CLI flag. Vega OS client support is currently VVD-only and covers device discovery, app open/close, `back`, `home`, and `tvRemote`. Physical Fire TV, capture, selector, install, logging, and performance methods report unsupported for Vega targets. diff --git a/website/docs/docs/replay-e2e.md b/website/docs/docs/replay-e2e.md index e96da2ee99..f423f6bd4e 100644 --- a/website/docs/docs/replay-e2e.md +++ b/website/docs/docs/replay-e2e.md @@ -62,11 +62,13 @@ agent-device replay ~/.agent-device/sessions/e2e-2026-02-09T12-00-00-000Z.ad --s they fail. A step recorded against a screen that was still loading passes on replay once the target shows up. The wait covers only a target that is not on screen yet: a target that is covered, off-screen, or matched by more than one element fails at once, as it does live. -- When the target never appears, the step fails with the same `selector_not_found` error as a live - command, and `error.details.readiness` says how long it waited and how many times it looked - (`waitedMs`, `polls`, `end`). If the app stops exposing an accessibility tree during the wait, - the step fails with `error.details.reason: capture_sparse` instead; take a snapshot to see - where the app is. +- When the target never appears, replay stops with `REPLAY_DIVERGENCE`. For a step recorded with + a target annotation (the `# agent-device:target-v1` line above it), `error.details.divergence.kind` + is `selector-miss` and the step is never sent. For a step without an annotation, + `error.details.reason` is `selector_not_found`, as for a live command, and + `error.details.readiness` says how long the step waited and how many times it looked (`waitedMs`, + `polls`, `end`). If the app shows an empty accessibility tree during that wait, + `error.details.reason` is `capture_sparse` instead; take a snapshot to see where the app is. ## Run Maestro compatibility flows @@ -393,11 +395,12 @@ Passing `--plan-digest` that no longer matches the current script — because yo - Leave the replay plan unchanged, repair app state so the reported failed step can be retried, then use its `--from`/`--plan-digest`. Resume starts at `--from`; it does not skip that step. - Replay file parse error: - Validate quoting in `.ad` lines (unclosed quotes are rejected). -- A `press` or `click` step fails with `selector_not_found`, but the element is on the screenshot: - - Check `error.details.readiness.end`. `expired` means the element was not in the accessibility - tree for the whole 2-second wait: the selector is wrong for this build, or the element is not - exposed to accessibility. `sparse` means the app showed an empty tree: the screen was - mid-transition or the app had left. Add a `wait` step for a landmark on the new screen before - the press. +- A `press` or `click` step fails because its target was not found, but the element is on the + screenshot: + - A `selector-miss` divergence, or `error.details.readiness.end: expired`, means the element was + not in the accessibility tree for the whole 2-second wait: the selector is wrong for this + build, or the element is not exposed to accessibility. `readiness.end: sparse` means the app + showed an empty tree: the screen was mid-transition or the app had left. Add a `wait` step for + a landmark on the new screen before the press. - Maestro compatibility flow fails on unsupported syntax: - Check [ADR 0015](https://github.com/callstack/agent-device/blob/main/docs/adr/0015-direct-maestro-engine.md). If the missing feature matters to your suite, open a focused issue with a small flow snippet. From a1e3461b7d3a8d3b0db8b6a871339d697ad0882d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:05:45 +0200 Subject: [PATCH 14/23] test(layering): record AdReplayTargetObservation on the ad-replay root facade --- scripts/layering/package-boundaries.test.ts | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/layering/package-boundaries.test.ts b/scripts/layering/package-boundaries.test.ts index 9657a0a870..3f41c6c817 100644 --- a/scripts/layering/package-boundaries.test.ts +++ b/scripts/layering/package-boundaries.test.ts @@ -653,6 +653,7 @@ test('the real tree parses, declares, and passes R11', () => { 'AdReplayStepRuntime', 'AdReplayTargetBindingEvidence', 'AdReplayTargetClassification', + 'AdReplayTargetObservation', 'AdReplayVarSources', 'AdReplayVerificationEntry', 'inspectAdReplay', From 16f15ccc47207e07d911a569206e743072f6fd6b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:17:45 +0200 Subject: [PATCH 15/23] refactor(interaction): drop the readiness forceFresh plumbing; the press capture route has no cache The readiness poll asked for a fresh capture from its second poll on, but no press, click, or longpress capture can be served from the selector capture cache, so the flag and the backend option that carried it are removed and selector-runtime-backend.ts, backend.ts, and backend-snapshot-options.ts return to main. Trace: - selector-readiness.ts pollSelectorReadinessOnce captured through attemptSelectorResolution, which calls captureInteractionSnapshot (interaction-snapshot-capture.ts:35) and so runtime.backend.captureSnapshot. - The daemon press route builds that runtime with createInteractionRuntimeForRoute (src/daemon/interaction/internal/interaction-touch-press.ts:10, used for longPress/click/press at :177-187). - Its backend is createInteractionBackend (src/daemon/interaction/internal/interaction-runtime.ts:126), whose captureSnapshot (:132-139) forwards interactiveOnly, preferredBackend, includeRects, and the signal, and nothing cache-related. - That capture calls captureSnapshotForSession (src/daemon/interaction/index.ts:21), which calls the daemon captureSnapshot (src/daemon/snapshot-capture.ts:77). - captureSnapshot runs captureSnapshotAttempt, which runs captureSnapshotData (snapshot-capture.ts:99) and captures live through the bound capture or captureSnapshotWithInteractor on every call. - The 750 ms cache (SELECTOR_CAPTURE_CACHE_TTL_MS, src/daemon/selector-capture-runtime.ts:24, read at :253 and :280) exists only inside createSelectorBackend (src/daemon/selector-runtime-backend.ts:142). - createSelectorRuntimeForDevice is reached from the selector read and wait routes (selector-runtime.ts, wait-runtime.ts:158), never from interaction-touch-press.ts. The runtime-selector targetReadiness contract scenario keyed its late target on the removed option; it now keys on the capture count. --- src/backend-snapshot-options.ts | 5 +--- src/backend.ts | 6 ----- .../runtime/interaction-snapshot-capture.ts | 7 ------ .../interaction/runtime/resolution.ts | 8 +------ .../interaction/runtime/selector-readiness.ts | 23 ++++--------------- src/daemon/selector-runtime-backend.ts | 19 ++++----------- .../runtime-selector.contract.test.ts | 10 ++++---- 7 files changed, 16 insertions(+), 62 deletions(-) diff --git a/src/backend-snapshot-options.ts b/src/backend-snapshot-options.ts index 04f8564aa4..e60f192310 100644 --- a/src/backend-snapshot-options.ts +++ b/src/backend-snapshot-options.ts @@ -1,10 +1,7 @@ import { snapshotFlagsFromOptions } from '@agent-device/kernel/snapshot'; import type { BackendSnapshotOptions } from './backend.ts'; -type RoutedSnapshotOption = keyof Omit< - BackendSnapshotOptions, - 'includeRects' | 'outPath' | 'forceFresh' ->; +type RoutedSnapshotOption = keyof Omit; /** * Which backend snapshot options route as request flags. Still pinned to diff --git a/src/backend.ts b/src/backend.ts index 57d3d7140b..2533e8f4d1 100644 --- a/src/backend.ts +++ b/src/backend.ts @@ -64,12 +64,6 @@ export type BackendSnapshotOptions = SnapshotOptions & { includeRects?: boolean; includeHiddenContentHints?: boolean; outPath?: string; - /** - * Bypass a cache-serving backend's short-lived stored capture, the way `wait`/`find` already force - * one (`src/daemon/selector-runtime-backend.ts`). A backend without such a cache has nothing to - * bypass and may ignore this. - */ - forceFresh?: boolean; }; export type BackendReadTextResult = { diff --git a/src/commands/interaction/runtime/interaction-snapshot-capture.ts b/src/commands/interaction/runtime/interaction-snapshot-capture.ts index 9f2da3ab25..5ec85592d0 100644 --- a/src/commands/interaction/runtime/interaction-snapshot-capture.ts +++ b/src/commands/interaction/runtime/interaction-snapshot-capture.ts @@ -19,12 +19,6 @@ export async function captureInteractionSnapshot( runtime: AgentDeviceRuntime, options: CommandContext, interactiveOnly: boolean, - /** - * Bypasses the daemon's short-lived selector capture cache, the way `wait`'s capture already does - * (`src/daemon/selector-runtime-backend.ts`). Defaults to false: every caller before the readiness - * poll relied on the cache, and the poll itself only sets this from its second capture on. - */ - forceFresh?: boolean, ): Promise { if (!runtime.backend.captureSnapshot) { throw new AppError('UNSUPPORTED_OPERATION', 'snapshot is not supported by this backend'); @@ -35,7 +29,6 @@ export async function captureInteractionSnapshot( const result = await runtime.backend.captureSnapshot(toBackendContext(runtime, options), { interactiveOnly, includeRects: true, - ...(forceFresh ? { forceFresh: true } : {}), }); const snapshot = result.snapshot ?? diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index 551ede1a8c..c0802f3bf8 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -160,13 +160,7 @@ async function resolveSelectorInteractionTarget( let resolved: SelectorResolution | null; const readinessSchedule = readinessScheduleFor(params.pipeline.poll, params.readinessTimeoutMs); if (!readinessSchedule) { - const attempt = await attemptSelectorResolution( - runtime, - options, - selectorExpression, - params, - false, - ); + const attempt = await attemptSelectorResolution(runtime, options, selectorExpression, params); capture = attempt.capture; resolved = attempt.resolved; if (!resolved || !resolved.node.rect) { diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts index 184013ff89..ce62382999 100644 --- a/src/commands/interaction/runtime/selector-readiness.ts +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -74,14 +74,8 @@ export async function attemptSelectorResolution( options: CommandContext, selectorExpression: string, params: ResolveInteractionTargetParams, - forceFresh: boolean, ): Promise { - let capture = await captureInteractionSnapshot( - runtime, - options, - params.requireInteractive, - forceFresh, - ); + let capture = await captureInteractionSnapshot(runtime, options, params.requireInteractive); let resolved = resolveActionSelector( capture.snapshot.nodes, selectorExpression, @@ -90,7 +84,7 @@ export async function attemptSelectorResolution( ); if ((!resolved || !resolved.node.rect) && params.requireInteractive) { const interactive = capture.snapshot; - capture = await captureInteractionSnapshot(runtime, options, false, forceFresh); + capture = await captureInteractionSnapshot(runtime, options, false); inheritPostGestureOutcome(interactive, capture.snapshot); resolved = resolveActionSelector( capture.snapshot.nodes, @@ -179,13 +173,7 @@ async function pollSelectorReadinessOnce( params: ResolveInteractionTargetParams, previousPoll: InteractionSnapshot | undefined, ): Promise { - const attempt = await attemptSelectorResolution( - runtime, - options, - selectorExpression, - params, - previousPoll !== undefined, - ); + const attempt = await attemptSelectorResolution(runtime, options, selectorExpression, params); if (previousPoll) inheritPostGestureOutcome(previousPoll.snapshot, attempt.capture.snapshot); if (attempt.resolved?.node.rect) return attempt; const quality = attempt.capture.snapshot.snapshotQuality; @@ -250,9 +238,8 @@ async function readinessExhaustedFailure( * off-screen, non-hittable, and keyboard refusals are not judged here: they run once, after this * loop returns a rect-bearing resolution, as `runInteractionPipelineStages` does. * - * The first poll is unbounded and reuses the snapshot cache (`forceFresh: false`), so a caller - * whose first capture already matches pays the one-or-two-capture cost of a single attempt. Only a poll - * that follows a miss (poll 2+) forces a fresh capture, mirroring how `wait` bypasses the cache. + * The first poll is unbounded, so a caller whose first capture already matches pays the + * one-or-two-capture cost of a single attempt. * * ADR 0011 registry anchor: interaction-guarantees.ts cites this as the runtime-selector * `targetReadiness` `via` symbol. diff --git a/src/daemon/selector-runtime-backend.ts b/src/daemon/selector-runtime-backend.ts index 80bd6af98a..4af5d63305 100644 --- a/src/daemon/selector-runtime-backend.ts +++ b/src/daemon/selector-runtime-backend.ts @@ -188,7 +188,10 @@ function createSelectorBackend(params: SelectorRuntimeDeviceParams): AgentDevice const includeRects = options?.includeRects === true; const snapshotScope = options?.scope ?? req.flags?.snapshotScope; const needsFreshSnapshot = - options?.forceFresh === true || requestNeedsFreshSnapshot(req, device, includeRects); + req.command === 'wait' || + req.command === 'find' || + isAbsentPredicateRequest(req) || + (includeRects && device.platform === 'web'); return await captureRuntime.capture({ flags, signal: context.signal, @@ -238,17 +241,3 @@ function isAbsentPredicateRequest(req: DaemonRequest): boolean { const checked = checkIsArgs(req.positionals ?? []); return checked.ok && checked.predicate === 'absent'; } - -/** The requests whose read must not be served from the short-lived selector capture cache. */ -function requestNeedsFreshSnapshot( - req: Parameters[0], - device: { platform: string }, - includeRects: boolean, -): boolean { - return ( - req.command === 'wait' || - req.command === 'find' || - isAbsentPredicateRequest(req) || - (includeRects && device.platform === 'web') - ); -} diff --git a/test/integration/interaction-contract/runtime-selector.contract.test.ts b/test/integration/interaction-contract/runtime-selector.contract.test.ts index 495de88fa2..90c03366a8 100644 --- a/test/integration/interaction-contract/runtime-selector.contract.test.ts +++ b/test/integration/interaction-contract/runtime-selector.contract.test.ts @@ -202,13 +202,13 @@ test(scenario('targetReadiness'), async () => { const taps: Point[] = []; let captures = 0; const device = createContractDevice(viewportOnlySnapshot(), { - captureSnapshot: async (_context, options) => { + captureSnapshot: async () => { captures += 1; return { - // Polls after the first bypass the selector capture cache (forceFresh); the button appears - // only once that bypass is in effect, so a resolution the loop never actually polled for - // would leave `captures` at 1 and the tap unsent. - snapshot: options?.forceFresh ? continueButtonSnapshot() : viewportOnlySnapshot(), + // The first poll's interactive capture and its full-capture fallback both miss; the button + // appears only from the second poll on, so a loop that never polled again leaves the tap + // unsent. + snapshot: captures >= 3 ? continueButtonSnapshot() : viewportOnlySnapshot(), }; }, tap: async (_context, point) => { From 09cbe1e43622dc9e7a88cdca644342f60328a875 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:21:29 +0200 Subject: [PATCH 16/23] fix(replay): share one readiness budget between the target gate and the dispatch A replayed step has one readiness budget. The pre-dispatch target gate and the dispatch each spent the full budget, so a target that rendered late could cost a step twice the budget. When the gate polled more than once, the dispatch now receives what is left, max(0, budget - gate waitedMs), as its readinessTimeoutMs; 0 makes the dispatch resolve its target once. A gate that matched on its first capture leaves the dispatch the whole budget. A gate that polled more than once emits the interaction_target_readiness diagnostic the dispatched wait emits, with polls, waitedMs, end, and the step's command. A selector-miss divergence after such a wait carries the same evidence at error.details.readiness, where an unannotated step's target-not-found failure already carries it. --- .../session-replay-action-runtime.ts | 8 +++ .../session-replay-runtime-engine-adapter.ts | 49 ++++++++++++++++++- ...replay-target-verification-runtime.test.ts | 37 ++++++++++++++ website/docs/docs/replay-e2e.md | 12 ++--- 4 files changed, 98 insertions(+), 8 deletions(-) diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index bbabaa0ee1..2e61460136 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -46,6 +46,11 @@ export async function invokeReplayAction(params: { /** The isolation scope the daemon already resolved for the request, when it did. */ resolvedSessionScope: SessionScope | undefined; dependencies: ReplayDaemonDependencies; + /** + * What is left of the step's readiness budget after the pre-dispatch target gate waited; 0 makes + * the dispatch resolve its target once. Absent: the step's full budget. + */ + readinessTimeoutMs?: number; }): Promise { const { req, @@ -87,6 +92,7 @@ export async function invokeReplayAction(params: { invoke, resolvedSessionScope, dependencies, + readinessTimeoutMs: params.readinessTimeoutMs, }); } catch (error) { // Only an expected AppError dispatch failure (e.g. a selector-miss) gets @@ -144,10 +150,12 @@ async function invokeResolvedReplayAction(params: { invoke: ReplayInvoke; resolvedSessionScope: SessionScope | undefined; dependencies: ReplayDaemonDependencies; + readinessTimeoutMs: number | undefined; }): Promise { const { req, sessionName, resolved, sourceAction, invoke, resolvedSessionScope, dependencies } = params; const flags = buildReplayActionFlags(req.flags, resolved.flags, resolved.command); + if (params.readinessTimeoutMs !== undefined) flags.readinessTimeoutMs = params.readinessTimeoutMs; const recordedInputVariable = sourceAction.command === 'fill' ? readRecordedInputVariableName(inferFillText(sourceAction)) diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts index 59f43b838c..f8d6822172 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts @@ -27,6 +27,7 @@ import type { AdReplayTargetObservation, } from '@agent-device/ad-replay'; import { observeUntil } from '@agent-device/capture-kit/observe-until'; +import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; import { collectReplayActionArtifactPaths } from '@agent-device/replay-port/session-replay-runtime-artifacts'; import { applyReplayDispatchGuard, @@ -123,6 +124,8 @@ export function createAdReplayStepRuntime(params: { const { ctx, req, artifactPaths, onStep, armSaveScript } = params; let lastResponse: DaemonResponse | undefined; let lastObservation: DivergenceObservation | undefined; + /** This step's pre-dispatch readiness wait, when the gate polled more than once. */ + let gateWait: GateReadinessWait | undefined; /** * The `TargetBindingDivergenceContext` every wire-builder needs — built @@ -236,8 +239,22 @@ export function createAdReplayStepRuntime(params: { schedule: { ...readiness, captureDeadline: 'none' }, ...(ctx.signal ? { signal: ctx.signal } : {}), ...(ctx.dependencies.clock ? { clock: ctx.dependencies.clock } : {}), - phase: 'replay_target_readiness', }); + if (observed.polls.length > 1) { + gateWait = { + budgetMs: readiness.budgetMs, + readiness: { + polls: observed.polls.length, + waitedMs: observed.waitedMs, + end: observed.kind, + }, + }; + emitDiagnostic({ + level: 'debug', + phase: 'interaction_target_readiness', + data: { ...gateWait.readiness, command: action.command }, + }); + } if (observed.kind === 'done') return observed.result; if (observed.last !== undefined) return observed.last; throw observed.kind === 'expired' ? observed.lastError : observed.error; @@ -249,6 +266,11 @@ export function createAdReplayStepRuntime(params: { async dispatchStep(action, resolvedAction, index, _stepArtifactPaths, guard) { const sourceLine = ctx.actionLines[index] ?? 1; const response = await invokeReplayAction({ + ...(gateWait + ? { + readinessTimeoutMs: Math.max(0, gateWait.budgetMs - gateWait.readiness.waitedMs), + } + : {}), req: applyReplayDispatchGuard(ctx.replayReq, guard), sessionName: ctx.sessionName, action, @@ -297,7 +319,11 @@ export function createAdReplayStepRuntime(params: { evidence, observation, ); - return recordFailure(response); + return recordFailure( + evidence.kind === 'selector-miss' && gateWait + ? withReadinessDetail(response, gateWait.readiness) + : response, + ); }, async buildPostDispatchTargetBindingFailure( @@ -350,6 +376,7 @@ export function createAdReplayStepRuntime(params: { // capabilities — the natural per-step boundary to clear the previous // step's capture (see this factory's own header). lastObservation = undefined; + gateWait = undefined; armSaveScript(); }, isRepairArmed: () => ctx.coordinator.view()?.repairBoundary !== undefined, @@ -496,3 +523,21 @@ function isTargetNotRenderedYet(observation: AdReplayTargetObservation): boolean observation.classification.kind === 'selector-miss' ); } + +type GateReadinessWait = { + budgetMs: number; + /** The same evidence a dispatched readiness wait reports (`SelectorReadinessDetails`). */ + readiness: { polls: number; waitedMs: number; end: string }; +}; + +/** Carries the gate's wait where a dispatched step's target-not-found failure carries its own. */ +function withReadinessDetail( + response: DaemonResponse, + readiness: GateReadinessWait['readiness'], +): DaemonResponse { + if (response.ok) return response; + return { + ...response, + error: { ...response.error, details: { ...response.error.details, readiness } }, + }; +} diff --git a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts index 8a4fe8afe2..0bd1076de1 100644 --- a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts +++ b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts @@ -28,7 +28,13 @@ vi.mock('@agent-device/host-kit/retry', async (importOriginal) => { return { ...actual, sleep: vi.fn(async () => {}) }; }); +vi.mock('@agent-device/host-kit/diagnostics', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, emitDiagnostic: vi.fn(actual.emitDiagnostic) }; +}); + import { AppError } from '@agent-device/kernel/errors'; +import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; import { legacyDispatchCapture, resetLegacySnapshotCapture, @@ -51,8 +57,16 @@ const mockCaptureSnapshotWithInteractor = vi.mocked(captureSnapshotWithInteracto beforeEach(() => { resetLegacySnapshotCapture(mockCaptureSnapshotWithInteractor); + vi.mocked(emitDiagnostic).mockClear(); }); +function readinessDiagnostics(): unknown[] { + return vi + .mocked(emitDiagnostic) + .mock.calls.filter(([event]) => event.phase === 'interaction_target_readiness') + .map(([event]) => event.data); +} + test('an unannotated action executes unchanged (old-script pass-through)', async () => { const scene = replayScriptScene('agent-device-replay-target-verify-passthrough-', [ 'click id="save"', @@ -80,6 +94,9 @@ test('a verified target proceeds to dispatch the action', async () => { expect(response.ok).toBe(true); expect(scene.invoked.map((req) => req.command)).toEqual(['click']); expect(scene.invoked[0]?.positionals).toEqual(['id="save"']); + // Present on the gate's first capture: the dispatch keeps the step's whole budget. + expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(2_000); + expect(readinessDiagnostics()).toEqual([]); }); test('a verified drag guards both source and destination before dispatch', async () => { @@ -179,6 +196,7 @@ test('a selector-miss divergence blocks dispatch and never sends the action', as expect(targetBinding.recorded).toEqual({ id: 'save', role: 'button', label: 'Save' }); // click waits for its target, so the gate re-captured before refusing. expect(mockDispatchCommand.mock.calls.length).toBeGreaterThan(1); + expect(response.error.details?.readiness).toMatchObject({ waitedMs: 2_000, end: 'expired' }); }); test('an annotated click whose target renders on the second capture waits for it and dispatches', async () => { @@ -197,6 +215,25 @@ test('an annotated click whose target renders on the second capture waits for it identity: { id: 'save', role: 'button', label: 'Save' }, }); expect(mockDispatchCommand).toHaveBeenCalledTimes(2); + expect(readinessDiagnostics()).toEqual([ + { polls: 2, waitedMs: 200, end: 'done', command: 'click' }, + ]); +}); + +test('the dispatch gets only the readiness budget the gate left', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-shared-budget-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + + // Eight misses, one interval apart, before the target renders 1.6 s into the 2 s budget. + for (let miss = 0; miss < 8; miss += 1) mockDispatchCommand.mockResolvedValueOnce(emptyCapture()); + mockDispatchCommand.mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay(); + + expect(response.ok).toBe(true); + expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(400); }); test('an annotated step whose command does not wait for its target refuses a selector miss on one capture', async () => { diff --git a/website/docs/docs/replay-e2e.md b/website/docs/docs/replay-e2e.md index f423f6bd4e..3023fa48d1 100644 --- a/website/docs/docs/replay-e2e.md +++ b/website/docs/docs/replay-e2e.md @@ -62,13 +62,13 @@ agent-device replay ~/.agent-device/sessions/e2e-2026-02-09T12-00-00-000Z.ad --s they fail. A step recorded against a screen that was still loading passes on replay once the target shows up. The wait covers only a target that is not on screen yet: a target that is covered, off-screen, or matched by more than one element fails at once, as it does live. -- When the target never appears, replay stops with `REPLAY_DIVERGENCE`. For a step recorded with - a target annotation (the `# agent-device:target-v1` line above it), `error.details.divergence.kind` - is `selector-miss` and the step is never sent. For a step without an annotation, - `error.details.reason` is `selector_not_found`, as for a live command, and +- When the target never appears, replay stops with `REPLAY_DIVERGENCE`, and `error.details.readiness` says how long the step waited and how many times it looked (`waitedMs`, - `polls`, `end`). If the app shows an empty accessibility tree during that wait, - `error.details.reason` is `capture_sparse` instead; take a snapshot to see where the app is. + `polls`, `end`). For a step recorded with a target annotation (the `# agent-device:target-v1` + line above it), `error.details.divergence.kind` is `selector-miss` and the step is never sent. + For a step without an annotation, `error.details.reason` is `selector_not_found`, as for a live + command. If the app shows an empty accessibility tree during that wait, `error.details.reason` + is `capture_sparse` instead; take a snapshot to see where the app is. ## Run Maestro compatibility flows From 1481f361d1e5cfa6af91661b5dedde19022cb112 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 16:43:54 +0200 Subject: [PATCH 17/23] fix(command-registry): pin the readiness cap without importing the selector policy The envelope widening imported the promotedTarget selector row to read its 2 s poll ceiling, which made every command-registry entry and src/cli.ts evaluate the selector policy module. The cap is now READINESS_BUDGET_MAX_MS in timeout-policy.ts, pinned by test to the row's maxTimeoutMs; selectors cannot import command-registry, so the test is the tie. The same eager-closure gate showed two more new edges from this branch: - replay's native and test command entries evaluated observe-until and the selector policy through the target gate. Both now load inside the gate's readiness path, the only code that uses them. - src/cli.ts evaluated the new target-readiness-grammar module. Its one function moves into interaction/metadata.ts, its one caller, and its tests into metadata.test.ts. --- .../command-registry/src/timeout-policy.ts | 9 +++- packages/command-registry/src/types.ts | 4 +- .../session-replay-action-runtime.ts | 8 ++-- .../session-replay-runtime-engine-adapter.ts | 4 +- .../command-descriptor-timeout-policy.test.ts | 11 ++++- src/commands/common-input-fields.ts | 2 +- src/commands/interaction/metadata.test.ts | 33 +++++++++++++++ src/commands/interaction/metadata.ts | 27 +++++++++++- src/commands/target-readiness-grammar.test.ts | 41 ------------------- src/commands/target-readiness-grammar.ts | 26 ------------ 10 files changed, 85 insertions(+), 80 deletions(-) delete mode 100644 src/commands/target-readiness-grammar.test.ts delete mode 100644 src/commands/target-readiness-grammar.ts diff --git a/packages/command-registry/src/timeout-policy.ts b/packages/command-registry/src/timeout-policy.ts index 42a3b4fb02..587c59315c 100644 --- a/packages/command-registry/src/timeout-policy.ts +++ b/packages/command-registry/src/timeout-policy.ts @@ -1,4 +1,3 @@ -import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import type { CommandTimeoutBudget, CommandTimeoutPolicy } from './types.ts'; // Request-envelope constants, relocated from src/daemon/request-timeouts.ts when @@ -87,6 +86,12 @@ export function resolveCommandRequestTimeoutMs( return envelopeMs + readinessBudgetMs(policy, input.flags); } +/** + * The ceiling of a `targetReadiness: 'budgeted'` command's readiness budget: the promotedTarget + * selector row's poll ceiling, pinned to it by test. + */ +export const READINESS_BUDGET_MAX_MS = 2_000; + /** * A readiness budget is target-poll time spent before the action itself, so it extends whatever * envelope the command otherwise has; the daemon's own poll must never outlive the client's clock. @@ -99,7 +104,7 @@ function readinessBudgetMs( if (policy.targetReadiness !== 'budgeted') return 0; const budgetMs = flags?.readinessTimeoutMs; if (typeof budgetMs !== 'number' || !Number.isInteger(budgetMs) || budgetMs <= 0) return 0; - return Math.min(budgetMs, SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs); + return Math.min(budgetMs, READINESS_BUDGET_MAX_MS); } function resolvePositionalBudgetTimeoutMs( diff --git a/packages/command-registry/src/types.ts b/packages/command-registry/src/types.ts index 5bcafffb4e..16911f71aa 100644 --- a/packages/command-registry/src/types.ts +++ b/packages/command-registry/src/types.ts @@ -85,8 +85,8 @@ export type CommandTimeoutPolicy = { /** * `budgeted`: the command's target resolution may poll for a target that does not exist yet, under - * a caller-supplied `readinessTimeoutMs`. Must match the commands the `targetReadiness` interaction - * guarantee cells mark `runtime`. + * a caller-supplied `readinessTimeoutMs`, capped at `READINESS_BUDGET_MAX_MS` (`timeout-policy.ts`). + * Must match the commands the `targetReadiness` interaction guarantee cells mark `runtime`. */ export type CommandTargetReadiness = 'budgeted'; diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index 2e61460136..f3d86177fa 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -7,7 +7,6 @@ import type { } from '@agent-device/replay-port/command-types'; import { mergeParentFlags } from '@agent-device/command-registry/batch'; import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; -import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import { AppError, normalizeError } from '@agent-device/kernel/errors'; import { gesturePayloadFromPositionals, @@ -236,12 +235,15 @@ const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; * step's command does not wait for its target. The pre-dispatch target gate polls under the same * schedule, so neither refuses a target the other would still wait for. */ -export function replayStepReadinessSchedule( +export async function replayStepReadinessSchedule( parentFlags: CommandFlags | undefined, action: SessionAction, -): { intervalMs: number; budgetMs: number } | undefined { +): Promise<{ intervalMs: number; budgetMs: number } | undefined> { const { readinessTimeoutMs } = buildReplayActionFlags(parentFlags, action.flags, action.command); if (readinessTimeoutMs === undefined || readinessTimeoutMs <= 0) return undefined; + // Loaded only by a step that waits, so the replay entry does not evaluate the selector policy. + const { SELECTOR_PIPELINE_POLICIES } = + await import('@agent-device/selectors/selector-pipeline-policy'); const poll = SELECTOR_PIPELINE_POLICIES.promotedTarget.poll; return { intervalMs: poll.intervalMs, budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs) }; } diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts index f8d6822172..10341b70d0 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts @@ -26,7 +26,6 @@ import type { AdReplayStepRuntime, AdReplayTargetObservation, } from '@agent-device/ad-replay'; -import { observeUntil } from '@agent-device/capture-kit/observe-until'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; import { collectReplayActionArtifactPaths } from '@agent-device/replay-port/session-replay-runtime-artifacts'; import { @@ -228,8 +227,9 @@ export function createAdReplayStepRuntime(params: { // command resolving a selector (a `@ref` resolves without a wait). const readiness = token.startsWith('@') ? undefined - : replayStepReadinessSchedule(ctx.replayReq.flags, action); + : await replayStepReadinessSchedule(ctx.replayReq.flags, action); if (!readiness) return await observeOnce(); + const { observeUntil } = await import('@agent-device/capture-kit/observe-until'); const observed = await observeUntil({ capture: observeOnce, verdict: (latest) => diff --git a/src/__tests__/command-descriptor-timeout-policy.test.ts b/src/__tests__/command-descriptor-timeout-policy.test.ts index 97d25c8706..692d6bc8c4 100644 --- a/src/__tests__/command-descriptor-timeout-policy.test.ts +++ b/src/__tests__/command-descriptor-timeout-policy.test.ts @@ -11,6 +11,7 @@ import { INTERACTION_DISPATCH_PATHS } from '@agent-device/contracts/interaction- import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; import { DEFAULT_TIMEOUT_POLICY, + READINESS_BUDGET_MAX_MS, resolveCommandRequestTimeoutMs, } from '@agent-device/command-registry/timeout-policy'; import { DEFAULT_STABLE_TIMEOUT_MS } from '../commands/interaction/runtime/stable-capture.ts'; @@ -428,9 +429,15 @@ test('a readiness budget widens the request envelope on top of the settle envelo assert.equal(resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 0 } }), 90_000); assert.equal( resolveCommandRequestTimeoutMs(press, { flags: { readinessTimeoutMs: 999_000 } }), - 90_000 + SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs, + 90_000 + 2_000, + ); +}); + +test('the envelope readiness cap is the promotedTarget row poll ceiling', () => { + assert.equal( + READINESS_BUDGET_MAX_MS, + SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs, ); - assert.equal(SELECTOR_PIPELINE_POLICIES.promotedTarget.poll.maxTimeoutMs, 2_000); assert.equal( resolveCommandRequestTimeoutMs(resolveCommandTimeoutPolicy('fill'), { flags: { readinessTimeoutMs: 2_000 }, diff --git a/src/commands/common-input-fields.ts b/src/commands/common-input-fields.ts index e28dd48773..ba88f9e0f2 100644 --- a/src/commands/common-input-fields.ts +++ b/src/commands/common-input-fields.ts @@ -34,7 +34,7 @@ export type CommonCommandInput = Pick< export type CommonInputReadOptions = { readTargetAlias?: boolean; - /** The command's own fields declare `readinessTimeoutMs` (`target-readiness-grammar.ts`). */ + /** The command's own fields declare `readinessTimeoutMs` (`interaction/metadata.ts`). */ readinessBudgetDeclared?: boolean; }; diff --git a/src/commands/interaction/metadata.test.ts b/src/commands/interaction/metadata.test.ts index 182265e17b..7d108661e5 100644 --- a/src/commands/interaction/metadata.test.ts +++ b/src/commands/interaction/metadata.test.ts @@ -1,5 +1,7 @@ import { expect, test } from 'vitest'; import { AppError } from '@agent-device/kernel/errors'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import { findCommandMetadata, listCommandMetadata } from '../command-metadata.ts'; import { interactionCommandMetadata } from './metadata.ts'; // `fill ""` is the clear-field primitive (#2063): before it existed, emptying an input @@ -50,3 +52,34 @@ test('type keeps refusing an empty text: appending nothing is not a clear', () = expect((error as AppError).code).toBe('INVALID_ARGS'); } }); + +const SELECTOR_TARGET = { kind: 'selector', selector: 'label=Continue' }; + +test('a command advertises readinessTimeoutMs exactly when its descriptor declares the budget', () => { + for (const metadata of listCommandMetadata()) { + const properties = metadata.inputSchema.properties ?? {}; + expect('readinessTimeoutMs' in properties, metadata.name).toBe( + commandAcceptsReadinessBudget(metadata.name), + ); + } +}); + +test('press reads readinessTimeoutMs', () => { + const input = findCommandMetadata('press').readInput({ + target: SELECTOR_TARGET, + readinessTimeoutMs: 2_000, + }) as { readinessTimeoutMs?: number }; + expect(input.readinessTimeoutMs).toBe(2_000); +}); + +test('fill refuses readinessTimeoutMs with an input error, and reads without it', () => { + const fill = findCommandMetadata('fill'); + expect(() => fill.readInput({ target: SELECTOR_TARGET, text: 'hi' })).not.toThrow(); + let refusal: unknown; + try { + fill.readInput({ target: SELECTOR_TARGET, text: 'hi', readinessTimeoutMs: 2_000 }); + } catch (error) { + refusal = error; + } + expect(refusal instanceof AppError && refusal.code).toBe('INVALID_ARGS'); +}); diff --git a/src/commands/interaction/metadata.ts b/src/commands/interaction/metadata.ts index 17659852b7..62a5b4ff26 100644 --- a/src/commands/interaction/metadata.ts +++ b/src/commands/interaction/metadata.ts @@ -21,6 +21,7 @@ import { SWIPE_REPETITION_MAX, } from '@agent-device/contracts/scroll-gesture'; import { FIND_LOCATORS } from '@agent-device/selectors'; +import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; import { booleanField, elementTargetField, @@ -28,6 +29,7 @@ import { integerField, interactionTargetField, numberField, + operatorField, pointField, repeatedFields, requiredField, @@ -40,7 +42,6 @@ import { readCommonInput, type CommonCommandInput } from '../common-input-fields import { readInputRecord } from '../input-readers.ts'; import { defineFieldCommandMetadata } from '../field-command-contract.ts'; import { postActionObservationFields } from '../post-action-observation-grammar.ts'; -import { targetReadinessFields } from '../target-readiness-grammar.ts'; const FIND_ACTION_VALUES = [ 'click', @@ -79,6 +80,30 @@ const interactionCommandDescriptions = { type InteractionCommandName = keyof typeof interactionCommandDescriptions; +/** + * The input field a command's `targetReadiness: 'budgeted'` descriptor trait entitles it to. Only + * those commands declare `readinessTimeoutMs`; the common input reader refuses the key for every + * command whose fields do not (`common-input-fields.ts`). Fails closed at module load for a command + * without the trait, so a field map cannot advertise a budget its runtime never polls under. + */ +function targetReadinessFields(command: InteractionCommandName) { + if (!commandAcceptsReadinessBudget(command)) { + throw new Error(`${command} does not declare targetReadiness: 'budgeted'`); + } + return { + readinessTimeoutMs: operatorField( + integerField( + "Operator-only: how long the command may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", + { min: 1 }, + ), + { + operatorPath: + 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', + }, + ), + }; +} + const clickFields = { target: requiredField(interactionTargetField()), button: enumField(CLICK_BUTTONS, 'Pointer button for platforms that support mouse buttons.'), diff --git a/src/commands/target-readiness-grammar.test.ts b/src/commands/target-readiness-grammar.test.ts deleted file mode 100644 index 1756340a26..0000000000 --- a/src/commands/target-readiness-grammar.test.ts +++ /dev/null @@ -1,41 +0,0 @@ -import assert from 'node:assert/strict'; -import { test } from 'vitest'; -import { AppError } from '@agent-device/kernel/errors'; -import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; -import { findCommandMetadata, listCommandMetadata } from './command-metadata.ts'; - -const SELECTOR_TARGET = { kind: 'selector', selector: 'label=Continue' }; - -function metadataFor(name: string) { - const metadata = findCommandMetadata(name); - assert.ok(metadata, `expected command metadata for ${name}`); - return metadata; -} - -test('a command advertises readinessTimeoutMs exactly when its descriptor declares the budget', () => { - for (const metadata of listCommandMetadata()) { - const properties = metadata.inputSchema.properties ?? {}; - assert.equal( - 'readinessTimeoutMs' in properties, - commandAcceptsReadinessBudget(metadata.name), - metadata.name, - ); - } -}); - -test('press reads readinessTimeoutMs', () => { - const input = metadataFor('press').readInput({ - target: SELECTOR_TARGET, - readinessTimeoutMs: 2_000, - }) as { readinessTimeoutMs?: number }; - assert.equal(input.readinessTimeoutMs, 2_000); -}); - -test('fill refuses readinessTimeoutMs with an input error, and reads without it', () => { - const fill = metadataFor('fill'); - assert.doesNotThrow(() => fill.readInput({ target: SELECTOR_TARGET, text: 'hi' })); - assert.throws( - () => fill.readInput({ target: SELECTOR_TARGET, text: 'hi', readinessTimeoutMs: 2_000 }), - (error: unknown) => error instanceof AppError && error.code === 'INVALID_ARGS', - ); -}); diff --git a/src/commands/target-readiness-grammar.ts b/src/commands/target-readiness-grammar.ts deleted file mode 100644 index 90e6e70fbb..0000000000 --- a/src/commands/target-readiness-grammar.ts +++ /dev/null @@ -1,26 +0,0 @@ -import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; -import { integerField, operatorField } from './command-input.ts'; - -/** - * The input field a command's `targetReadiness: 'budgeted'` descriptor trait entitles it to. Only - * those commands declare `readinessTimeoutMs`; the common input reader refuses the key for every - * command whose fields do not (`common-input-fields.ts`). Fails closed at module load for a command - * without the trait, so a field map cannot advertise a budget its runtime never polls under. - */ -export function targetReadinessFields(command: string) { - if (!commandAcceptsReadinessBudget(command)) { - throw new Error(`${command} does not declare targetReadiness: 'budgeted'`); - } - return { - readinessTimeoutMs: operatorField( - integerField( - "Operator-only: how long the command may poll for a target that does not exist yet, in milliseconds. Capped at the promotedTarget row's maxTimeoutMs; omitted takes the one-attempt resolution path.", - { min: 1 }, - ), - { - operatorPath: - 'Pass readinessTimeoutMs directly as CLI/Node.js command input; it is not exposed to model-facing tools.', - }, - ), - }; -} From 1f8f6270309f7ea6695775767441666cad2605ab Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 19:49:16 +0200 Subject: [PATCH 18/23] refactor(selectors): one owner for the readiness schedule The replay target gate and the dispatch each built the promotedTarget readiness schedule from the row's poll budget. readinessScheduleFor now lives next to SELECTOR_PIPELINE_POLICIES and both callers use it: the dispatch directly, the replay gate through its existing dynamic import. The two waits also counted the budget from different origins: the gate from the start of its wait, the dispatch from the end of its first capture. The schedule now carries budgetFrom: 'first-capture' for both, and the budget the gate leaves the dispatch is the step budget minus the gate's time after its first capture ended. The first capture is the one-attempt lookup a step pays without a wait, in the gate as in the dispatch. --- .../session-replay-action-runtime.ts | 14 +++--- .../session-replay-runtime-engine-adapter.ts | 14 ++++-- .../src/selector-pipeline-policy.test.ts | 30 ++++++++++++ .../selectors/src/selector-pipeline-policy.ts | 29 ++++++++++++ .../interaction/runtime/resolution.ts | 22 +-------- .../interaction/runtime/selector-readiness.ts | 20 ++------ .../replay-runtime/replay-command-fixture.ts | 6 ++- .../session-replay-action-runtime.test.ts | 47 ++++++++++++++++++- .../session-replay-scenario.fixtures.ts | 8 ++-- ...replay-target-verification-runtime.test.ts | 30 ++++++++++++ 10 files changed, 168 insertions(+), 52 deletions(-) create mode 100644 packages/selectors/src/selector-pipeline-policy.test.ts diff --git a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts index f3d86177fa..e826dc7cc5 100644 --- a/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts +++ b/packages/replay-port/src/daemon-port/session-replay-action-runtime.ts @@ -7,6 +7,7 @@ import type { } from '@agent-device/replay-port/command-types'; import { mergeParentFlags } from '@agent-device/command-registry/batch'; import { commandAcceptsReadinessBudget } from '@agent-device/command-registry/registry'; +import type { ReadinessSchedule } from '@agent-device/selectors/selector-pipeline-policy'; import { AppError, normalizeError } from '@agent-device/kernel/errors'; import { gesturePayloadFromPositionals, @@ -232,20 +233,19 @@ const REPLAY_DEFAULT_READINESS_TIMEOUT_MS = 2_000; /** * The readiness schedule a replayed step's dispatch polls its target under, or undefined when the - * step's command does not wait for its target. The pre-dispatch target gate polls under the same - * schedule, so neither refuses a target the other would still wait for. + * step's command does not wait for its target. The pre-dispatch target gate polls under this same + * schedule, built by the selector policy's `readinessScheduleFor` as the dispatch builds its own. */ export async function replayStepReadinessSchedule( parentFlags: CommandFlags | undefined, action: SessionAction, -): Promise<{ intervalMs: number; budgetMs: number } | undefined> { +): Promise { const { readinessTimeoutMs } = buildReplayActionFlags(parentFlags, action.flags, action.command); - if (readinessTimeoutMs === undefined || readinessTimeoutMs <= 0) return undefined; + if (readinessTimeoutMs === undefined) return undefined; // Loaded only by a step that waits, so the replay entry does not evaluate the selector policy. - const { SELECTOR_PIPELINE_POLICIES } = + const { SELECTOR_PIPELINE_POLICIES, readinessScheduleFor } = await import('@agent-device/selectors/selector-pipeline-policy'); - const poll = SELECTOR_PIPELINE_POLICIES.promotedTarget.poll; - return { intervalMs: poll.intervalMs, budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs) }; + return readinessScheduleFor(SELECTOR_PIPELINE_POLICIES.promotedTarget.poll, readinessTimeoutMs); } function buildReplayActionFlags( diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts index 10341b70d0..f277bab00e 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts @@ -27,6 +27,7 @@ import type { AdReplayTargetObservation, } from '@agent-device/ad-replay'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; +import type { ObservationEvidence } from '@agent-device/capture-kit/observe-until'; import { collectReplayActionArtifactPaths } from '@agent-device/replay-port/session-replay-runtime-artifacts'; import { applyReplayDispatchGuard, @@ -242,7 +243,7 @@ export function createAdReplayStepRuntime(params: { }); if (observed.polls.length > 1) { gateWait = { - budgetMs: readiness.budgetMs, + remainingBudgetMs: Math.max(0, readiness.budgetMs - budgetSpentMs(observed)), readiness: { polls: observed.polls.length, waitedMs: observed.waitedMs, @@ -268,7 +269,7 @@ export function createAdReplayStepRuntime(params: { const response = await invokeReplayAction({ ...(gateWait ? { - readinessTimeoutMs: Math.max(0, gateWait.budgetMs - gateWait.readiness.waitedMs), + readinessTimeoutMs: gateWait.remainingBudgetMs, } : {}), req: applyReplayDispatchGuard(ctx.replayReq, guard), @@ -525,11 +526,18 @@ function isTargetNotRenderedYet(observation: AdReplayTargetObservation): boolean } type GateReadinessWait = { - budgetMs: number; + /** The step's readiness budget the gate left for the dispatch. */ + remainingBudgetMs: number; /** The same evidence a dispatched readiness wait reports (`SelectorReadinessDetails`). */ readiness: { polls: number; waitedMs: number; end: string }; }; +/** The budget a `budgetFrom: 'first-capture'` wait spent: its time after the first capture ended. */ +function budgetSpentMs(observed: ObservationEvidence): number { + const first = observed.polls[0]; + return first ? observed.waitedMs - (first.startedMs + first.durationMs) : 0; +} + /** Carries the gate's wait where a dispatched step's target-not-found failure carries its own. */ function withReadinessDetail( response: DaemonResponse, diff --git a/packages/selectors/src/selector-pipeline-policy.test.ts b/packages/selectors/src/selector-pipeline-policy.test.ts new file mode 100644 index 0000000000..c074c1a223 --- /dev/null +++ b/packages/selectors/src/selector-pipeline-policy.test.ts @@ -0,0 +1,30 @@ +import { describe, expect, test } from 'vitest'; +import { SELECTOR_PIPELINE_POLICIES, readinessScheduleFor } from './selector-pipeline-policy.ts'; + +const promotedPoll = SELECTOR_PIPELINE_POLICIES.promotedTarget.poll; + +describe('readinessScheduleFor', () => { + test('polls at the row cadence with the supplied budget, counted from the first capture', () => { + expect(readinessScheduleFor(promotedPoll, 800)).toEqual({ + intervalMs: promotedPoll.intervalMs, + budgetMs: 800, + budgetFrom: 'first-capture', + }); + }); + + test('caps the supplied budget at the row ceiling', () => { + expect(readinessScheduleFor(promotedPoll, promotedPoll.maxTimeoutMs + 5_000)?.budgetMs).toBe( + promotedPoll.maxTimeoutMs, + ); + }); + + test('a row that resolves against one capture has no schedule', () => { + expect(readinessScheduleFor(SELECTOR_PIPELINE_POLICIES.resolvedTarget.poll, 800)).toBe( + undefined, + ); + }); + + test.each([undefined, 0, -1, 1.5])('a %s budget takes the one-attempt path', (budget) => { + expect(readinessScheduleFor(promotedPoll, budget)).toBe(undefined); + }); +}); diff --git a/packages/selectors/src/selector-pipeline-policy.ts b/packages/selectors/src/selector-pipeline-policy.ts index 68c5fec2b3..aed75591b7 100644 --- a/packages/selectors/src/selector-pipeline-policy.ts +++ b/packages/selectors/src/selector-pipeline-policy.ts @@ -2,6 +2,7 @@ import { SELECTOR_RESOLUTION_POLICIES, type SelectorResolutionPolicy, } from '@agent-device/selectors'; +import type { ObservationSchedule } from '@agent-device/capture-kit/observe-until'; /** * The structural half of the per-caller selector policy (#1656), companion to @@ -231,6 +232,34 @@ export const SELECTOR_PIPELINE_POLICIES = { export type SelectorPipelinePolicyName = keyof typeof SELECTOR_PIPELINE_POLICIES; +/** + * The schedule a readiness wait polls its target under. The budget counts from the end of the + * first capture: that capture is the one-attempt lookup a step pays without any wait, so the + * replay target gate and the dispatch spend the budget only on retries. + */ +export type ReadinessSchedule = Readonly< + Pick & { budgetFrom: 'first-capture' } +>; + +/** + * The one owner of a readiness schedule, for the replay target gate and the dispatch alike: + * `undefined` for a row that resolves against one capture or when no positive integer budget was + * supplied, otherwise the row's cadence and the supplied budget capped at the row's `maxTimeoutMs`. + */ +export function readinessScheduleFor( + poll: ReadinessPollBudget | 'none', + readinessTimeoutMs: number | undefined, +): ReadinessSchedule | undefined { + if (poll === 'none') return undefined; + if (readinessTimeoutMs === undefined || !Number.isInteger(readinessTimeoutMs)) return undefined; + if (readinessTimeoutMs <= 0) return undefined; + return { + intervalMs: poll.intervalMs, + budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs), + budgetFrom: 'first-capture', + }; +} + /** * The two questions a row asks the engine, derived from its ambiguity contract * rather than from a hand-kept list of row names: `reject-candidates` rows go diff --git a/src/commands/interaction/runtime/resolution.ts b/src/commands/interaction/runtime/resolution.ts index c0802f3bf8..a141a486ee 100644 --- a/src/commands/interaction/runtime/resolution.ts +++ b/src/commands/interaction/runtime/resolution.ts @@ -1,12 +1,11 @@ import { AppError } from '@agent-device/kernel/errors'; import type { AgentDeviceRuntime, CommandContext } from '../../../runtime-contract.ts'; import type { SelectorResolution } from '@agent-device/selectors'; -import type { ActingPipelinePolicy } from '@agent-device/selectors/selector-pipeline-policy'; +import { readinessScheduleFor } from '@agent-device/selectors/selector-pipeline-policy'; import { attemptSelectorResolution, pollForSelectorReadiness, selectorInteractionFailure, - type ReadinessSchedule, } from './selector-readiness.ts'; import { captureInteractionSnapshot, @@ -130,25 +129,6 @@ async function tryCaptureEvidenceBaseline( } } -function isPositiveInteger(value: number | undefined): value is number { - return typeof value === 'number' && Number.isInteger(value) && value > 0; -} - -/** - * The schedule THIS call may poll `promotedTarget` under: `undefined` when the row resolves - * against one capture (`resolvedTarget`'s `poll: 'none'`), or when the caller supplied no - * `readinessTimeoutMs` — both take the one-attempt path. Otherwise the row's own cadence, and a - * budget capped at the row's `maxTimeoutMs`, so a caller-supplied value can shrink the wait but - * never stretch past what the row allows. - */ -function readinessScheduleFor( - poll: ActingPipelinePolicy['poll'], - readinessTimeoutMs: number | undefined, -): ReadinessSchedule | undefined { - if (poll === 'none' || !isPositiveInteger(readinessTimeoutMs)) return undefined; - return { intervalMs: poll.intervalMs, budgetMs: Math.min(readinessTimeoutMs, poll.maxTimeoutMs) }; -} - async function resolveSelectorInteractionTarget( runtime: AgentDeviceRuntime, options: CommandContext, diff --git a/src/commands/interaction/runtime/selector-readiness.ts b/src/commands/interaction/runtime/selector-readiness.ts index ce62382999..cddb669138 100644 --- a/src/commands/interaction/runtime/selector-readiness.ts +++ b/src/commands/interaction/runtime/selector-readiness.ts @@ -7,7 +7,10 @@ import { type SelectorResolution, } from '@agent-device/selectors'; import { resolveSelectorPipeline } from '@agent-device/selectors/selector-pipeline'; -import { SELECTOR_PIPELINE_POLICIES } from '@agent-device/selectors/selector-pipeline-policy'; +import { + SELECTOR_PIPELINE_POLICIES, + type ReadinessSchedule, +} from '@agent-device/selectors/selector-pipeline-policy'; import { INTERACTION_ERROR_REASONS } from '@agent-device/selectors/interaction-error'; import { observeUntil } from '@agent-device/capture-kit/observe-until'; import { isSparseSnapshotQualityVerdict } from '@agent-device/capture-kit/snapshot-quality-verdict'; @@ -50,17 +53,6 @@ export type SelectorReadinessDetails = { end: 'expired' | 'stalled' | 'sparse'; }; -/** - * The schedule one call may poll under: the row's own cadence, and a budget the caller - * (`resolution.ts#resolveSelectorInteractionTarget`) has already capped at the row's own ceiling. - * This module never reads the row or the caller-supplied `readinessTimeoutMs` directly — deciding - * whether and how long to poll is the caller's job; this module only runs the loop. - */ -export type ReadinessSchedule = { - intervalMs: number; - budgetMs: number; -}; - /** * One capture-and-resolve attempt: interactive capture, resolve; on a miss that requires * interactivity, fall back to a full (non-interactive) capture and resolve again. Identical body @@ -270,9 +262,7 @@ export async function pollForSelectorReadiness( ? { kind: 'done', result: { capture: latest.capture, resolved: latest.resolved } } : { kind: 'continue' }, schedule: { - intervalMs: schedule.intervalMs, - budgetMs: schedule.budgetMs, - budgetFrom: 'first-capture', + ...schedule, // The poll signal reaches the platform as CaptureSnapshotInput.signal, which the snapshot // binding joins (captureSnapshotSignal): the same per-capture cancellation `wait` relies on. captureDeadline: 'cancel', diff --git a/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts b/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts index 5cddac764e..1e49f6d0d8 100644 --- a/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts +++ b/src/daemon/__tests__/replay-runtime/replay-command-fixture.ts @@ -21,6 +21,8 @@ export type ReplayCommandTestInput = Readonly<{ invoke: DaemonInvokeFn; tracePath?: string; onStep?: ReplayTestAttemptStepSink; + /** Paces the target-readiness wait; default: `instantReplayClock`. */ + clock?: ObservationClock; }>; /** @@ -29,12 +31,12 @@ export type ReplayCommandTestInput = Readonly<{ * test's `invoke` sees the same `DaemonRequest` the daemon would. */ export function replayCommandForTest(params: ReplayCommandTestInput): ReplayCommand { - const { req, sessionName, logPath, sessionStore, invoke, tracePath, onStep } = params; + const { req, sessionName, logPath, sessionStore, invoke, tracePath, onStep, clock } = params; return { ...splitReplayCommandRequest(req), session: createReplaySession(sessionName, logPath, sessionStore), invoke: replayInvokeOverDispatch(invoke, req), - dependencies: { ...replayDaemonDependencies, clock: instantReplayClock() }, + dependencies: { ...replayDaemonDependencies, clock: clock ?? instantReplayClock() }, ...(tracePath === undefined ? {} : { tracePath }), ...(onStep === undefined ? {} : { onStep }), }; diff --git a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts index 4c4092464f..e50b53df9f 100644 --- a/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts +++ b/src/daemon/__tests__/replay-runtime/session-replay-action-runtime.test.ts @@ -3,7 +3,14 @@ import { expect, test } from 'vitest'; import { makeIosSession } from '../../../__tests__/test-utils/session-factories.ts'; import { recordActionEntry } from '../../session-action-recorder.ts'; import type { DaemonRequest } from '../../daemon-request.ts'; -import { invokeReplayAction } from '@agent-device/replay-port/session-replay-action-runtime'; +import { + invokeReplayAction, + replayStepReadinessSchedule, +} from '@agent-device/replay-port/session-replay-action-runtime'; +import { + SELECTOR_PIPELINE_POLICIES, + readinessScheduleFor, +} from '@agent-device/selectors/selector-pipeline-policy'; import { replayDaemonDependencies } from '../../handlers/session-replay-command.ts'; import { resolveReplayAction } from '@agent-device/ad-script'; import { @@ -94,6 +101,44 @@ test.each(READINESS_BUDGETED_COMMANDS)( }, ); +// The dispatch resolves a press/click/longpress target under the promotedTarget row with the +// readinessTimeoutMs it receives; the pre-dispatch gate must poll under that same schedule. +test.each([ + ...READINESS_BUDGETED_COMMANDS.map((command) => ({ command, flags: {} })), + { command: 'click', flags: { readinessTimeoutMs: 700 } }, + { command: 'press', flags: { readinessTimeoutMs: 9_000 } }, +])('the replay gate and the dispatch poll a $command step under one schedule', async (step) => { + const action: SessionAction = { + ts: 0, + command: step.command, + positionals: ['label="Continue"'], + flags: step.flags, + }; + let dispatchedFlags: Record | undefined; + await invokeReplayAction({ + req: REPLAY_REQUEST, + sessionName: 'default', + action, + resolved: action, + filePath: 'flow.ad', + line: 1, + step: 1, + resolvedSessionScope: undefined, + dependencies: replayDaemonDependencies, + invoke: async (request) => { + dispatchedFlags = request.flags; + return { ok: true, data: {} }; + }, + }); + + const dispatchSchedule = readinessScheduleFor( + SELECTOR_PIPELINE_POLICIES.promotedTarget.poll, + dispatchedFlags?.readinessTimeoutMs as number | undefined, + ); + expect(dispatchSchedule).toBeDefined(); + expect(await replayStepReadinessSchedule(REPLAY_REQUEST.flags, action)).toEqual(dispatchSchedule); +}); + test('replay keeps a readinessTimeoutMs the step already carries instead of overwriting it', async () => { const action: SessionAction = { ts: 0, diff --git a/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts b/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts index c822ffd05b..271f2546bc 100644 --- a/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts +++ b/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts @@ -2,6 +2,7 @@ import path from 'node:path'; import { mkdtempForTestSync } from '../../../__tests__/test-utils/tmp-dir.ts'; import { makeIosAppSession } from '../../../__tests__/test-utils/session-factories.ts'; import { SessionStore } from '../../session-store.ts'; +import type { ObservationClock } from '@agent-device/capture-kit/observe-until'; import type { DaemonInvokeFn, DaemonRequest, DaemonResponse } from '../../daemon-request.ts'; import { runReplayForTest } from '../replay-runtime/replay-command-fixture.ts'; import { @@ -33,9 +34,9 @@ export type ReplayScriptScene = ReplaySessionScene & /** * Runs the script through `runReplayCommand`. Each request is appended to * `invoked` before `invoke` answers it; without `invoke` every request succeeds - * with empty data. + * with empty data. `clock` paces its target-readiness waits. */ - replay(options?: { invoke?: DaemonInvokeFn }): Promise; + replay(options?: { invoke?: DaemonInvokeFn; clock?: ObservationClock }): Promise; }>; /** An iOS app session plus a written `.ad` script, ready to replay. */ @@ -47,9 +48,10 @@ export function replayScriptScene(prefix: string, lines: string[]): ReplayScript ...scene, filePath, invoked, - replay: ({ invoke } = {}) => + replay: ({ invoke, clock } = {}) => runReplayForTest({ req: baseReplayRequest({ positionals: [filePath] }), + ...(clock === undefined ? {} : { clock }), sessionName: scene.sessionName, logPath: scene.logPath, sessionStore: scene.sessionStore, diff --git a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts index 0bd1076de1..a57494e5cd 100644 --- a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts +++ b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts @@ -236,6 +236,36 @@ test('the dispatch gets only the readiness budget the gate left', async () => { expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(400); }); +test('the budget the gate hands the dispatch counts from the end of its first capture', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-budget-origin-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + let nowMs = 0; + const clock = { + now: () => nowMs, + sleep: async (ms: number) => { + nowMs += ms; + }, + }; + + // A 500 ms first capture, then seven more misses one interval apart: 1.4 s of the 2 s budget. + mockDispatchCommand.mockImplementationOnce(async () => { + nowMs += 500; + return emptyCapture(); + }); + for (let miss = 0; miss < 7; miss += 1) mockDispatchCommand.mockResolvedValueOnce(emptyCapture()); + mockDispatchCommand.mockResolvedValue(saveButtonCapture()); + + const response = await scene.replay({ clock }); + + expect(response.ok).toBe(true); + expect(readinessDiagnostics()).toEqual([ + { polls: 9, waitedMs: 2_100, end: 'done', command: 'click' }, + ]); + expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(400); +}); + test('an annotated step whose command does not wait for its target refuses a selector miss on one capture', async () => { const scene = replayScriptScene('agent-device-replay-target-verify-no-wait-', [ SAVE_ANNOTATION, From 1b93af6adcdc3ebf2177fb1ab60d31ce0125a6a2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 19:50:07 +0200 Subject: [PATCH 19/23] refactor(capture-kit): observeUntil defaults to no per-capture deadline; cancel is an opt-in captureDeadline is now optional and defaults to 'none'. A capture only ends the loop stalled when its caller declares 'cancel', so an adapter whose capture ignores the signal cannot claim cancellation by omission. The selector readiness poll keeps its explicit 'cancel': its poll signal reaches the platform as CaptureSnapshotInput.signal. The replay target gate drops its explicit 'none'. --- packages/capture-kit/src/observe-until.test.ts | 14 ++++++++------ packages/capture-kit/src/observe-until.ts | 5 +++-- .../session-replay-runtime-engine-adapter.ts | 4 +--- 3 files changed, 12 insertions(+), 11 deletions(-) diff --git a/packages/capture-kit/src/observe-until.test.ts b/packages/capture-kit/src/observe-until.test.ts index a1bdde49e2..41cff27d1a 100644 --- a/packages/capture-kit/src/observe-until.test.ts +++ b/packages/capture-kit/src/observe-until.test.ts @@ -21,7 +21,9 @@ function fakeClock(): ObservationClock & { advance(ms: number): void; slept: num }; } -const SCHEDULE = { intervalMs: 200, budgetMs: 1_000, captureDeadline: 'cancel' } as const; +/** No per-capture deadline: the engine default. */ +const UNBOUNDED_SCHEDULE = { intervalMs: 200, budgetMs: 1_000 } as const; +const SCHEDULE = { ...UNBOUNDED_SCHEDULE, captureDeadline: 'cancel' } as const; /** The Android helper's content verdict: the capture ran but held no readable app content. */ function unreadableContent(): AppError { @@ -165,7 +167,7 @@ describe('observeUntil', () => { previous !== undefined ? { kind: 'done', result: [previous, latest] } : { kind: 'continue' }, - schedule: { intervalMs: 200, budgetMs: 1_000, minPolls: 2, captureDeadline: 'none' }, + schedule: { ...UNBOUNDED_SCHEDULE, minPolls: 2 }, clock, }); assert.equal(observed.kind, 'done'); @@ -262,12 +264,12 @@ describe('observeUntil captureDeadline', () => { }; } - test("'none' judges a capture that finishes past the budget and can end done", async () => { + test('by default judges a capture that finishes past the budget and can end done', async () => { const clock = fakeClock(); const observed = await observeUntil({ capture: lateSecondCapture(clock), verdict: (latest) => (latest === 2 ? { kind: 'done', result: latest } : { kind: 'continue' }), - schedule: { ...SCHEDULE, captureDeadline: 'none' }, + schedule: UNBOUNDED_SCHEDULE, clock, }); assert.equal(observed.kind, 'done'); @@ -277,12 +279,12 @@ describe('observeUntil captureDeadline', () => { ); }); - test("'none' ends expired, never stalled, when the late capture's verdict continues", async () => { + test("by default ends expired, never stalled, when the late capture's verdict continues", async () => { const clock = fakeClock(); const observed = await observeUntil({ capture: lateSecondCapture(clock), verdict: () => ({ kind: 'continue' }), - schedule: { ...SCHEDULE, captureDeadline: 'none' }, + schedule: UNBOUNDED_SCHEDULE, clock, }); assert.equal(observed.kind, 'expired'); diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts index b6138ba61f..6730be3842 100644 --- a/packages/capture-kit/src/observe-until.ts +++ b/packages/capture-kit/src/observe-until.ts @@ -43,9 +43,10 @@ export type ObservationSchedule = Readonly<{ * remaining budget (at least one interval) as an abort signal, and a capture that ends at or past * that deadline ends the loop `stalled`: declare it only when the capture honors the signal. * `'none'` hands the capture no deadline; a capture that finishes past the budget is still judged, - * so the loop ends `done` on an accepting verdict and `expired` otherwise. + * so the loop ends `done` on an accepting verdict and `expired` otherwise. Default `'none'`, so a + * capture that ignores the signal cannot end the loop `stalled` by omission. */ - captureDeadline: 'cancel' | 'none'; + captureDeadline?: 'cancel' | 'none'; }>; export type ObservationClock = Readonly<{ diff --git a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts index f277bab00e..d46a0cdf09 100644 --- a/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts +++ b/packages/replay-port/src/daemon-port/session-replay-runtime-engine-adapter.ts @@ -235,9 +235,7 @@ export function createAdReplayStepRuntime(params: { capture: observeOnce, verdict: (latest) => isTargetNotRenderedYet(latest) ? { kind: 'continue' } : { kind: 'done', result: latest }, - // The divergence capture takes no per-capture signal, so a late - // capture is judged rather than cancelled. - schedule: { ...readiness, captureDeadline: 'none' }, + schedule: readiness, ...(ctx.signal ? { signal: ctx.signal } : {}), ...(ctx.dependencies.clock ? { clock: ctx.dependencies.clock } : {}), }); From b0921253b4ea1e27f986cd837b68eb6598ebd431 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 19:53:57 +0200 Subject: [PATCH 20/23] fix(replay): keep a thrown error's typed details on the replay response An error thrown out of the replay run reached the wire as code and message only; the catch rebuilt it with errorResponse and dropped details, hint, and diagnostic references. A request canceled during the pre-dispatch target gate therefore lost its request_canceled reason. The catch now normalizes the error with normalizeError and appends the run's artifactPaths to its details. --- .../src/daemon-port/native-command.ts | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/packages/replay-port/src/daemon-port/native-command.ts b/packages/replay-port/src/daemon-port/native-command.ts index 50ba57430a..80b28f2ab3 100644 --- a/packages/replay-port/src/daemon-port/native-command.ts +++ b/packages/replay-port/src/daemon-port/native-command.ts @@ -1,4 +1,4 @@ -import { asAppError } from '@agent-device/kernel/errors'; +import { normalizeError } from '@agent-device/kernel/errors'; import { runAdReplay } from '@agent-device/ad-replay'; import type { SnapshotTimingSample } from '@agent-device/contracts/capture'; import { summarizeSnapshotTimingSamples } from '@agent-device/contracts/capture'; @@ -183,12 +183,15 @@ export async function runReplayCommand(command: ReplayCommand): Promise 0 ? { artifactPaths: [...artifactPaths] } : undefined, - ); + const normalized = normalizeError(error); + if (artifactPaths.size === 0) return { ok: false, error: normalized }; + return { + ok: false, + error: { + ...normalized, + details: { ...normalized.details, artifactPaths: [...artifactPaths] }, + }, + }; } } From 9c76e9a3d878c457970d8a7ac2e460b4c4c0898e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 19:53:57 +0200 Subject: [PATCH 21/23] test(replay): cancelling the readiness gate ends at the next poll The target gate's divergence capture takes no signal, so the gate honors a canceled request only at a poll boundary. A request canceled during the gate's second capture ends the wait when that capture returns: two captures, the request_canceled reason on the response, and no dispatch. --- .../session-replay-scenario.fixtures.ts | 16 +++++++--- ...replay-target-verification-runtime.test.ts | 31 +++++++++++++++++++ 2 files changed, 43 insertions(+), 4 deletions(-) diff --git a/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts b/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts index 271f2546bc..f129fc90fb 100644 --- a/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts +++ b/src/daemon/__tests__/replay-target/session-replay-scenario.fixtures.ts @@ -34,9 +34,14 @@ export type ReplayScriptScene = ReplaySessionScene & /** * Runs the script through `runReplayCommand`. Each request is appended to * `invoked` before `invoke` answers it; without `invoke` every request succeeds - * with empty data. `clock` paces its target-readiness waits. + * with empty data. `clock` paces its target-readiness waits; `requestId` lets a test cancel the + * replay through the request registry. */ - replay(options?: { invoke?: DaemonInvokeFn; clock?: ObservationClock }): Promise; + replay(options?: { + invoke?: DaemonInvokeFn; + clock?: ObservationClock; + requestId?: string; + }): Promise; }>; /** An iOS app session plus a written `.ad` script, ready to replay. */ @@ -48,9 +53,12 @@ export function replayScriptScene(prefix: string, lines: string[]): ReplayScript ...scene, filePath, invoked, - replay: ({ invoke, clock } = {}) => + replay: ({ invoke, clock, requestId } = {}) => runReplayForTest({ - req: baseReplayRequest({ positionals: [filePath] }), + req: baseReplayRequest({ + positionals: [filePath], + ...(requestId === undefined ? {} : { meta: { requestId } }), + }), ...(clock === undefined ? {} : { clock }), sessionName: scene.sessionName, logPath: scene.logPath, diff --git a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts index a57494e5cd..ddffb39004 100644 --- a/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts +++ b/src/daemon/__tests__/replay-target/session-replay-target-verification-runtime.test.ts @@ -34,6 +34,11 @@ vi.mock('@agent-device/host-kit/diagnostics', async (importOriginal) => { }); import { AppError } from '@agent-device/kernel/errors'; +import { + clearRequestCanceled, + markRequestCanceled, + registerRequestAbort, +} from '@agent-device/host-kit/request'; import { emitDiagnostic } from '@agent-device/host-kit/diagnostics'; import { legacyDispatchCapture, @@ -266,6 +271,32 @@ test('the budget the gate hands the dispatch counts from the end of its first ca expect(scene.invoked[0]?.flags?.readinessTimeoutMs).toBe(400); }); +test('cancelling the request during the gate wait ends it at the next poll with no dispatch', async () => { + const scene = replayScriptScene('agent-device-replay-target-verify-gate-cancel-', [ + SAVE_ANNOTATION, + 'click id="save"', + ]); + const requestId = 'replay-gate-cancel'; + const registration = registerRequestAbort(requestId); + mockDispatchCommand.mockResolvedValueOnce(emptyCapture()).mockImplementationOnce(async () => { + markRequestCanceled(requestId); + return emptyCapture(); + }); + mockDispatchCommand.mockResolvedValue(saveButtonCapture()); + + try { + const response = await scene.replay({ requestId }); + + expect(response.ok).toBe(false); + if (response.ok) return; + expect(response.error.details?.reason).toBe('request_canceled'); + expect(mockDispatchCommand).toHaveBeenCalledTimes(2); + expect(scene.invoked).toEqual([]); + } finally { + clearRequestCanceled(requestId, registration); + } +}); + test('an annotated step whose command does not wait for its target refuses a selector miss on one capture', async () => { const scene = replayScriptScene('agent-device-replay-target-verify-no-wait-', [ SAVE_ANNOTATION, From fa7f36af7d19d8d162adfb8bdc409ecfb5f0a4b9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 19:54:18 +0200 Subject: [PATCH 22/23] docs(replay): state that the readiness clock is test injection like AgentDeviceRuntime.clock --- packages/replay-port/src/daemon-port/command-types.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/replay-port/src/daemon-port/command-types.ts b/packages/replay-port/src/daemon-port/command-types.ts index 9082fc1be5..e34948b0f2 100644 --- a/packages/replay-port/src/daemon-port/command-types.ts +++ b/packages/replay-port/src/daemon-port/command-types.ts @@ -128,7 +128,10 @@ export type ReplayInvoke = (request: ReplayDispatchRequest) => Promise SessionScope; - /** The time source the pre-dispatch target-readiness wait paces itself by. Absent: wall clock. */ + /** + * The time source the pre-dispatch target-readiness wait paces itself by. Absent: wall clock. + * The daemon never sets it; tests inject one, as `AgentDeviceRuntime.clock` does for the dispatch. + */ clock?: ObservationClock; }>; From d476097d601c5283dd76e78e95fbcb02d6f72249 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Pierzcha=C5=82a?= Date: Wed, 30 Sep 2026 20:20:45 +0200 Subject: [PATCH 23/23] fix(capture-kit): a poll never starts after the signal aborted observeUntil checked the caller's signal only at the top of each iteration, before the sleep. A cancel that arrived during the sleep still paid one more capture before the loop noticed. The loop now checks the signal after the sleep and ends with the typed request-canceled error before starting a poll. The host-kit sleep takes no signal, so the check after the sleep is the boundary. --- .../capture-kit/src/observe-until.test.ts | 28 +++++++++++++++++++ packages/capture-kit/src/observe-until.ts | 1 + 2 files changed, 29 insertions(+) diff --git a/packages/capture-kit/src/observe-until.test.ts b/packages/capture-kit/src/observe-until.test.ts index 41cff27d1a..3a1426e622 100644 --- a/packages/capture-kit/src/observe-until.test.ts +++ b/packages/capture-kit/src/observe-until.test.ts @@ -210,6 +210,34 @@ describe('observeUntil', () => { }); }); +describe('observeUntil cancellation', () => { + test('a signal aborted during the sleep ends the loop canceled before another capture', async () => { + const clock = fakeClock(); + const controller = new AbortController(); + let captures = 0; + await assert.rejects( + observeUntil({ + capture: async () => { + captures += 1; + return captures; + }, + verdict: () => ({ kind: 'continue' }), + schedule: UNBOUNDED_SCHEDULE, + signal: controller.signal, + clock: { + now: clock.now, + sleep: async (ms) => { + await clock.sleep(ms); + controller.abort(); + }, + }, + }), + (error) => error instanceof AppError && error.details?.reason === 'request_canceled', + ); + assert.equal(captures, 1); + }); +}); + describe('observeUntil budgetFrom first-capture', () => { test('never bounds the first capture and spends the budget only on retries', async () => { const clock = fakeClock(); diff --git a/packages/capture-kit/src/observe-until.ts b/packages/capture-kit/src/observe-until.ts index 6730be3842..89d7bdfbc9 100644 --- a/packages/capture-kit/src/observe-until.ts +++ b/packages/capture-kit/src/observe-until.ts @@ -149,6 +149,7 @@ export async function observeUntil( if (params.signal?.aborted) throw createRequestCanceledError(); const expired = await waitBeforePoll(loop); if (expired) return expired; + if (params.signal?.aborted) throw createRequestCanceledError(); const ended = await pollOnce(loop); if (ended) return ended; }