Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions packages/command-registry/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,10 @@
"types": "./src/flag-groups.ts",
"default": "./src/flag-groups.ts"
},
"./flag-definitions-workflow": {
"types": "./src/flag-definitions-workflow.ts",
"default": "./src/flag-definitions-workflow.ts"
},
"./command-text": {
"types": "./src/command-text.ts",
"default": "./src/command-text.ts"
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,7 @@ test('every command that deviates from require-owner is a reviewed, diffable set
],
observe: ['apps', 'appstate', 'capabilities', 'device', 'devices', 'doctor', 'takeover'],
none: [
'act',
'artifacts',
'auth',
'batch',
Expand All @@ -74,6 +75,7 @@ test('every command that deviates from require-owner is a reviewed, diffable set
'session',
'session_list',
'session_save_script',
'suggest',
'web',
],
});
Expand Down
51 changes: 50 additions & 1 deletion packages/command-registry/src/flag-definitions-workflow.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,24 @@
import { SCREENSHOT_SPECIFIC_FLAG_DEFINITIONS } from '@agent-device/contracts/capture';
import type { FlagDefinition } from './flag-types.ts';

/**
* The policy head's CLI vocabulary and defaults.
*
* They live beside the flags that publish them so the prose and the runtime read one declaration:
* `--max-steps` and `--min-confidence` state these numbers in their help, and the act loop applies
* them. Kept here rather than in the policy modules because the CLI surface is evaluated at
* startup while the policy runtime loads on demand, and a constant may not drag the second into
* the first.
*/
export const POLICY_PROVIDER_NAMES = ['jev'] as const;
export const DEFAULT_POLICY_PROVIDER: (typeof POLICY_PROVIDER_NAMES)[number] = 'jev';
export const DEFAULT_POLICY_MAX_STEPS = 12;
export const DEFAULT_POLICY_MIN_CONFIDENCE = 0.4;
/** Environment entry the jev provider reads its credential from; never a flag. */
export const POLICY_API_KEY_ENV = 'TYPESAFE_API_KEY';
/** Prefix for per-key text entries, so a secret never has to appear in argv. */
export const POLICY_INPUT_ENV_PREFIX = 'AGENT_DEVICE_INPUT_';

export const WORKFLOW_FLAG_DEFINITIONS: readonly FlagDefinition[] = [
{
key: 'replayUpdate',
Expand Down Expand Up @@ -195,7 +213,7 @@ export const WORKFLOW_FLAG_DEFINITIONS: readonly FlagDefinition[] = [
min: 1,
max: 1000,
usageLabel: '--max-steps <n>',
usageDescription: 'Batch: maximum number of allowed steps',
usageDescription: `Batch: maximum allowed steps; act: stop the loop after this many steps (default ${DEFAULT_POLICY_MAX_STEPS})`,
projectConfig: true,
recorded: false,
},
Expand Down Expand Up @@ -322,6 +340,37 @@ export const WORKFLOW_FLAG_DEFINITIONS: readonly FlagDefinition[] = [
projectConfig: false,
recorded: false,
},
{
key: 'policy',
names: ['--policy'],
type: 'string',
usageLabel: '--policy <name>',
usageDescription: `suggest/act: policy head that decides the next element (default ${DEFAULT_POLICY_PROVIDER})`,
projectConfig: true,
recorded: false,
},
{
key: 'minConfidence',
names: ['--min-confidence'],
type: 'number',
usageLabel: '--min-confidence <n>',
usageDescription: `act: escalate instead of acting below this policy confidence (default ${DEFAULT_POLICY_MIN_CONFIDENCE})`,
projectConfig: false,
recorded: false,
},
{
key: 'policyInput',
names: ['--input'],
type: 'string',
multiple: true,
usageLabel: '--input <key=value>',
usageDescription:
'act: repeatable text the loop may enter, matched to a field by identifier or label; the loop never generates text',
inputDescription:
'Text the loop may enter, as key=value. The key matches a field identifier or label.',
projectConfig: false,
recorded: false,
},
{
key: 'overlayRefs',
names: ['--overlay-refs'],
Expand Down
36 changes: 36 additions & 0 deletions packages/command-registry/src/registry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1538,6 +1538,42 @@ export const RAW_COMMAND_DESCRIPTORS = [
},

// -- local client-backed CLI/MCP commands (no daemon route/capability) --
// The policy commands compose `snapshot`, `press`, and `fill` from the client process, so they
// own no daemon route of their own and stay out of the public catalog, which requires one.
{
name: 'suggest',
deviceClaimPolicy: 'none',
...(ownerFilesEnabled ? { ownerFiles: ['src/commands/interaction/index.ts'] as const } : {}),
catalog: { group: 'local-cli' },
recordsSessionAction: false,
timeoutPolicy: DEFAULT_TIMEOUT_POLICY,
batchable: false,
platformExecution: NO_PLATFORM_EXECUTION,
},
{
name: 'act',
deviceClaimPolicy: 'none',
...(ownerFilesEnabled
? {
ownerFiles: [
'src/commands/interaction/index.ts',
'src/commands/policy/act-loop.ts',
] as const,
}
: {}),
catalog: { group: 'local-cli' },
recordsSessionAction: false,
// A run is a sequence of ordinary commands, each already under its own envelope; a client
// envelope over the whole loop would abort a run that is still making progress. `--max-steps`
// is the budget that bounds it.
timeoutPolicy: {
...DEFAULT_TIMEOUT_POLICY,
envelopeMs: 'unbounded',
onTimeout: 'preserve-daemon',
},
batchable: false,
platformExecution: NO_PLATFORM_EXECUTION,
},
{
name: 'debug',
deviceClaimPolicy: 'none',
Expand Down
6 changes: 6 additions & 0 deletions packages/contracts/src/cli-flags.ts
Original file line number Diff line number Diff line change
Expand Up @@ -160,6 +160,12 @@ export type CliFlags = CloudProviderProfileFields &
runtime?: SessionRuntimeHints;
}>;
out?: string;
/** suggest/act: which policy head answers "what next". */
policy?: string;
/** act: escalate instead of acting below this confidence. */
minConfidence?: number;
/** act: repeated `key=value` text the loop may enter; it never generates text itself. */
policyInput?: string[];
help: boolean;
version: boolean;
};
Expand Down
3 changes: 3 additions & 0 deletions scripts/integration-progress-model.ts
Original file line number Diff line number Diff line change
Expand Up @@ -326,6 +326,9 @@ function summarizeProviderScenarioFlagExclusions() {
'shardAll',
'shardSplit',
'searchPath',
'policy',
'minConfidence',
'policyInput',
'stepsFile',
'proxyHost',
'proxyPort',
Expand Down
2 changes: 2 additions & 0 deletions skills/agent-device/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -19,4 +19,6 @@ Reaching an off-screen target is one command, not a scroll-and-check loop: `scro

Copy refs byte-for-byte: `@e12`, `@e12~s4` — keep the `@` and any `~sN`. Prefer current refs, then `id`/`label`/`role` selectors; coordinates are a last resort. If snapshot reports sparse/AX-unavailable, its refs and selectors are invalid: run `agent-device screenshot`, inspect the image, use coordinates, then retry `snapshot -i` after navigating. Otherwise run `snapshot -i` only when the diff lacks the next target.

With `TYPESAFE_API_KEY` set, `suggest "<goal>"` names the next element to act on in roughly the time a snapshot takes, and `act "<goal>" --input key=value` runs the whole snapshot-decide-act loop. Text is never generated; a field with no `--input` entry escalates. Both are always listed; without the key each refuses with a typed error and the loop above is the path. On `Jev unavailable: <status>`, fall back to that loop.

Error output includes corrective hints; follow them instead of re-planning. Only when the task is specialized (for example gestures, scripting, TV, macOS, remote, or debugging) or a command shape is unclear, run `agent-device help <topic>`. `agent-device --help` lists topics, but is not a startup step.
5 changes: 2 additions & 3 deletions src/__tests__/cli-client-commands.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1168,10 +1168,8 @@ function createStubClient(params: {
return {
command,
devices: {
...createThrowingMethodGroup<AgentDeviceClient['devices']>(),
list: async () => [],
capabilities: unexpectedCommandCall,
boot: unexpectedCommandCall,
shutdown: unexpectedCommandCall,
},
sessions: {
list: async () => [],
Expand Down Expand Up @@ -1272,6 +1270,7 @@ function createStubClient(params: {
events: params.events ?? unexpectedCommandCall,
},
debug: createThrowingMethodGroup<AgentDeviceClient['debug']>(),
policy: createThrowingMethodGroup<AgentDeviceClient['policy']>(),
recording: createThrowingMethodGroup<AgentDeviceClient['recording']>(),
settings: {
update: params.updateSettings ?? unexpectedCommandCall,
Expand Down
7 changes: 7 additions & 0 deletions src/__tests__/command-descriptor-timeout-policy.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -67,10 +67,14 @@ test('daemon-preserving timeout commands are a bounded, reviewed set', () => {
// sessions the daemon owns, so a client-side timeout must not SIGKILL the
// daemon mid-create/mid-release and orphan them (and every other provider
// session held).
// `act` joined with the policy head: a run is a sequence of ordinary commands, each already
// under its own envelope, so a client-side timeout over the whole loop must not reset the
// daemon and destroy the session the remaining steps still need.
const preserving = commandDescriptors
.filter((descriptor) => descriptor.timeoutPolicy.onTimeout === 'preserve-daemon')
.map((descriptor) => descriptor.name);
assert.deepEqual(preserving.sort(), [
'act',
'back',
'click',
'fill',
Expand Down Expand Up @@ -146,6 +150,9 @@ test('request envelopes deviating from the default are bounded, reviewed sets',
// #1774: base allocation budget (300s) + client/daemon race margin (30s).
lease_allocate: 330_000,
test: 'unbounded',
// A policy run is bounded by --max-steps, not by wall clock; each step it issues carries its
// own envelope, and an outer one would abort a run that is still making progress.
act: 'unbounded',
};
for (const descriptor of commandDescriptors) {
const expected = EXPECTED_ENVELOPES[descriptor.name] ?? 90_000;
Expand Down
1 change: 1 addition & 0 deletions src/__tests__/remote-connection.fixtures.ts
Original file line number Diff line number Diff line change
Expand Up @@ -116,6 +116,7 @@ export function createTestClient(
batch: createThrowingMethodGroup<AgentDeviceClient['batch']>(),
observability: createThrowingMethodGroup<AgentDeviceClient['observability']>(),
debug: createThrowingMethodGroup<AgentDeviceClient['debug']>(),
policy: createThrowingMethodGroup<AgentDeviceClient['policy']>(),
recording: createThrowingMethodGroup<AgentDeviceClient['recording']>(),
settings: createThrowingMethodGroup<AgentDeviceClient['settings']>(),
};
Expand Down
40 changes: 40 additions & 0 deletions src/agent-device-client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,7 @@ import {
resolveSessionName,
} from './client/client-normalizers.ts';
import type { AgentDeviceClient, MetroPrepareResult } from './client/client-types.ts';
import type { PolicyClientCalls } from './client/client-policy.ts';
import { INTERNAL_COMMANDS } from '@agent-device/command-registry/catalog';
import { buildRequestFlags } from './commands/command-flags.ts';
import {
Expand Down Expand Up @@ -135,6 +136,24 @@ export function createAgentDeviceClient(
const resolveRequestSession = (options: InternalRequestOptions = {}) =>
resolveSessionName(mergeClientOptions(config, options).session);

/**
* The three commands the policy loop composes. They are the ordinary daemon commands, so a
* policy-driven run carries the same claims, ref frames, and recording behaviour as a run an
* agent drives by hand.
*/
const policyClientCalls: PolicyClientCalls = {
snapshot: async (options) => {
const session = resolveRequestSession(options);
const data = await executeCommand<Record<string, unknown>>('snapshot', options);
const result = normalizeSnapshotResult(data, session);
return { nodes: result.nodes, refsGeneration: result.refsGeneration };
},
press: async ({ ref, ...options }) =>
await executeCommand('press', { ...options, ...refTarget(ref) }),
fill: async ({ ref, text, ...options }) =>
await executeCommand('fill', { ...options, ...refTarget(ref, { text }) }),
};

return {
command: {
wait: async (options) => await executeCommand<CommandResult<'wait'>>('wait', options),
Expand Down Expand Up @@ -423,6 +442,18 @@ export function createAgentDeviceClient(
return symbolicateCrashArtifact({ cwd: options.cwd ?? config.cwd, ...options });
},
},
policy: {
// Loaded on demand: the policy head is optional, and nothing about it belongs in the
// module closure every CLI invocation evaluates at startup.
suggest: async (options) => {
const { suggestPolicyAction } = await import('./client/client-policy.ts');
return await suggestPolicyAction(policyClientCalls, options, process.env);
},
act: async (options) => {
const { runPolicyAct } = await import('./client/client-policy.ts');
return await runPolicyAct(policyClientCalls, options, process.env);
},
},
recording: {
record: async (options) => await executeCommand<CommandResult<'record'>>('record', options),
trace: async (options) => await executeCommand<CommandResult<'trace'>>('trace', options),
Expand All @@ -433,6 +464,15 @@ export function createAgentDeviceClient(
};
}

/**
* An `@ref` interaction target plus any command-specific fields, in the shape `executeCommand`
* forwards to the daemon. The request options type is deliberately narrow; interaction fields
* reach the wire through the same untyped bag every other interaction call uses.
*/
function refTarget(ref: string, extra: Record<string, unknown> = {}): InternalRequestOptions {
return { ref, ...extra } as InternalRequestOptions;
}

function panGestureInput(options: PanOptions): InternalRequestOptions & Record<string, unknown> {
const { x, y, dx, dy, ...common } = options;
return { ...common, kind: 'pan', origin: { x, y }, delta: { x: dx, y: dy } };
Expand Down
Loading