diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b645f4b8..4f41900f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -194,6 +194,27 @@ jobs: - name: Tests run: npm test + dev-tools-package-tests: + name: Dev Tools Package Tests + runs-on: ubuntu-latest + defaults: + run: + working-directory: packages/dev-tools + steps: + - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 + + - uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6 + with: + node-version: 22.21.0 + cache: npm + cache-dependency-path: packages/dev-tools/package-lock.json + + - name: Install dependencies + run: npm ci + + - name: Tests + run: npm test + macos-storage-tests: name: macOS Storage ACL Tests runs-on: macos-14 diff --git a/README.md b/README.md index 3856b254..29647586 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,10 @@ Code Interpreter (internally `codeapi`, the prefix used by its env vars, images, development - **Remote Code Bridge** - Lets an operator-owned VM connect outbound and serve as a fenced, stateful sandbox through the `@librechat/code` worker +- **Coding Tool Definitions** - The `@librechat/dev-tools` npm package: the + LLM-visible surface (canonical names, schemas, descriptions) of the + harness-native coding tools, shared by `@librechat/agents` and LibreChat + provisioning ## Architecture diff --git a/packages/dev-tools/README.md b/packages/dev-tools/README.md new file mode 100644 index 00000000..660721a1 --- /dev/null +++ b/packages/dev-tools/README.md @@ -0,0 +1,81 @@ +# `@librechat/dev-tools` + +The LLM-visible surface of the harness-native coding tools: the canonical +tool names, JSON schemas, and descriptions the model sees — `execute_code`, +`bash_tool`, `run_tools_with_code`, `run_tools_with_bash`, `read_file`, and +the local-engine file/edit/search tools (`read_file`, `write_file`, +`edit_file`, `grep_search`, `glob_search`, `list_directory`, +`compile_check`). + +This package is the single versioned source of truth for that surface. The +schemas describe the execution environments this service provides, so they +live next to the Code API rather than inside the agent harness: +`@librechat/agents` and LibreChat (BYOM provisioning) both depend on the +definitions without the harness owning them. + +## What is here + +| Module | Contents | +| --- | --- | +| `constants` | Canonical `ToolNames`, `CONTENT_AND_ARTIFACT`, and the `CODE_EXECUTION_TOOLS` / `LOCAL_CODING_TOOL_NAMES` / `LOCAL_CODING_BUNDLE_NAMES` sets | +| `types` | `JsonSchemaType`, `ToolDefinition` (mirrors the agents SDK's `LCTool`), `AllowedCaller`, `OutcomePatch` | +| `intent` | The intent-label contract embedded in every schema: `INTENT_PROPERTY`, `withIntent`/`withoutIntent`, arg readers/strippers, and outcome resolution | +| `guidance` | Shared `/mnt/data` and bash guidance embedded across schemas and descriptions | +| `timeout` | The programmatic-run `timeout` schema with environment-resolved defaults and clamping | +| `execute-code`, `bash-tool`, `read-file` | Remote (Code API) engine tool surfaces, including the stateful and attached-workspace description/schema builders | +| `run-tools-with-code`, `run-tools-with-bash` | Remote programmatic tool calling surfaces, including attached-workspace builders | +| `local` | Local-engine surfaces: file/edit/search tools, `compile_check`, the local execution descriptions, and the local programmatic tool calling schemas | + +Zero runtime dependencies: everything is plain data and pure functions, so +harness, host, and worker consumers pay nothing to read the schemas. + +## What is deliberately not here + +Execution. The Code API client, `ToolNode` event dispatch, the local +execution engine, tool-result replay, and output shaping (artifact-delivery +warnings, code-session file summaries) remain harness-native in +`@librechat/agents`. This package answers one question: what does the LLM +see? The sibling `@librechat/code` package answers the other side of +provisioning — the worker that turns an operator-owned VM into a stateful, +fenced execution environment. + +## Provenance + +Ported from `@librechat/agents` (`src/tools/`) with export-name parity so +the harness swap is mechanical. Intentional differences: + +- `Constants` tool-name members became `ToolNames` (the agents enum mixes + orchestration constants; only the coding-tool members moved). +- Schemas that were module-private in the harness (`CompileCheckSchema`, the + local programmatic tool calling schema builders) are exported here — the + package's purpose is sharing them. +- Local-engine descriptions that lived inline inside tool factories are + first-class `Local*ToolDescription` constants. +- `JsonSchemaType` is widened with the JSON-Schema keywords these schemas + use (`minLength`, `minimum`, `maximum`, `default`, `uniqueItems`) so every + schema type-checks as written. + +Every ported string was verified byte-for-byte against the harness source at +port time; descriptions are prompt surface, so drift is behavior change. + +## Usage + +```ts +import { + CodeExecutionToolDefinition, + LocalCodingBundleNames, + buildBashExecutionToolDescription, +} from '@librechat/dev-tools'; +``` + +Subpath exports mirror the module list above (`@librechat/dev-tools/local`, +`@librechat/dev-tools/intent`, and so on). Tests pin the canonical names, +required properties, intent-first property ordering, and the +stateful/attached description builders. + +## Development + +```bash +npm install +npm test +``` diff --git a/packages/dev-tools/package-lock.json b/packages/dev-tools/package-lock.json new file mode 100644 index 00000000..1566fad8 --- /dev/null +++ b/packages/dev-tools/package-lock.json @@ -0,0 +1,51 @@ +{ + "name": "@librechat/dev-tools", + "version": "0.1.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "@librechat/dev-tools", + "version": "0.1.0", + "license": "Apache-2.0", + "devDependencies": { + "@types/node": "^22.5.5", + "typescript": "^5.5.4" + }, + "engines": { + "node": ">=20.11" + } + }, + "node_modules/@types/node": { + "version": "22.20.4", + "resolved": "https://registry.npmjs.org/@types/node/-/node-22.20.4.tgz", + "integrity": "sha512-zJRE40jpHtKqE/C4fgHrAKQLJuSpzEnP9ff9Y7YtoR3Wd2pwqzlekDeEuUQXjRd+QCYnVnNwuJYmhdk9XV8gvA==", + "dev": true, + "license": "MIT", + "dependencies": { + "undici-types": "~6.21.0" + } + }, + "node_modules/typescript": { + "version": "5.9.3", + "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", + "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", + "dev": true, + "license": "Apache-2.0", + "bin": { + "tsc": "bin/tsc", + "tsserver": "bin/tsserver" + }, + "engines": { + "node": ">=14.17" + } + }, + "node_modules/undici-types": { + "version": "6.21.0", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-6.21.0.tgz", + "integrity": "sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==", + "dev": true, + "license": "MIT" + } + } +} diff --git a/packages/dev-tools/package.json b/packages/dev-tools/package.json new file mode 100644 index 00000000..dc4bd093 --- /dev/null +++ b/packages/dev-tools/package.json @@ -0,0 +1,75 @@ +{ + "name": "@librechat/dev-tools", + "version": "0.1.0", + "description": "LLM-visible definitions (names, schemas, descriptions) of the harness-native coding tools for LibreChat Code API", + "license": "Apache-2.0", + "type": "module", + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js" + }, + "./constants": { + "types": "./dist/constants.d.ts", + "import": "./dist/constants.js" + }, + "./types": { + "types": "./dist/types.d.ts", + "import": "./dist/types.js" + }, + "./intent": { + "types": "./dist/intent.d.ts", + "import": "./dist/intent.js" + }, + "./guidance": { + "types": "./dist/guidance.d.ts", + "import": "./dist/guidance.js" + }, + "./timeout": { + "types": "./dist/timeout.d.ts", + "import": "./dist/timeout.js" + }, + "./execute-code": { + "types": "./dist/execute-code.d.ts", + "import": "./dist/execute-code.js" + }, + "./bash-tool": { + "types": "./dist/bash-tool.d.ts", + "import": "./dist/bash-tool.js" + }, + "./read-file": { + "types": "./dist/read-file.d.ts", + "import": "./dist/read-file.js" + }, + "./run-tools-with-code": { + "types": "./dist/run-tools-with-code.d.ts", + "import": "./dist/run-tools-with-code.js" + }, + "./run-tools-with-bash": { + "types": "./dist/run-tools-with-bash.d.ts", + "import": "./dist/run-tools-with-bash.js" + }, + "./local": { + "types": "./dist/local/index.d.ts", + "import": "./dist/local/index.js" + } + }, + "files": [ + "dist", + "!dist/*.test.*" + ], + "scripts": { + "build": "tsc -p tsconfig.json", + "test": "npm run build && node --test dist/*.test.js", + "prepack": "npm run build" + }, + "devDependencies": { + "@types/node": "^22.5.5", + "typescript": "^5.5.4" + }, + "engines": { + "node": ">=20.11" + } +} diff --git a/packages/dev-tools/src/bash-tool.ts b/packages/dev-tools/src/bash-tool.ts new file mode 100644 index 00000000..f67050c2 --- /dev/null +++ b/packages/dev-tools/src/bash-tool.ts @@ -0,0 +1,196 @@ +/** + * `bash_tool` — the LLM-visible surface of the remote (Code API) bash + * execution tool: canonical name, schema, the stateless/stateful/ + * attached-workspace description variants, and the tool-output-references + * guide. Ported from `@librechat/agents` (`src/tools/BashExecutor.ts`); + * the execution client stays harness-native in the agents SDK. + */ + +import { ToolNames } from './constants.js'; +import { + BASH_SHELL_GUIDANCE, + CODE_ARTIFACT_PATH_GUIDANCE, +} from './guidance.js'; +import { INTENT_PROPERTY } from './intent.js'; + +export const BashExecutionToolSchema = { + type: 'object', + properties: { + intent: { ...INTENT_PROPERTY }, + command: { + type: 'string', + description: `The bash command or script to execute. +- The environment is stateless; variables and state don't persist between executions. +- Prior /mnt/data files are available and can be modified in place. +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- ${BASH_SHELL_GUIDANCE} +- Input code **IS ALREADY** displayed to the user, so **DO NOT** repeat it in your response unless asked. +- Output code **IS NOT** displayed to the user, so **DO** write all desired output explicitly. +- IMPORTANT: You MUST explicitly print/output ALL results you want the user to see. +- Use \`echo\`, \`printf\`, or \`cat\` for all outputs.`, + }, + args: { + type: 'array', + items: { type: 'string' }, + description: + 'Additional arguments to execute the command with. This should only be used if the input command requires additional arguments to run.', + }, + }, + required: ['command'], +} as const; + +export const BashExecutionToolDescription = ` +Runs bash commands and returns stdout/stderr output from a stateless execution environment, similar to running scripts in a command-line interface. Each execution is isolated and independent. + +Usage: +- No network access available. +- Generated files are automatically delivered; **DO NOT** provide download links. +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- ${BASH_SHELL_GUIDANCE} +- NEVER use this tool to execute malicious commands. +`.trim(); + +/** + * Bash statefulness is filesystem-tier and scoped to `/mnt/data`. The machine + * is warm across calls, but each call runs in a fresh sandbox (new process + * tree + private /tmp), so background processes are reaped when the call ends + * and anything written outside /mnt/data is discarded. The note must not + * promise otherwise: a model told background processes survive will start a + * server in one call and assume it is listening in the next. + */ +export const STATEFUL_BASH_NOTE = + 'Session state: commands in this conversation run on the same warm machine, so files written to /mnt/data persist between calls. Each call runs in a fresh, isolated sandbox: shell variables, the working directory, /tmp, and background processes do NOT survive after the call returns — a process started in one call is terminated when that call ends. Only /mnt/data is durable (the machine itself may also be reset at any time).'; + +export const StatefulBashExecutionToolDescription = ` +Runs bash commands and returns stdout/stderr output. Commands in this conversation share one warm machine with a persistent /mnt/data, but each command runs in its own isolated sandbox (not a persistent shell session). + +${STATEFUL_BASH_NOTE} + +Usage: +- No network access available. +- Generated files are automatically delivered; **DO NOT** provide download links. +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- ${BASH_SHELL_GUIDANCE} +- NEVER use this tool to execute malicious commands. +`.trim(); + +export const AttachedWorkspaceBashExecutionToolDescription = ` +Runs bash commands in the selected persistent project through an isolated sandbox process. + +Usage: +- Project file changes persist between calls; shell variables, background processes, and execution-private temporary files do not. +- Injected files and generated artifacts use \${LIBRECHAT_CODE_DATA_DIR:-/mnt/data}; write durable files to the project root. +- Generated artifacts are automatically delivered; **DO NOT** provide download links. +- ${BASH_SHELL_GUIDANCE} +- NEVER use this tool to execute malicious commands. +`.trim(); + +/** + * Supplemental prompt documenting the tool-output reference feature. + * + * Hosts should append this (separated by a blank line) to the base + * {@link BashExecutionToolDescription} only when + * `RunConfig.toolOutputReferences.enabled` is `true`. When the feature + * is disabled, including this text would tell the LLM to emit + * `{{tool0turn0}}` placeholders that pass through unsubstituted and + * leak into the shell. + */ +export const BashToolOutputReferencesGuide = ` +Referencing previous tool outputs: +- Every successful tool result is tagged with a reference key of the form \`toolturn\` (e.g., \`tool0turn0\`). The key appears either as a \`[ref: tool0turn0]\` prefix line or, when the output is a JSON object, as a \`_ref\` field on the object. +- To pipe a previous tool output into this tool, embed the placeholder \`{{toolturn}}\` literally anywhere in the \`command\` string (or any string arg). It will be substituted with the stored output verbatim before the command runs. +- The substituted value is the original output string (no \`[ref: …]\` prefix, no \`_ref\` key), so it is safe to pipe directly into \`jq\`, \`grep\`, \`awk\`, etc. +- Example (simple ASCII output): \`echo '{{tool0turn0}}' | jq '.foo'\` takes the full output of the first tool from the first turn and pipes it into jq. +- For payloads that may contain quotes, parentheses, backticks, or arbitrary bytes (random/binary data, JSON with embedded quotes, multi-line strings), prefer a quoted-delimiter heredoc over \`echo '…'\`. The heredoc body is not interpreted by the shell, so substituted payloads pass through unchanged. +- Heredoc example: \`wc -c << 'EOF'\\n{{tool0turn0}}\\nEOF\` (the quotes around \`'EOF'\` disable interpolation inside the body). +- Unknown reference keys are left in place and surfaced as \`[unresolved refs: …]\` after the output. +`.trim(); + +/** + * Composes the bash tool description, optionally appending the + * tool-output references guide. Hosts that enable + * `RunConfig.toolOutputReferences` should pass `enableToolOutputReferences: true` + * when registering the tool so the LLM learns the `{{…}}` syntax it + * will actually be able to use. + */ +export function buildBashExecutionToolDescription(options?: { + enableToolOutputReferences?: boolean; + statefulSessions?: boolean; + attachedWorkspace?: boolean; +}): string { + let base = BashExecutionToolDescription; + if (options?.attachedWorkspace === true) { + base = AttachedWorkspaceBashExecutionToolDescription; + } else if (options?.statefulSessions === true) { + base = StatefulBashExecutionToolDescription; + } + if (options?.enableToolOutputReferences === true) { + return `${base}\n\n${BashToolOutputReferencesGuide}`; + } + return base; +} + +const STATELESS_BASH_PARAM_NOTE = + "The environment is stateless; variables and state don't persist between executions."; +const STATEFUL_BASH_PARAM_NOTE = + 'Files written to /mnt/data persist between calls on the same warm machine. Each call runs in a fresh sandbox: shell variables, cwd, /tmp, and background processes do NOT survive the call. Only /mnt/data is durable.'; +const ATTACHED_BASH_PARAM_NOTE = + 'Commands start in the selected persistent project. Project file changes persist, but shell variables, background processes, and execution-private temporary files do not.'; +const ATTACHED_BASH_ARTIFACT_PATH_GUIDANCE = + 'Injected files and generated artifacts use `${LIBRECHAT_CODE_DATA_DIR:-/mnt/data}` for this execution only. Write anything needed later into the selected project.'; + +export function buildBashExecutionToolSchema(opts?: { + statefulSessions?: boolean; + attachedWorkspace?: boolean; +}): typeof BashExecutionToolSchema { + let note = STATELESS_BASH_PARAM_NOTE; + if (opts?.attachedWorkspace === true) { + note = ATTACHED_BASH_PARAM_NOTE; + } else if (opts?.statefulSessions === true) { + note = STATEFUL_BASH_PARAM_NOTE; + } + let commandDescription = + BashExecutionToolSchema.properties.command.description.replace( + STATELESS_BASH_PARAM_NOTE, + note + ); + if (opts?.attachedWorkspace === true) { + commandDescription = commandDescription + .replace( + '- Prior /mnt/data files are available and can be modified in place.\n', + '' + ) + .replace( + CODE_ARTIFACT_PATH_GUIDANCE, + ATTACHED_BASH_ARTIFACT_PATH_GUIDANCE + ); + } + return { + ...BashExecutionToolSchema, + properties: { + ...BashExecutionToolSchema.properties, + command: { + ...BashExecutionToolSchema.properties.command, + description: commandDescription, + }, + }, + } as typeof BashExecutionToolSchema; +} + +export const BashExecutionToolName = ToolNames.BASH_TOOL; + +/** + * Default bash tool definition using the base description. + * + * When `RunConfig.toolOutputReferences.enabled` is `true`, build a + * reference-aware description with + * {@link buildBashExecutionToolDescription} + * (`{ enableToolOutputReferences: true }`) and construct a custom + * definition using it — using this constant as-is leaves the LLM + * unaware of the `{{toolturn}}` syntax. + */ +export const BashExecutionToolDefinition = { + name: BashExecutionToolName, + description: BashExecutionToolDescription, + schema: BashExecutionToolSchema, +} as const; diff --git a/packages/dev-tools/src/constants.test.ts b/packages/dev-tools/src/constants.test.ts new file mode 100644 index 00000000..14dbe7d0 --- /dev/null +++ b/packages/dev-tools/src/constants.test.ts @@ -0,0 +1,64 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { + CODE_EXECUTION_TOOLS, + CONTENT_AND_ARTIFACT, + LOCAL_CODING_BUNDLE_NAMES, + LOCAL_CODING_TOOL_NAMES, + ToolNames, +} from './constants.js'; + +test('canonical tool names keep their wire-level string values', () => { + assert.equal(ToolNames.EXECUTE_CODE, 'execute_code'); + assert.equal(ToolNames.PROGRAMMATIC_TOOL_CALLING, 'run_tools_with_code'); + assert.equal(ToolNames.READ_FILE, 'read_file'); + assert.equal(ToolNames.BASH_TOOL, 'bash_tool'); + assert.equal( + ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING, + 'run_tools_with_bash' + ); + assert.equal(ToolNames.WRITE_FILE, 'write_file'); + assert.equal(ToolNames.EDIT_FILE, 'edit_file'); + assert.equal(ToolNames.GREP_SEARCH, 'grep_search'); + assert.equal(ToolNames.GLOB_SEARCH, 'glob_search'); + assert.equal(ToolNames.LIST_DIRECTORY, 'list_directory'); + assert.equal(ToolNames.COMPILE_CHECK, 'compile_check'); +}); + +test('response format constant keeps its wire-level value', () => { + assert.equal(CONTENT_AND_ARTIFACT, 'content_and_artifact'); +}); + +test('code execution tools are exactly the four sandbox-engine tools', () => { + assert.deepEqual([...CODE_EXECUTION_TOOLS].sort(), [ + 'bash_tool', + 'execute_code', + 'run_tools_with_bash', + 'run_tools_with_code', + ]); +}); + +test('local coding tool names are the local-only surface', () => { + assert.deepEqual(LOCAL_CODING_TOOL_NAMES, [ + ToolNames.READ_FILE, + ToolNames.WRITE_FILE, + ToolNames.EDIT_FILE, + ToolNames.GREP_SEARCH, + ToolNames.GLOB_SEARCH, + ToolNames.LIST_DIRECTORY, + ToolNames.COMPILE_CHECK, + ]); + assert.equal(LOCAL_CODING_TOOL_NAMES.length, 7); +}); + +test('local coding bundle names are the local tools plus the execution pair', () => { + assert.deepEqual(LOCAL_CODING_BUNDLE_NAMES, [ + ...LOCAL_CODING_TOOL_NAMES, + ToolNames.BASH_TOOL, + ToolNames.EXECUTE_CODE, + ToolNames.PROGRAMMATIC_TOOL_CALLING, + ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING, + ]); + assert.equal(LOCAL_CODING_BUNDLE_NAMES.length, 11); +}); diff --git a/packages/dev-tools/src/constants.ts b/packages/dev-tools/src/constants.ts new file mode 100644 index 00000000..8be7e412 --- /dev/null +++ b/packages/dev-tools/src/constants.ts @@ -0,0 +1,71 @@ +/** + * Canonical names of the coding tools the LLM sees, plus the tool-name sets + * the harness keys behavior off. Ported from `Constants` in + * `@librechat/agents` (`src/common/enum.ts`) with the same member names so + * an import swap is mechanical; only the coding-tool members live here. + * + * The string values are wire-level tool names — consumer UIs (most + * importantly LibreChat's `getToolIconType`) match against them, so a rename + * here is a breaking change for every consumer. + */ + +/** Canonical coding tool names as the LLM and consumer UIs see them. */ +export enum ToolNames { + EXECUTE_CODE = 'execute_code', + PROGRAMMATIC_TOOL_CALLING = 'run_tools_with_code', + READ_FILE = 'read_file', + BASH_TOOL = 'bash_tool', + BASH_PROGRAMMATIC_TOOL_CALLING = 'run_tools_with_bash', + WRITE_FILE = 'write_file', + EDIT_FILE = 'edit_file', + GREP_SEARCH = 'grep_search', + GLOB_SEARCH = 'glob_search', + LIST_DIRECTORY = 'list_directory', + COMPILE_CHECK = 'compile_check', +} + +/** Response format for tools that return content plus a structured artifact. */ +export const CONTENT_AND_ARTIFACT = 'content_and_artifact'; + +/** Tool names that use the code execution environment (shared session, file tracking). */ +export const CODE_EXECUTION_TOOLS: ReadonlySet = new Set([ + ToolNames.EXECUTE_CODE, + ToolNames.BASH_TOOL, + ToolNames.PROGRAMMATIC_TOOL_CALLING, + ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING, +]); + +/** + * Canonical names of the local-engine-specific coding tools — the + * file/edit/search/typecheck surface that doesn't exist in the remote + * (sandbox-API) engine. Single source of truth; the per-tool definitions and + * the harness workspace-policy defaults all key off these. + * + * `read_file` is on this list (the remote ReadFile tool is skill/execution + * output oriented; the local engine's `read_file` is a parallel + * implementation that shares the canonical name so consumer UIs render both + * with the same icon). + */ +export const LOCAL_CODING_TOOL_NAMES: readonly string[] = [ + ToolNames.READ_FILE, + ToolNames.WRITE_FILE, + ToolNames.EDIT_FILE, + ToolNames.GREP_SEARCH, + ToolNames.GLOB_SEARCH, + ToolNames.LIST_DIRECTORY, + ToolNames.COMPILE_CHECK, +]; + +/** + * Every tool name the local coding bundle exposes — the local-specific tools + * above plus the bash/code/PTC pair that the local engine wraps around the + * remote factories. Any addition/removal in the bundle must be accompanied + * by a deliberate canonical-name update here. + */ +export const LOCAL_CODING_BUNDLE_NAMES: readonly string[] = [ + ...LOCAL_CODING_TOOL_NAMES, + ToolNames.BASH_TOOL, + ToolNames.EXECUTE_CODE, + ToolNames.PROGRAMMATIC_TOOL_CALLING, + ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING, +]; diff --git a/packages/dev-tools/src/execute-code.ts b/packages/dev-tools/src/execute-code.ts new file mode 100644 index 00000000..e4664cd1 --- /dev/null +++ b/packages/dev-tools/src/execute-code.ts @@ -0,0 +1,140 @@ +/** + * `execute_code` — the LLM-visible surface of the remote (Code API) code + * execution tool: canonical name, schema, descriptions, and the + * stateful-session description/schema builders. Ported from + * `@librechat/agents` (`src/tools/CodeExecutor.ts`); the execution client + * itself stays harness-native in the agents SDK. + */ + +import { ToolNames } from './constants.js'; +import { CODE_ARTIFACT_PATH_GUIDANCE } from './guidance.js'; +import { INTENT_PROPERTY } from './intent.js'; + +export const SUPPORTED_LANGUAGES = [ + 'py', + 'js', + 'ts', + 'c', + 'cpp', + 'java', + 'php', + 'rs', + 'go', + 'd', + 'f90', + 'r', + 'bash', +] as const; + +export const CodeExecutionToolSchema = { + type: 'object', + properties: { + intent: { ...INTENT_PROPERTY }, + lang: { + type: 'string', + enum: SUPPORTED_LANGUAGES, + description: + 'The programming language or runtime to execute the code in.', + }, + code: { + type: 'string', + description: `The complete, self-contained code to execute, without any truncation or minimization. +- The environment is stateless; variables and imports don't persist between executions. +- Prior /mnt/data files are available and can be modified in place. +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- Input code **IS ALREADY** displayed to the user, so **DO NOT** repeat it in your response unless asked. +- Output code **IS NOT** displayed to the user, so **DO** write all desired output explicitly. +- IMPORTANT: You MUST explicitly print/output ALL results you want the user to see. +- py: This is not a Jupyter notebook environment. Use \`print()\` for all outputs. +- py: Matplotlib: Use \`plt.savefig()\` to save plots as files. +- js: use the \`console\` or \`process\` methods for all outputs. +- r: IMPORTANT: No X11 display available. ALL graphics MUST use Cairo library (library(Cairo)). +- Other languages: use appropriate output functions.`, + }, + args: { + type: 'array', + items: { type: 'string' }, + description: + 'Additional arguments to execute the code with. This should only be used if the input code requires additional arguments to run.', + }, + }, + required: ['lang', 'code'], +} as const; + +export const CodeExecutionToolDescription = ` +Runs code and returns stdout/stderr output from a stateless execution environment, similar to running scripts in a command-line interface. Each execution is isolated and independent. + +Usage: +- No network access available. +- Generated files are automatically delivered; **DO NOT** provide download links. +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- NEVER use this tool to execute malicious code. +`.trim(); + +/** + * Statefulness here is FILESYSTEM-tier, not runtime-tier. Executions in a + * session reuse one warm machine, so `/mnt/data` carries across calls — but + * every execution is a brand-new interpreter process in a fresh sandbox, so + * variables and imports never survive. The note must not imply otherwise: a + * model told its in-memory state persists writes `df = ...` in one call and + * `df.head()` in the next, then hits a NameError it was told to treat as rare. + */ +export const STATEFUL_ENV_NOTE = + 'Session state: executions in this conversation run on the same warm machine, so files persist between calls — but each execution is a NEW process. Variables, imports, and in-memory data NEVER carry over: every call must re-import and rebuild the state it needs. Only /mnt/data is durable (the machine itself may also be reset at any time), so write anything that must survive there and read it back next call.'; + +export const StatefulCodeExecutionToolDescription = ` +Runs code and returns stdout/stderr output. Executions in this conversation share one warm machine with a persistent /mnt/data, but each execution runs as a separate process (not a notebook-style kernel). + +${STATEFUL_ENV_NOTE} + +Usage: +- No network access available. +- Generated files are automatically delivered; **DO NOT** provide download links. +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- NEVER use this tool to execute malicious code. +`.trim(); + +export function buildCodeExecutionToolDescription(opts?: { + statefulSessions?: boolean; +}): string { + return opts?.statefulSessions === true + ? StatefulCodeExecutionToolDescription + : CodeExecutionToolDescription; +} + +const STATELESS_CODE_PARAM_NOTE = + "The environment is stateless; variables and imports don't persist between executions."; +const STATEFUL_CODE_PARAM_NOTE = + 'Executions in this conversation share one warm machine, so files written to /mnt/data persist between calls. Each execution is a new process: variables and imports do NOT carry over — re-import and reload from /mnt/data every call.'; + +export function buildCodeExecutionToolSchema(opts?: { + statefulSessions?: boolean; +}): typeof CodeExecutionToolSchema { + const note = + opts?.statefulSessions === true + ? STATEFUL_CODE_PARAM_NOTE + : STATELESS_CODE_PARAM_NOTE; + const codeDescription = + CodeExecutionToolSchema.properties.code.description.replace( + STATELESS_CODE_PARAM_NOTE, + note + ); + return { + ...CodeExecutionToolSchema, + properties: { + ...CodeExecutionToolSchema.properties, + code: { + ...CodeExecutionToolSchema.properties.code, + description: codeDescription, + }, + }, + } as typeof CodeExecutionToolSchema; +} + +export const CodeExecutionToolName = ToolNames.EXECUTE_CODE; + +export const CodeExecutionToolDefinition = { + name: CodeExecutionToolName, + description: CodeExecutionToolDescription, + schema: CodeExecutionToolSchema, +} as const; diff --git a/packages/dev-tools/src/guidance.ts b/packages/dev-tools/src/guidance.ts new file mode 100644 index 00000000..f9f8fc97 --- /dev/null +++ b/packages/dev-tools/src/guidance.ts @@ -0,0 +1,14 @@ +/** + * Reusable guidance embedded across the coding tool schemas and + * descriptions. Single source of truth: every description that needs to tell + * the model where durable files live says it with + * {@link CODE_ARTIFACT_PATH_GUIDANCE}, so the wording cannot drift per tool. + */ + +/** Where generated files must be written so later calls can read them back. */ +export const CODE_ARTIFACT_PATH_GUIDANCE = + 'Anything a later call needs (data, helper scripts/modules, partial results) MUST be written under `/mnt/data` in the same call that produces it; `/tmp` never survives the call. `/mnt/data` keeps files with recognized extensions, covering common source, text, data, document, image, and archive formats (.py/.sh/.sql/.md/.json/.csv/.parquet/.png/.pdf/.zip and similar); extensionless or unusual extensions are not kept. Failed executions register nothing; fix the error and rerun before relying on new files.'; + +/** How to produce multi-line files and Python one-liners from bash. */ +export const BASH_SHELL_GUIDANCE = + 'Bash: multi-line files use heredoc/printf; run Python via python3 -c/heredoc, not bare Python.'; diff --git a/packages/dev-tools/src/index.ts b/packages/dev-tools/src/index.ts new file mode 100644 index 00000000..ac651512 --- /dev/null +++ b/packages/dev-tools/src/index.ts @@ -0,0 +1,11 @@ +export * from './constants.js'; +export * from './types.js'; +export * from './intent.js'; +export * from './guidance.js'; +export * from './timeout.js'; +export * from './execute-code.js'; +export * from './bash-tool.js'; +export * from './read-file.js'; +export * from './run-tools-with-code.js'; +export * from './run-tools-with-bash.js'; +export * from './local/index.js'; diff --git a/packages/dev-tools/src/intent.test.ts b/packages/dev-tools/src/intent.test.ts new file mode 100644 index 00000000..53b58f28 --- /dev/null +++ b/packages/dev-tools/src/intent.test.ts @@ -0,0 +1,216 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import type { JsonSchemaType } from './types.js'; +import { + applyOutcome, + INTENT_ARG, + INTENT_DESCRIPTION, + INTENT_LABEL_MARKER, + INTENT_PROPERTY, + isIntentLabelProperty, + outcomeFieldsFromResult, + readIntent, + readOutcomeFields, + resolveToolOutcome, + stripIntent, + withIntent, + withoutIntent, +} from './intent.js'; + +const BASE_SCHEMA: JsonSchemaType = { + type: 'object', + properties: { + path: { type: 'string', description: 'File path.' }, + }, + required: ['path'], +}; + +test('the intent property is frozen and its description keeps the marker prefix', () => { + assert.equal(Object.isFrozen(INTENT_PROPERTY), true); + assert.equal(INTENT_DESCRIPTION.startsWith(INTENT_LABEL_MARKER), true); + assert.equal(isIntentLabelProperty(INTENT_PROPERTY), true); + assert.equal( + isIntentLabelProperty({ type: 'string', description: 'other' }), + false + ); + assert.equal(isIntentLabelProperty(null), false); +}); + +test('withIntent prepends the label property first and never requires it', () => { + const schema = withIntent(BASE_SCHEMA); + assert.deepEqual(Object.keys(schema.properties ?? {}), [ + INTENT_ARG, + 'path', + ]); + assert.equal(schema.properties?.intent?.description, INTENT_DESCRIPTION); + assert.deepEqual(schema.required, ['path']); + assert.equal( + BASE_SCHEMA.properties && INTENT_ARG in BASE_SCHEMA.properties, + false + ); +}); + +test('withIntent is a no-op when the schema already declares intent', () => { + const withOwn: JsonSchemaType = { + ...BASE_SCHEMA, + properties: { + intent: { + type: 'string', + description: 'Business intent, not the label.', + }, + path: { type: 'string', description: 'File path.' }, + }, + }; + assert.equal(withIntent(withOwn), withOwn); +}); + +test('withoutIntent removes the label property and its required entry', () => { + const withLabel = withIntent(BASE_SCHEMA); + const stripped = withoutIntent(withLabel); + assert.equal(INTENT_ARG in (stripped?.properties ?? {}), false); + assert.deepEqual(stripped?.required, ['path']); +}); + +test('withoutIntent prunes required when intent was its only member', () => { + const onlyIntent = withIntent({ type: 'object', properties: {} }); + const stripped = withoutIntent(onlyIntent); + assert.equal(stripped?.required, undefined); +}); + +test('withoutIntent never strips a business parameter named intent', () => { + const businessIntent: JsonSchemaType = { + type: 'object', + properties: { + intent: { + type: 'string', + description: 'Which of several goals to pursue', + }, + }, + required: ['intent'], + }; + assert.equal(withoutIntent(businessIntent), businessIntent); +}); + +test('readIntent reads object args and stringified JSON args', () => { + assert.equal( + readIntent({ intent: 'Renaming the callback router' }), + 'Renaming the callback router' + ); + assert.equal( + readIntent('{"intent":"Renaming the callback router"}'), + 'Renaming the callback router' + ); + assert.equal(readIntent({ intent: ' ' }), undefined); + assert.equal(readIntent({}), undefined); + assert.equal(readIntent('not json'), undefined); +}); + +test('stripIntent removes the label key and leaves everything else', () => { + assert.deepEqual(stripIntent({ intent: 'label', path: 'a.ts' }), { + path: 'a.ts', + }); + assert.deepEqual(stripIntent('{"intent":"label","path":"a.ts"}'), { + path: 'a.ts', + }); + const unchanged = { path: 'a.ts' }; + assert.equal(stripIntent(unchanged), unchanged); + assert.equal(stripIntent('plain string'), 'plain string'); +}); + +test('applyOutcome resolves in precedence order: outcome, patch, unchanged intent', () => { + assert.equal( + applyOutcome('Searching the router', { + outcome: 'Searched the router', + }), + 'Searched the router' + ); + assert.equal( + applyOutcome('Searching the router', { + outcome_patch: { from: 'Searching', to: 'Searched' }, + }), + 'Searched the router' + ); + assert.equal( + applyOutcome('Searching the router', {}), + 'Searching the router' + ); + assert.equal(applyOutcome(undefined, {}), undefined); + assert.equal(applyOutcome(undefined, { outcome: 'Done' }), 'Done'); + assert.equal( + applyOutcome('Searching the router', { + outcome_patch: { from: 'Editing', to: 'Edited' }, + }), + 'Searching the router' + ); +}); + +test('resolveToolOutcome emits only tool-authored labels for failed calls', () => { + const args = { intent: 'Searching the callback router' }; + assert.equal( + resolveToolOutcome( + args, + { outcome: 'Searched the callback router' }, + { isError: true } + ), + 'Searched the callback router' + ); + assert.equal( + resolveToolOutcome( + args, + { outcome_patch: { from: 'Searching', to: 'Searched' } }, + { isError: true } + ), + 'Searched the callback router' + ); + assert.equal( + resolveToolOutcome( + args, + { outcome_patch: { from: 'Editing', to: 'Edited' } }, + { isError: true } + ), + undefined + ); +}); + +test('resolveToolOutcome collapses and bounds the emitted label', () => { + assert.equal( + resolveToolOutcome( + { intent: 'Searching\n the router' }, + { outcome_patch: { from: 'Searching', to: 'Searched' } } + ), + 'Searched the router' + ); + const long = 'a'.repeat(400); + const bounded = resolveToolOutcome({ intent: 'x' }, { outcome: long }); + assert.equal(bounded?.length, 256); + assert.equal(bounded?.endsWith('…'), true); + assert.equal(resolveToolOutcome({}, { outcome: ' ' }), undefined); +}); + +test('resolveToolOutcome returns undefined without tool-authored fields', () => { + assert.equal(resolveToolOutcome({ intent: 'x' }, undefined), undefined); + assert.equal(resolveToolOutcome({ intent: 'x' }, {}), undefined); +}); + +test('outcome fields read through the artifact channel', () => { + const artifact = { outcome: 'Searched the router' }; + assert.deepEqual(readOutcomeFields(artifact), { + outcome: 'Searched the router', + outcome_patch: undefined, + }); + assert.deepEqual( + readOutcomeFields({ outcome_patch: { from: 'a', to: 'b', extra: 1 } }), + { outcome: undefined, outcome_patch: { from: 'a', to: 'b' } } + ); + assert.equal(readOutcomeFields('not an object'), undefined); + assert.equal(readOutcomeFields({ unrelated: true }), undefined); + assert.deepEqual(outcomeFieldsFromResult({ outcome: 'typed wins' }), { + outcome: 'typed wins', + }); + assert.deepEqual(outcomeFieldsFromResult({ artifact }), { + outcome: 'Searched the router', + outcome_patch: undefined, + }); + assert.equal(outcomeFieldsFromResult({}), undefined); +}); diff --git a/packages/dev-tools/src/intent.ts b/packages/dev-tools/src/intent.ts new file mode 100644 index 00000000..2dec4403 --- /dev/null +++ b/packages/dev-tools/src/intent.ts @@ -0,0 +1,374 @@ +/** + * @fileoverview Tool intent labels. + * + * Lets a tool declare, as the FIRST property of its input schema, an `intent` + * string: one model-authored sentence stating what that specific call is about + * to do ("Searching for OAuth handling in the callback router"). Because the + * property is first, it is the first key providers stream in the tool-call + * args, so a host UI can render it as the call's live status label before the + * rest of the args exist. When the call settles, {@link applyOutcome} edits + * the sentence in place into its outcome form — a tool-supplied replacement + * (`outcome`) or a tool-supplied span edit (`outcome_patch`). Absent either, + * the label is left exactly as the model wrote it: completion is a UI state + * (the shimmer stopping, the icon settling), not a tense change. + * + * The arg is always optional (never listed in `required`): the same schemas + * are callable from programmatic tool calling, where no UI renders a label + * and forcing generated code to fabricate one would be pure cost. Tool bodies + * must call {@link stripIntent} before using their args so no tool receives a + * parameter it did not declare. + */ + +import type { JsonSchemaType, OutcomePatch } from './types.js'; + +/** Argument carrying the model-authored label for a tool call. */ +export const INTENT_ARG = 'intent'; + +/** + * Opening words of {@link INTENT_DESCRIPTION}, and the discriminator that + * tells the injected LABEL apart from a tool's own business parameter that + * merely shares the name `intent`. + * + * Exported because host applications reimplement the same strip/sanitize + * passes and would otherwise duplicate this as a string literal: if the two + * copies drift, the host silently stops recognizing SDK-native labels and + * fails OPEN (labels stay in schemas, opt-outs stop working) with no error. + * Any edit to the description must preserve this prefix verbatim. + */ +export const INTENT_LABEL_MARKER = 'ALWAYS write this field FIRST'; + +/** + * Model-facing instruction for the injected `intent` property. + * + * Deliberately terse — it is repeated on every opted-in tool schema, on every + * request, so each sentence is paid for many times over. What remains is + * load-bearing: first-position placement (the entire streaming mechanism), + * the one-sentence present-progressive form, who reads it, and the sibling + * rule, without which models emit identical labels for parallel calls to one + * tool and defeat the feature's headline case. + */ +export const INTENT_DESCRIPTION = + `${INTENT_LABEL_MARKER}, before any other argument. One present-progressive ` + + 'sentence saying what THIS call is about to do: "Searching for OAuth handling ' + + 'in the callback router". Shown to the user as this call\'s live status. ' + + 'Never name the tool. Sibling calls to one tool must differ.'; + +/** + * Canonical (frozen) shape of the injected property. Always embed a COPY + * (`{ ...INTENT_PROPERTY }`): LangChain's JSON-schema validator stamps a + * `__absolute_uri__` marker onto every subschema it dereferences, which + * throws on a frozen object — and a single shared instance would be stamped + * with one schema's URI while embedded in many. + */ +export const INTENT_PROPERTY: JsonSchemaType = Object.freeze({ + type: 'string', + description: INTENT_DESCRIPTION, +}); + +/** + * Discriminates the intent LABEL property from a tool's own business + * parameter that merely shares the name: the label contract always opens + * with the same instruction. Removal/sanitize passes must never strip a + * parameter the tool actually needs. + */ +export function isIntentLabelProperty(property: unknown): boolean { + if (property == null || typeof property !== 'object') { + return false; + } + const record = property as { type?: unknown; description?: unknown }; + return ( + record.type === 'string' && + typeof record.description === 'string' && + record.description.startsWith(INTENT_LABEL_MARKER) + ); +} + +/** + * Schema shape accepted by {@link withoutIntent}. + * + * `required` is widened to `readonly string[]` because the SDK's own native + * schemas are declared `as const` — their `required` is a readonly tuple, and + * a mutable `string[]` parameter would reject the very schemas this helper + * exists for (TS2345), forcing embedders to cast to use the advertised API. + */ +export type IntentStrippableSchema = Omit & { + required?: readonly string[]; +}; + +/** + * Returns a copy of `parameters` without the injected intent LABEL — the + * opt-out for consumers that render no status label and should not pay for + * the property. + * + * The SDK's native schemas carry the label unconditionally, so without this + * an embedder has no lever at all: `withIntent` is applied at module scope. + * Marker-guarded, so a tool's own business parameter named `intent` is never + * removed. Returns the input unchanged when there is nothing to strip. + * + * `required` is pruned alongside the property: a schema that lists `intent` + * as required (strict-mode normalization does exactly that, since OpenAI + * strict function schemas require every property to appear in `required`) + * would otherwise be left naming a property it no longer declares, which is + * invalid JSON Schema and gets rejected by the provider instead of quietly + * opting out. + */ +export function withoutIntent( + parameters?: IntentStrippableSchema +): JsonSchemaType | undefined { + const props = parameters?.properties; + if ( + parameters == null || + props == null || + !isIntentLabelProperty(props[INTENT_ARG]) + ) { + return parameters as JsonSchemaType | undefined; + } + const { [INTENT_ARG]: _omit, ...rest } = props; + const next: JsonSchemaType = { + ...(parameters as JsonSchemaType), + properties: rest, + }; + if (parameters.required != null) { + const required = parameters.required.filter(key => key !== INTENT_ARG); + if (required.length > 0) { + next.required = required; + } else { + delete next.required; + } + } + return next; +} + +/** + * Returns a copy of the parameters schema with `intent` prepended as the + * FIRST property (object key order is insertion order and every provider + * serializer preserves it — first key in the schema means first key in the + * streamed input). Never mutates the input; no-op when the schema already + * declares `intent`. The property is not added to `required`. + */ +export function withIntent(parameters?: JsonSchemaType): JsonSchemaType { + const existingProps = parameters?.properties ?? {}; + if (INTENT_ARG in existingProps) { + return parameters as JsonSchemaType; + } + return { + ...parameters, + type: 'object', + properties: { [INTENT_ARG]: { ...INTENT_PROPERTY }, ...existingProps }, + }; +} + +/** + * Coerces tool-call args to an object, parsing a stringified JSON object + * (some providers deliver args as a string). Returns undefined otherwise. + */ +function coerceArgsObject(args: unknown): Record | undefined { + if (typeof args === 'object' && args !== null && !Array.isArray(args)) { + return args as Record; + } + if (typeof args === 'string' && args.trim().startsWith('{')) { + try { + const parsed = JSON.parse(args) as unknown; + if ( + parsed != null && + typeof parsed === 'object' && + !Array.isArray(parsed) + ) { + return parsed as Record; + } + } catch { + return undefined; + } + } + return undefined; +} + +/** + * Reads the model-authored intent from tool-call args (handles stringified + * args). Returns undefined when absent, empty, or not a string. + */ +export function readIntent(args: unknown): string | undefined { + const value = coerceArgsObject(args)?.[INTENT_ARG]; + if (typeof value !== 'string') { + return undefined; + } + const trimmed = value.trim(); + return trimmed === '' ? undefined : trimmed; +} + +/** + * Returns the args without the `intent` key so downstream consumers that did + * not declare it never receive it. Parses stringified JSON object args; + * returns the value unchanged when the key is absent. + */ +export function stripIntent(args: unknown): unknown { + const obj = coerceArgsObject(args); + if (!obj || !(INTENT_ARG in obj)) { + return args; + } + const { [INTENT_ARG]: _omit, ...rest } = obj; + return rest; +} + +/** + * Resolves the settled label for a call from its model-authored `intent` and + * the tool's result fields, in precedence order: + * + * 1. `outcome` — full replacement authored by the tool. + * 2. `outcome_patch` — first occurrence of `from` in the intent replaced + * with `to` (case-sensitive); no-op when `from` is absent or empty. + * 3. Otherwise the intent is returned UNCHANGED. + * + * There is deliberately no mechanical present-progressive→past-tense rewrite. + * Such a transform can only be a closed list of English verbs, which makes it + * wrong in three ways at once: it never fires for the non-English labels this + * feature expects (the model answers in the user's language), it fires for + * some sibling calls and not others inside one group — "Searched…" beside + * "Recording…" — and it quietly enumerates a vocabulary in a feature whose + * premise is that the sentence is free-form. Completion is conveyed by UI + * state (the shimmer stopping, the icon settling), which is language-neutral + * and always consistent; a tool that wants past tense says so explicitly via + * `outcome` or `outcome_patch`. + * + * Returns undefined when there is neither an intent nor an outcome, so + * callers fall back to their default label. Pure and dependency-free — host + * UIs needing identical logic can import or mirror it. + */ +export function applyOutcome( + intent: string | undefined, + result?: { outcome?: string; outcome_patch?: OutcomePatch } +): string | undefined { + const outcome = result?.outcome; + if (typeof outcome === 'string' && outcome.trim() !== '') { + return outcome; + } + if (intent == null || intent === '') { + return undefined; + } + const patch = result?.outcome_patch; + if (patch != null && patch.from !== '' && intent.includes(patch.from)) { + /** Replacement callback keeps `to` verbatim — a direct string second + * argument would interpret `$&`/`$'`-style tokens in tool-authored + * text (e.g. labels derived from shell syntax). */ + return intent.replace(patch.from, () => patch.to); + } + return intent; +} + +/** + * Hard cap on an emitted outcome label. The label is a single progress line + * in UI chrome; a tool that derives it from data (or a malformed patch) + * must not be able to inflate completion events or persisted parts. + */ +const MAX_OUTCOME_CHARS = 256; + +function boundOutcomeLabel(label: string | undefined): string | undefined { + if (label == null) { + return undefined; + } + const singleLine = label.replace(/\s+/g, ' ').trim(); + if (singleLine === '') { + return undefined; + } + if (singleLine.length <= MAX_OUTCOME_CHARS) { + return singleLine; + } + return `${singleLine.slice(0, MAX_OUTCOME_CHARS - 1)}…`; +} + +/** + * Resolves the settled label to emit on a completion event: only when the + * tool actually authored `outcome`/`outcome_patch` fields. Returns undefined + * otherwise, so the wire never carries a label the host already has — a bare + * intent needs no settled form, because it is displayed unchanged and the UI + * conveys completion through its own state. Hosts must NOT rewrite it (see + * {@link applyOutcome} for why a tense transform is deliberately absent). The + * result is collapsed to a bounded single line before emission. + * + * For failed calls (`isError`), only tool-AUTHORED text may label the call: + * an explicit `outcome`, or a patch whose `from` actually matches the intent. + * An unmatched patch resolves to undefined rather than silently reusing the + * in-flight intent, so a failure is never labelled as though it succeeded. + */ +export function resolveToolOutcome( + args: unknown, + fields?: { outcome?: string; outcome_patch?: OutcomePatch } | null, + options?: { isError?: boolean } +): string | undefined { + if ( + fields == null || + (fields.outcome == null && fields.outcome_patch == null) + ) { + return undefined; + } + if (options?.isError !== true) { + return boundOutcomeLabel(applyOutcome(readIntent(args), fields)); + } + const outcome = fields.outcome; + if (typeof outcome === 'string' && outcome.trim() !== '') { + return boundOutcomeLabel(outcome); + } + const intent = readIntent(args); + const patch = fields.outcome_patch; + if ( + intent != null && + patch != null && + patch.from !== '' && + intent.includes(patch.from) + ) { + return boundOutcomeLabel(intent.replace(patch.from, () => patch.to)); + } + return undefined; +} + +/** + * Reads the outcome fields off a tool-execution result: the typed + * `outcome`/`outcome_patch` fields when present, else the artifact channel + * (see {@link readOutcomeFields}) — so a `content_and_artifact` tool authors + * its label the same way on the direct and event-driven paths. + */ +export function outcomeFieldsFromResult(result: { + outcome?: string; + outcome_patch?: OutcomePatch; + artifact?: unknown; +}): { outcome?: string; outcome_patch?: OutcomePatch } | undefined { + if (result.outcome != null || result.outcome_patch != null) { + return result; + } + return readOutcomeFields(result.artifact); +} + +/** + * Extracts validated `outcome`/`outcome_patch` fields from an arbitrary + * value — the artifact channel through which an in-process + * `content_and_artifact` tool authors its settled label. Returns undefined + * when neither field is usable. + */ +export function readOutcomeFields( + source: unknown +): { outcome?: string; outcome_patch?: OutcomePatch } | undefined { + if (source == null || typeof source !== 'object' || Array.isArray(source)) { + return undefined; + } + const record = source as Record; + const outcome = + typeof record.outcome === 'string' && record.outcome.trim() !== '' + ? record.outcome + : undefined; + let outcome_patch: OutcomePatch | undefined; + const rawPatch = record.outcome_patch; + if ( + rawPatch != null && + typeof rawPatch === 'object' && + !Array.isArray(rawPatch) + ) { + const patch = rawPatch as Record; + if (typeof patch.from === 'string' && typeof patch.to === 'string') { + outcome_patch = { from: patch.from, to: patch.to }; + } + } + if (outcome == null && outcome_patch == null) { + return undefined; + } + return { outcome, outcome_patch }; +} diff --git a/packages/dev-tools/src/local-definitions.test.ts b/packages/dev-tools/src/local-definitions.test.ts new file mode 100644 index 00000000..3cd21901 --- /dev/null +++ b/packages/dev-tools/src/local-definitions.test.ts @@ -0,0 +1,118 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { + CompileCheckToolDefinition, + CompileCheckToolDescription, + CompileCheckToolSchema, +} from './local/compile-check.js'; +import { + LocalCodeExecutionToolDescription, + LocalBashExecutionToolDescription, +} from './local/execution-tools.js'; +import { + LocalEditFileToolDefinition, + LocalEditFileToolDescription, + LocalEditFileToolSchema, + LocalGlobSearchToolDefinition, + LocalGlobSearchToolSchema, + LocalGrepSearchToolDefinition, + LocalGrepSearchToolSchema, + LocalListDirectoryToolDefinition, + LocalListDirectoryToolSchema, + LocalReadFileToolDefinition, + LocalReadFileToolDescription, + LocalReadFileToolSchema, + LocalWriteFileToolDefinition, + LocalWriteFileToolDescription, + LocalWriteFileToolSchema, +} from './local/file-tools.js'; +import { ToolNames } from './constants.js'; +import { INTENT_ARG } from './intent.js'; + +const intentIsFirst = (schema: { properties?: Record }) => + Object.keys(schema.properties ?? {})[0] === INTENT_ARG; + +const assertRegistryShape = ( + definition: { + name: string; + allowed_callers?: string[]; + responseFormat?: string; + toolType?: string; + }, + name: string +) => { + assert.equal(definition.name, name); + assert.deepEqual(definition.allowed_callers, ['direct', 'code_execution']); + assert.equal(definition.responseFormat, 'content_and_artifact'); + assert.equal(definition.toolType, 'builtin'); +}; + +test('local read_file schema and definition', () => { + assert.deepEqual(LocalReadFileToolSchema.required, ['path']); + assert.equal(intentIsFirst(LocalReadFileToolSchema), true); + assertRegistryShape(LocalReadFileToolDefinition, ToolNames.READ_FILE); + assert.equal( + LocalReadFileToolDefinition.parameters, + LocalReadFileToolSchema + ); + assert.match(LocalReadFileToolDescription, /line numbers/); +}); + +test('local write_file schema and definition', () => { + assert.deepEqual(LocalWriteFileToolSchema.required, ['path', 'content']); + assert.equal(intentIsFirst(LocalWriteFileToolSchema), true); + assertRegistryShape(LocalWriteFileToolDefinition, ToolNames.WRITE_FILE); + assert.match(LocalWriteFileToolDescription, /unified diff/); +}); + +test('local edit_file schema supports single and batched edits', () => { + assert.deepEqual(LocalEditFileToolSchema.required, ['path']); + assert.equal(intentIsFirst(LocalEditFileToolSchema), true); + assert.equal( + LocalEditFileToolSchema.properties?.edits?.items?.required?.includes( + 'old_text' + ), + true + ); + assertRegistryShape(LocalEditFileToolDefinition, ToolNames.EDIT_FILE); + assert.match(LocalEditFileToolDescription, /whitespace-normalized/); +}); + +test('local grep and glob search schemas and definitions', () => { + assert.deepEqual(LocalGrepSearchToolSchema.required, ['pattern']); + assert.deepEqual(LocalGlobSearchToolSchema.required, ['pattern']); + assert.equal(intentIsFirst(LocalGrepSearchToolSchema), true); + assert.equal(intentIsFirst(LocalGlobSearchToolSchema), true); + assertRegistryShape(LocalGrepSearchToolDefinition, ToolNames.GREP_SEARCH); + assertRegistryShape(LocalGlobSearchToolDefinition, ToolNames.GLOB_SEARCH); +}); + +test('local list_directory schema and definition', () => { + assert.equal(LocalListDirectoryToolSchema.required, undefined); + assert.equal(intentIsFirst(LocalListDirectoryToolSchema), true); + assertRegistryShape( + LocalListDirectoryToolDefinition, + ToolNames.LIST_DIRECTORY + ); +}); + +test('compile_check schema is optional-args only', () => { + assert.equal(CompileCheckToolSchema.required, undefined); + assert.equal(intentIsFirst(CompileCheckToolSchema), true); + assertRegistryShape(CompileCheckToolDefinition, ToolNames.COMPILE_CHECK); + assert.match(CompileCheckToolDescription, /tsconfig\.json/); +}); + +test('local execution descriptions describe the local machine semantics', () => { + assert.match(LocalCodeExecutionToolDescription, /on the local machine/); + assert.match(LocalBashExecutionToolDescription, /on the local machine/); + assert.match( + LocalCodeExecutionToolDescription, + /local execution mode is enabled/ + ); + assert.match( + LocalBashExecutionToolDescription, + /Prefer project-native commands/ + ); +}); diff --git a/packages/dev-tools/src/local-programmatic-calling.test.ts b/packages/dev-tools/src/local-programmatic-calling.test.ts new file mode 100644 index 00000000..8caff496 --- /dev/null +++ b/packages/dev-tools/src/local-programmatic-calling.test.ts @@ -0,0 +1,96 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { + createLocalBashProgrammaticToolCallingSchema, + createLocalProgrammaticToolCallingSchema, + LocalBashProgrammaticToolCallingDefinition, + LocalBashProgrammaticToolCallingDescription, + LocalProgrammaticToolCallingDefinition, + LocalProgrammaticToolCallingDescription, +} from './local/programmatic-tool-calling.js'; +import { INTENT_ARG } from './intent.js'; +import { ProgrammaticToolCallingDescription } from './run-tools-with-code.js'; +import { BashProgrammaticToolCallingDescription } from './run-tools-with-bash.js'; + +test('local run_tools_with_code schema defaults to bash with a 60s timeout', () => { + const schema = createLocalProgrammaticToolCallingSchema(); + assert.equal(Object.keys(schema.properties)[0], INTENT_ARG); + assert.deepEqual(schema.required, ['code']); + assert.deepEqual(schema.properties.lang.enum, [ + 'py', + 'python', + 'bash', + 'sh', + ]); + assert.equal(schema.properties.lang.default, 'bash'); + assert.match(schema.properties.lang.description, /Defaults to bash/); + assert.equal(schema.properties.timeout.default, 60_000); + assert.equal(schema.properties.timeout.minimum, 1_000); + assert.equal(schema.properties.timeout.maximum, 300_000); + assert.match(schema.properties.timeout.description, /Default: 60 seconds/); +}); + +test('local run_tools_with_code schema honours a configured timeout', () => { + const schema = createLocalProgrammaticToolCallingSchema({ + timeoutMs: 90_000, + }); + assert.equal(schema.properties.timeout.default, 90_000); + assert.match(schema.properties.timeout.description, /Default: 90 seconds/); +}); + +test('local run_tools_with_bash schema swaps the timeout and keeps the code param', () => { + const schema = createLocalBashProgrammaticToolCallingSchema({ + timeoutMs: 120_000, + }); + assert.equal('lang' in schema.properties, false); + assert.equal(schema.properties.timeout.default, 120_000); + assert.deepEqual(schema.required, ['code']); +}); + +test('local PTC descriptions extend the remote descriptions with the local suffix', () => { + assert.ok( + LocalProgrammaticToolCallingDescription.startsWith( + ProgrammaticToolCallingDescription + ) + ); + assert.match( + LocalProgrammaticToolCallingDescription, + /in-process localhost bridge/ + ); + assert.ok( + LocalBashProgrammaticToolCallingDescription.startsWith( + BashProgrammaticToolCallingDescription + ) + ); + assert.match( + LocalBashProgrammaticToolCallingDescription, + /this bash orchestration code/ + ); +}); + +test('default local PTC definitions bind names, descriptions, and default schemas', () => { + assert.equal( + LocalProgrammaticToolCallingDefinition.name, + 'run_tools_with_code' + ); + assert.equal( + LocalProgrammaticToolCallingDefinition.description, + LocalProgrammaticToolCallingDescription + ); + assert.deepEqual(LocalProgrammaticToolCallingDefinition.schema.required, [ + 'code', + ]); + assert.equal( + LocalBashProgrammaticToolCallingDefinition.name, + 'run_tools_with_bash' + ); + assert.equal( + LocalBashProgrammaticToolCallingDefinition.description, + LocalBashProgrammaticToolCallingDescription + ); + assert.deepEqual( + LocalBashProgrammaticToolCallingDefinition.schema.required, + ['code'] + ); +}); diff --git a/packages/dev-tools/src/local/compile-check.ts b/packages/dev-tools/src/local/compile-check.ts new file mode 100644 index 00000000..c2a0cb14 --- /dev/null +++ b/packages/dev-tools/src/local/compile-check.ts @@ -0,0 +1,43 @@ +/** + * `compile_check` — the LLM-visible surface of the local-engine typecheck + * tool: canonical name, schema, description, and registry definition. + * Ported from `@librechat/agents` (`src/tools/local/CompileCheckTool.ts`), + * where the schema was module-private and only the definition factory was + * exported; the toolchain auto-detection and command execution stay + * harness-native in the agents SDK. + */ + +import { ToolNames, CONTENT_AND_ARTIFACT } from '../constants.js'; +import { withIntent } from '../intent.js'; +import type { JsonSchemaType, ToolDefinition } from '../types.js'; + +/** Back-compat alias; canonical name lives on `ToolNames.COMPILE_CHECK`. */ +export const CompileCheckToolName = ToolNames.COMPILE_CHECK; + +export const CompileCheckToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + command: { + type: 'string', + description: + 'Optional explicit command to run instead of the auto-detected one. Runs verbatim from the local engine cwd; honours the standard sandbox/AST gate.', + }, + timeout_ms: { + type: 'integer', + description: + 'Optional timeout in milliseconds. Defaults to 120000 (2 min).', + }, + }, +}); + +export const CompileCheckToolDescription = + "Run the project's standard typecheck or lint pass and return its output. Auto-detects from project markers (tsconfig.json/package.json -> tsc; Cargo.toml -> cargo check; go.mod -> go vet; pyproject.toml -> mypy or py_compile). Pass `command` to override."; + +export const CompileCheckToolDefinition: ToolDefinition = { + name: CompileCheckToolName, + description: CompileCheckToolDescription, + parameters: CompileCheckToolSchema, + allowed_callers: ['direct', 'code_execution'], + responseFormat: CONTENT_AND_ARTIFACT, + toolType: 'builtin', +}; diff --git a/packages/dev-tools/src/local/execution-tools.ts b/packages/dev-tools/src/local/execution-tools.ts new file mode 100644 index 00000000..73d904c4 --- /dev/null +++ b/packages/dev-tools/src/local/execution-tools.ts @@ -0,0 +1,31 @@ +/** + * Local-engine variants of the code and bash execution tool descriptions. + * The local engine reuses the remote schemas (`../execute-code.js`, + * `../bash-tool.js`) and swaps only the descriptions, because the schema + * shape is identical while the semantics (local machine, configured + * working directory, host filesystem) differ. Ported from + * `@librechat/agents` (`src/tools/local/LocalExecutionTools.ts`); the + * execution engine stays harness-native in the agents SDK. + */ + +export const LocalCodeExecutionToolDescription = ` +Runs code on the local machine in the configured working directory. Unlike the remote Code API sandbox, this tool can see the local repository, installed runtimes, environment variables, and filesystem available to the host process. + +Usage: +- The remote sandbox API remains the default; this description applies only when local execution mode is enabled. +- Local commands can use the Anthropic sandbox runtime when local.sandbox.enabled=true and @anthropic-ai/sandbox-runtime is installed. +- Commands execute in the local working directory and may modify local files. +- Input code is already displayed to the user, so do not repeat it unless asked. +- Output is not displayed unless you print it explicitly. +`.trim(); + +export const LocalBashExecutionToolDescription = ` +Runs bash commands on the local machine in the configured working directory. Unlike the remote Code API sandbox, this tool can see the local repository, installed tools, environment variables, and filesystem available to the host process. + +Usage: +- The remote sandbox API remains the default; this description applies only when local execution mode is enabled. +- Local commands can use the Anthropic sandbox runtime when local.sandbox.enabled=true and @anthropic-ai/sandbox-runtime is installed. +- Commands execute in the local working directory and may modify local files. +- Output is not displayed unless you print it explicitly. +- Prefer project-native commands and inspect files before changing them. +`.trim(); diff --git a/packages/dev-tools/src/local/file-tools.ts b/packages/dev-tools/src/local/file-tools.ts new file mode 100644 index 00000000..84b49585 --- /dev/null +++ b/packages/dev-tools/src/local/file-tools.ts @@ -0,0 +1,234 @@ +/** + * `read_file`, `write_file`, `edit_file`, `grep_search`, `glob_search`, + * `list_directory` — the LLM-visible surface of the local-engine file/edit/ + * search tools: canonical names (with the harness back-compat aliases), + * schemas, descriptions, and registry definitions. Ported from + * `@librechat/agents` (`src/tools/local/LocalCodingTools.ts`), where the + * descriptions previously lived inline in the tool factories; execution + * (the workspace filesystem engine, edit strategies, syntax checks) stays + * harness-native in the agents SDK. + * + * The remote engine's parallel `read_file` (skill and code-execution output + * oriented) lives in `../read-file.js`; both share the canonical name so + * consumer UIs render them with the same icon. + */ + +import { ToolNames, CONTENT_AND_ARTIFACT } from '../constants.js'; +import { withIntent } from '../intent.js'; +import type { JsonSchemaType, ToolDefinition } from '../types.js'; + +/** + * Tool name aliases retained for back-compat with consumers that imported + * the per-file `Local*ToolName` constants. The canonical names live on + * `Constants.*` (see `src/common/enum.ts`); these aliases just point at + * them so a typo upstream gets caught at the type level. + */ +export const LocalWriteFileToolName = ToolNames.WRITE_FILE; +export const LocalEditFileToolName = ToolNames.EDIT_FILE; +export const LocalGrepSearchToolName = ToolNames.GREP_SEARCH; +export const LocalGlobSearchToolName = ToolNames.GLOB_SEARCH; +export const LocalListDirectoryToolName = ToolNames.LIST_DIRECTORY; + +export const LocalReadFileToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + path: { + type: 'string', + description: + 'Path to a local file, relative to the configured cwd unless absolute paths are allowed.', + }, + offset: { + type: 'integer', + description: 'Optional 1-indexed line offset for large files.', + }, + limit: { + type: 'integer', + description: 'Optional maximum number of lines to return.', + }, + }, + required: ['path'], +}); + +export const LocalWriteFileToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + path: { + type: 'string', + description: + 'Path to write, relative to the configured cwd unless absolute paths are allowed.', + }, + content: { + type: 'string', + description: 'Complete file contents to write.', + }, + }, + required: ['path', 'content'], +}); + +export const LocalEditFileToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + path: { + type: 'string', + description: + 'Path to edit, relative to the configured cwd unless absolute paths are allowed.', + }, + old_text: { + type: 'string', + description: 'Exact text to replace. Must appear exactly once.', + }, + new_text: { + type: 'string', + description: 'Replacement text.', + }, + edits: { + type: 'array', + description: + 'Optional batch of exact replacements. Each old_text must appear exactly once in the original file.', + items: { + type: 'object', + properties: { + old_text: { type: 'string' }, + new_text: { type: 'string' }, + }, + required: ['old_text', 'new_text'], + }, + }, + }, + required: ['path'], +}); + +export const LocalGrepSearchToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + pattern: { + type: 'string', + description: 'Regex pattern to search for.', + }, + path: { + type: 'string', + description: 'Directory or file to search. Defaults to cwd.', + }, + glob: { + type: 'string', + description: 'Optional file glob passed to rg -g.', + }, + max_results: { + type: 'integer', + description: 'Maximum matching lines to return.', + }, + }, + required: ['pattern'], +}); + +export const LocalGlobSearchToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + pattern: { + type: 'string', + description: 'File glob pattern, for example "src/**/*.ts".', + }, + path: { + type: 'string', + description: 'Directory to search. Defaults to cwd.', + }, + max_results: { + type: 'integer', + description: 'Maximum file paths to return.', + }, + }, + required: ['pattern'], +}); + +export const LocalListDirectoryToolSchema: JsonSchemaType = withIntent({ + type: 'object', + properties: { + path: { + type: 'string', + description: 'Directory to list. Defaults to cwd.', + }, + }, +}); + +/** + * Full tool descriptions as the LLM sees them on the bound tools. Ported + * from the inline factory descriptions in `LocalCodingTools.ts`; previously + * these strings were not exported and could not be shared with hosts. + */ +export const LocalReadFileToolDescription = + 'Read a local text file from the configured working directory with line numbers. When `attachReadAttachments` is enabled (e.g. images-only), reading an image returns an `image_url` content block so vision-capable models can see the file directly.'; + +export const LocalWriteFileToolDescription = + 'Create or overwrite a local text file in the configured working directory. Preserves the existing BOM and line endings when overwriting; defaults to LF without BOM for new files. Returns a unified diff of the changes when overwriting.'; + +export const LocalEditFileToolDescription = + 'Apply exact text replacements to a local file. The matcher tries exact, line-trimmed, whitespace-normalized, and indentation-flexible strategies in order so common LLM whitespace mistakes are recoverable. Each old_text must still match exactly one location. Returns a unified diff of the changes.'; + +export const LocalGrepSearchToolDescription = + 'Search local files for a regex pattern (ripgrep when available, Node fallback otherwise).'; + +export const LocalGlobSearchToolDescription = + 'Find local files matching a glob pattern (ripgrep when available, Node fallback otherwise).'; + +export const LocalListDirectoryToolDescription = + 'List files and directories in a local directory.'; + +/** + * Registry definitions for the local file/edit/search tools, shaped like the + * harness `toolDefinition()` helper output: `parameters` (not the LangChain + * `schema` option), callable both directly and from programmatic code + * execution, content-and-artifact response, builtin tool type. + */ +function localToolDefinition( + name: string, + description: string, + parameters: JsonSchemaType +): ToolDefinition { + return { + name, + description, + parameters, + allowed_callers: ['direct', 'code_execution'], + responseFormat: CONTENT_AND_ARTIFACT, + toolType: 'builtin', + }; +} + +export const LocalReadFileToolDefinition: ToolDefinition = localToolDefinition( + ToolNames.READ_FILE, + LocalReadFileToolDescription, + LocalReadFileToolSchema +); + +export const LocalWriteFileToolDefinition: ToolDefinition = localToolDefinition( + LocalWriteFileToolName, + LocalWriteFileToolDescription, + LocalWriteFileToolSchema +); + +export const LocalEditFileToolDefinition: ToolDefinition = localToolDefinition( + LocalEditFileToolName, + LocalEditFileToolDescription, + LocalEditFileToolSchema +); + +export const LocalGrepSearchToolDefinition: ToolDefinition = + localToolDefinition( + LocalGrepSearchToolName, + LocalGrepSearchToolDescription, + LocalGrepSearchToolSchema + ); + +export const LocalGlobSearchToolDefinition: ToolDefinition = + localToolDefinition( + LocalGlobSearchToolName, + LocalGlobSearchToolDescription, + LocalGlobSearchToolSchema + ); + +export const LocalListDirectoryToolDefinition: ToolDefinition = + localToolDefinition( + LocalListDirectoryToolName, + LocalListDirectoryToolDescription, + LocalListDirectoryToolSchema + ); diff --git a/packages/dev-tools/src/local/index.ts b/packages/dev-tools/src/local/index.ts new file mode 100644 index 00000000..de18c6fb --- /dev/null +++ b/packages/dev-tools/src/local/index.ts @@ -0,0 +1,4 @@ +export * from './compile-check.js'; +export * from './execution-tools.js'; +export * from './file-tools.js'; +export * from './programmatic-tool-calling.js'; diff --git a/packages/dev-tools/src/local/programmatic-tool-calling.ts b/packages/dev-tools/src/local/programmatic-tool-calling.ts new file mode 100644 index 00000000..ebbeb8f6 --- /dev/null +++ b/packages/dev-tools/src/local/programmatic-tool-calling.ts @@ -0,0 +1,142 @@ +/** + * Local-engine variants of the programmatic tool calling schemas and + * descriptions: the local `run_tools_with_code` schema adds a `lang` runtime + * selector (bash by default) and a local timeout, and the local + * `run_tools_with_bash` schema swaps the timeout; the descriptions append the + * local-engine suffix. Ported from `@librechat/agents` + * (`src/tools/local/LocalProgrammaticToolCalling.ts`); the in-process + * localhost tool bridge and execution stay harness-native in the agents SDK. + */ + +import { ToolNames, CONTENT_AND_ARTIFACT } from '../constants.js'; +import type { ToolDefinition } from '../types.js'; +import { + ProgrammaticToolCallingDescription, + ProgrammaticToolCallingName, + ProgrammaticToolCallingSchema, +} from '../run-tools-with-code.js'; +import { + BashProgrammaticToolCallingDescription, + BashProgrammaticToolCallingSchema, +} from '../run-tools-with-bash.js'; + +/** + * Local engine configuration fields these schema builders read. The harness + * `LocalExecutionConfig` is a superset; only `timeoutMs` shapes the schema. + */ +export type LocalProgrammaticToolCallingConfig = { + timeoutMs?: number; +}; + +const DEFAULT_TIMEOUT = 60000; +const LOCAL_MIN_TIMEOUT = 1000; +const LOCAL_MAX_TIMEOUT = 300000; + +type LocalTimeoutSchema = { + type: 'integer'; + minimum: number; + maximum: number; + default: number; + description: string; +}; + +type LocalProgrammaticToolCallingJsonSchema = { + type: 'object'; + properties: typeof ProgrammaticToolCallingSchema.properties & { + timeout: LocalTimeoutSchema; + lang: { + type: 'string'; + enum: readonly ['py', 'python', 'bash', 'sh']; + default: 'bash'; + description: string; + }; + }; + required: readonly ['code']; +}; + +type LocalBashProgrammaticToolCallingJsonSchema = { + type: 'object'; + properties: typeof BashProgrammaticToolCallingSchema.properties & { + timeout: LocalTimeoutSchema; + }; + required: readonly ['code']; +}; + +function normalizeLocalTimeout(timeoutMs: number | undefined): number { + if (timeoutMs == null || !Number.isFinite(timeoutMs)) { + return DEFAULT_TIMEOUT; + } + + return Math.max(LOCAL_MIN_TIMEOUT, Math.floor(timeoutMs)); +} + +function formatLocalTimeout(timeoutMs: number): string { + return timeoutMs % 1000 === 0 + ? `${timeoutMs / 1000} seconds` + : `${timeoutMs} milliseconds`; +} + +function createLocalTimeoutSchema(timeoutMs?: number): LocalTimeoutSchema { + const defaultTimeout = normalizeLocalTimeout(timeoutMs); + const maxTimeout = Math.max(LOCAL_MAX_TIMEOUT, defaultTimeout); + const formattedDefault = formatLocalTimeout(defaultTimeout); + const formattedMax = formatLocalTimeout(maxTimeout); + + return { + type: 'integer', + minimum: LOCAL_MIN_TIMEOUT, + maximum: maxTimeout, + default: defaultTimeout, + description: + 'Maximum local execution time in milliseconds. ' + + `Default: ${formattedDefault}. Max: ${formattedMax}.`, + }; +} + +export function createLocalProgrammaticToolCallingSchema( + localConfig: LocalProgrammaticToolCallingConfig = {} +): LocalProgrammaticToolCallingJsonSchema { + return { + ...ProgrammaticToolCallingSchema, + properties: { + ...ProgrammaticToolCallingSchema.properties, + timeout: createLocalTimeoutSchema(localConfig.timeoutMs), + lang: { + type: 'string', + enum: ['py', 'python', 'bash', 'sh'], + default: 'bash', + description: + 'Local engine runtime for orchestration code. Defaults to bash; use py/python for Python orchestration.', + }, + }, + } as const; +} + +export function createLocalBashProgrammaticToolCallingSchema( + localConfig: LocalProgrammaticToolCallingConfig = {} +): LocalBashProgrammaticToolCallingJsonSchema { + return { + ...BashProgrammaticToolCallingSchema, + properties: { + ...BashProgrammaticToolCallingSchema.properties, + timeout: createLocalTimeoutSchema(localConfig.timeoutMs), + }, + } as const; +} + +export const LocalProgrammaticToolCallingDescription = `${ProgrammaticToolCallingDescription}\n\nLocal engine: runs bash by default, or Python when \`lang\` is \`py\` or \`python\`, on the host machine and calls tools through an in-process localhost bridge.`; + +export const LocalBashProgrammaticToolCallingDescription = `${BashProgrammaticToolCallingDescription}\n\nLocal engine: runs this bash orchestration code on the host machine and calls tools through an in-process localhost bridge.`; + +/** Default local-engine definitions: default timeout, bash runtime for the unified tool. */ +export const LocalProgrammaticToolCallingDefinition = { + name: ProgrammaticToolCallingName, + description: LocalProgrammaticToolCallingDescription, + schema: createLocalProgrammaticToolCallingSchema(), +} as const; + +export const LocalBashProgrammaticToolCallingDefinition = { + name: ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING, + description: LocalBashProgrammaticToolCallingDescription, + schema: createLocalBashProgrammaticToolCallingSchema(), +} as const; diff --git a/packages/dev-tools/src/read-file.ts b/packages/dev-tools/src/read-file.ts new file mode 100644 index 00000000..f8f87712 --- /dev/null +++ b/packages/dev-tools/src/read-file.ts @@ -0,0 +1,47 @@ +/** + * `read_file` — the LLM-visible surface of the remote file reading tool for + * invoked skill files and code-execution output. Ported from + * `@librechat/agents` (`src/tools/ReadFile.ts`). The local engine's parallel + * `read_file` (same canonical name) lives in `./local/file-tools.js`. + */ + +import { ToolNames } from './constants.js'; +import { INTENT_PROPERTY } from './intent.js'; + +export const ReadFileToolName = ToolNames.READ_FILE; + +export const ReadFileToolDescription = `Read the contents of a file. Returns text content with line numbers for easy reference. + +For skill files, use the path format: {skillName}/{filePath} (e.g. "pdf-analyzer/src/utils.py", "code-review/SKILL.md"). + +BEHAVIOR: +- Text files: returned with numbered lines. +- Images (png, jpeg, gif, webp): returned as visual content the model can see. +- PDFs: returned as document content. +- Other binary files: metadata returned with a note to use bash for processing. +- Large files (>256KB text, >10MB binary): metadata only. +- SKILL.md: returns the skill's instructions directly. + +CONSTRAINTS: +- Only files from invoked skills or code execution output are accessible. +- Do not guess file paths. Use paths from the skill documentation or tool output.`; + +export const ReadFileToolSchema = { + type: 'object', + properties: { + intent: { ...INTENT_PROPERTY }, + path: { + type: 'string', + description: + 'Path to the file. For skill files: "{skillName}/{path}" (e.g. "pdf-analyzer/src/utils.py"). For code execution output: the path as returned by the execution tool.', + }, + }, + required: ['path'], +} as const; + +export const ReadFileToolDefinition = { + name: ReadFileToolName, + description: ReadFileToolDescription, + parameters: ReadFileToolSchema, + responseFormat: 'content_and_artifact' as const, +} as const; diff --git a/packages/dev-tools/src/remote-definitions.test.ts b/packages/dev-tools/src/remote-definitions.test.ts new file mode 100644 index 00000000..c833d8c3 --- /dev/null +++ b/packages/dev-tools/src/remote-definitions.test.ts @@ -0,0 +1,206 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { + AttachedWorkspaceBashExecutionToolDescription, + BashExecutionToolDefinition, + BashExecutionToolDescription, + BashExecutionToolSchema, + buildBashExecutionToolDescription, + buildBashExecutionToolSchema, + BashToolOutputReferencesGuide, + StatefulBashExecutionToolDescription, +} from './bash-tool.js'; +import { ToolNames } from './constants.js'; +import { INTENT_ARG } from './intent.js'; +import { + CodeExecutionToolDefinition, + CodeExecutionToolDescription, + CodeExecutionToolSchema, + buildCodeExecutionToolDescription, + buildCodeExecutionToolSchema, + StatefulCodeExecutionToolDescription, + SUPPORTED_LANGUAGES, +} from './execute-code.js'; +import { ReadFileToolDefinition, ReadFileToolSchema } from './read-file.js'; +import { + buildBashProgrammaticToolCallingDescription, + buildBashProgrammaticToolCallingSchema, + BashProgrammaticToolCallingDefinition, + BashProgrammaticToolCallingSchema, +} from './run-tools-with-bash.js'; +import { + ProgrammaticToolCallingDefinition, + ProgrammaticToolCallingSchema, +} from './run-tools-with-code.js'; + +const intentIsFirst = (schema: { properties?: Record }) => + Object.keys(schema.properties ?? {})[0] === INTENT_ARG; + +test('execute_code definition matches its schema and description', () => { + assert.equal(CodeExecutionToolDefinition.name, 'execute_code'); + assert.equal( + CodeExecutionToolDefinition.description, + CodeExecutionToolDescription + ); + assert.equal(CodeExecutionToolDefinition.schema, CodeExecutionToolSchema); + assert.deepEqual(CodeExecutionToolSchema.required, ['lang', 'code']); + assert.equal(intentIsFirst(CodeExecutionToolSchema), true); + assert.deepEqual(SUPPORTED_LANGUAGES, [ + 'py', + 'js', + 'ts', + 'c', + 'cpp', + 'java', + 'php', + 'rs', + 'go', + 'd', + 'f90', + 'r', + 'bash', + ]); + assert.match( + CodeExecutionToolSchema.properties.code.description, + /\/mnt\/data/ + ); +}); + +test('execute_code description and schema builders swap stateful wording', () => { + assert.equal( + buildCodeExecutionToolDescription(), + CodeExecutionToolDescription + ); + assert.equal( + buildCodeExecutionToolDescription({ statefulSessions: true }), + StatefulCodeExecutionToolDescription + ); + const statefulSchema = buildCodeExecutionToolSchema({ + statefulSessions: true, + }); + assert.match( + statefulSchema.properties.code.description, + /files written to \/mnt\/data persist between calls/ + ); + assert.match( + CodeExecutionToolSchema.properties.code.description, + /stateless/ + ); +}); + +test('bash_tool definition matches its schema and description', () => { + assert.equal(BashExecutionToolDefinition.name, ToolNames.BASH_TOOL); + assert.equal( + BashExecutionToolDefinition.description, + BashExecutionToolDescription + ); + assert.equal(BashExecutionToolDefinition.schema, BashExecutionToolSchema); + assert.deepEqual(BashExecutionToolSchema.required, ['command']); + assert.equal(intentIsFirst(BashExecutionToolSchema), true); + assert.match( + BashExecutionToolSchema.properties.command.description, + /heredoc\/printf/ + ); +}); + +test('bash_tool description builder composes stateful, attached, and reference variants', () => { + assert.equal( + buildBashExecutionToolDescription(), + BashExecutionToolDescription + ); + assert.equal( + buildBashExecutionToolDescription({ statefulSessions: true }), + StatefulBashExecutionToolDescription + ); + assert.equal( + buildBashExecutionToolDescription({ attachedWorkspace: true }), + AttachedWorkspaceBashExecutionToolDescription + ); + const withReferences = buildBashExecutionToolDescription({ + enableToolOutputReferences: true, + }); + assert.equal( + withReferences, + `${BashExecutionToolDescription}\n\n${BashToolOutputReferencesGuide}` + ); + const attachedSchema = buildBashExecutionToolSchema({ + attachedWorkspace: true, + }); + assert.match( + attachedSchema.properties.command.description, + /selected persistent project/ + ); + assert.doesNotMatch( + attachedSchema.properties.command.description, + /stateless; variables and state/ + ); +}); + +test('read_file definition carries the content_and_artifact response format', () => { + assert.equal(ReadFileToolDefinition.name, ToolNames.READ_FILE); + assert.equal(ReadFileToolDefinition.responseFormat, 'content_and_artifact'); + assert.equal(ReadFileToolDefinition.parameters, ReadFileToolSchema); + assert.deepEqual(ReadFileToolSchema.required, ['path']); + assert.equal(intentIsFirst(ReadFileToolSchema), true); + assert.match(ReadFileToolDefinition.description, /skill files/i); +}); + +test('run_tools_with_code schema carries code, manifest, and timeout', () => { + assert.equal( + ProgrammaticToolCallingDefinition.name, + ToolNames.PROGRAMMATIC_TOOL_CALLING + ); + assert.deepEqual(ProgrammaticToolCallingSchema.required, ['code']); + assert.equal(intentIsFirst(ProgrammaticToolCallingSchema), true); + assert.equal(ProgrammaticToolCallingSchema.properties.code.minLength, 1); + assert.equal( + ProgrammaticToolCallingSchema.properties.tool_manifest.uniqueItems, + true + ); + assert.equal( + ProgrammaticToolCallingSchema.properties.timeout.minimum, + 1_000 + ); + assert.match( + ProgrammaticToolCallingDefinition.description, + /STATELESS EXECUTION/ + ); +}); + +test('run_tools_with_bash schema and its attached-workspace builders', () => { + assert.equal( + BashProgrammaticToolCallingDefinition.name, + ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING + ); + assert.deepEqual(BashProgrammaticToolCallingSchema.required, ['code']); + assert.equal(intentIsFirst(BashProgrammaticToolCallingSchema), true); + assert.equal( + BashProgrammaticToolCallingSchema.properties.tool_manifest.uniqueItems, + true + ); + assert.match( + BashProgrammaticToolCallingDefinition.description, + /bash functions/ + ); + + const attachedDescription = buildBashProgrammaticToolCallingDescription({ + attachedWorkspace: true, + }); + assert.match(attachedDescription, /ATTACHED WORKSPACE EXECUTION:/); + assert.doesNotMatch(attachedDescription, /CRITICAL - STATELESS EXECUTION:/); + + const attachedSchema = buildBashProgrammaticToolCallingSchema({ + attachedWorkspace: true, + maxRunTimeoutMs: 45_000, + }); + assert.match( + attachedSchema.properties.code.description, + /selected persistent workspace/ + ); + assert.equal(attachedSchema.properties.timeout.default, 45_000); + assert.notEqual( + attachedSchema.properties.code.description, + BashProgrammaticToolCallingSchema.properties.code.description + ); +}); diff --git a/packages/dev-tools/src/run-tools-with-bash.ts b/packages/dev-tools/src/run-tools-with-bash.ts new file mode 100644 index 00000000..27c03667 --- /dev/null +++ b/packages/dev-tools/src/run-tools-with-bash.ts @@ -0,0 +1,205 @@ +/** + * `run_tools_with_bash` — the LLM-visible surface of the remote (Code API) + * bash programmatic tool calling tool: canonical name, schema (with the + * runtime-configurable timeout property), description, and the + * attached-workspace description/schema builders the harness factory applies + * when the tool runs against a selected persistent workspace. Ported + * verbatim from `@librechat/agents` + * (`src/tools/BashProgrammaticToolCalling.ts`); execution stays + * harness-native in the agents SDK. + */ + +import { ToolNames } from './constants.js'; +import { + BASH_SHELL_GUIDANCE, + CODE_ARTIFACT_PATH_GUIDANCE, +} from './guidance.js'; +import { INTENT_PROPERTY } from './intent.js'; +import { + createCodeApiRunTimeoutSchema, + resolveCodeApiRunTimeoutMs, +} from './timeout.js'; +import type { ProgrammaticToolCallingJsonSchema } from './timeout.js'; + +/** Default programmatic run timeout, resolved from the environment at load. */ +const DEFAULT_RUN_TIMEOUT_MS = resolveCodeApiRunTimeoutMs(); + +const ATTACHED_BASH_DATA_DIRECTORY = '"${LIBRECHAT_CODE_DATA_DIR:-/mnt/data}"'; +const ATTACHED_BASH_ARTIFACT_PATH_GUIDANCE = + `Use ${ATTACHED_BASH_DATA_DIRECTORY} for injected files and generated artifacts. ` + + 'The directory is execution-scoped; the selected workspace is the persistent project root.'; + +// Description Components +// ============================================================================ + +const STATELESS_WARNING = `CRITICAL - STATELESS EXECUTION: +Each call is a fresh bash shell. Variables and state do NOT persist between calls. +You MUST complete your entire workflow in ONE code block. +DO NOT split work across multiple calls expecting to reuse variables.`; + +const ATTACHED_WORKSPACE_WARNING = `ATTACHED WORKSPACE EXECUTION: +- Commands start in the selected persistent workspace; project file changes persist between calls. +- Each sandbox run is a fresh process, so shell variables, background processes, and temporary execution data do not persist. +- Injected files and generated artifacts use \${LIBRECHAT_CODE_DATA_DIR:-/mnt/data}; do not copy them into the project unless the task requires it.`; + +const CORE_RULES = `Rules: +- One call: state does not persist +- Tools are pre-defined as bash functions—DO NOT redefine them +- Each tool function accepts a JSON string argument +- Save tool output with raw=$(tool '{}'); printf '%s\n' "$raw" > /mnt/data/file.json; direct tool > file may be empty +- Tool stdout is normalized to one compact JSON value when possible; parse saved stdout once, then use fromjson? // . only for JSON-string fields +- Only echo/printf output returns to the model +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- ${BASH_SHELL_GUIDANCE} +- timeout caps one sandbox run/replay iteration, not the total multi-round-trip workflow`; + +const ADDITIONAL_RULES = + '- Tool names normalized: hyphens→underscores, reserved words get `_tool` suffix'; + +const EXAMPLES = `Example (Complete workflow in one call): + # Query data and process + data=$(query_database '{"sql": "SELECT * FROM users"}') + echo "$data" | jq '.[] | .name' + +Example (Parallel calls): + { sf=$(web_search '{"query": "SF weather"}'); printf '%s\n' "$sf" > /mnt/data/sf.json; } & + { ny=$(web_search '{"query": "NY weather"}'); printf '%s\n' "$ny" > /mnt/data/ny.json; } & + wait + echo "SF: $(jq -r . /mnt/data/sf.json)" + echo "NY: $(jq -r . /mnt/data/ny.json)"`; + +const ATTACHED_CORE_RULES = `Rules: +- One call: process state does not persist; project files do +- Tools are pre-defined as bash functions—DO NOT redefine them +- Each tool function accepts a JSON string argument +- Resolve tool calls into variables before changing project files; do not redirect a tool call directly into the project +- Set data_dir=${ATTACHED_BASH_DATA_DIRECTORY}; save generated artifacts there, and write durable project files relative to the working directory +- Tool stdout is normalized to one compact JSON value when possible; parse saved stdout once, then use fromjson? // . only for JSON-string fields +- Only echo/printf output returns to the model +- ${ATTACHED_BASH_ARTIFACT_PATH_GUIDANCE} +- ${BASH_SHELL_GUIDANCE} +- timeout caps one sandbox run/replay iteration, not the total multi-round-trip workflow`; + +const ATTACHED_EXAMPLES = `Example (Complete workflow in one call): + data=$(query_database '{"sql": "SELECT * FROM users"}') + echo "$data" | jq '.[] | .name' + +Example (Parallel calls): + data_dir=${ATTACHED_BASH_DATA_DIRECTORY} + { sf=$(web_search '{"query": "SF weather"}'); printf '%s\n' "$sf" > "$data_dir/sf.json"; } & + { ny=$(web_search '{"query": "NY weather"}'); printf '%s\n' "$ny" > "$data_dir/ny.json"; } & + wait + echo "SF: $(jq -r . "$data_dir/sf.json")" + echo "NY: $(jq -r . "$data_dir/ny.json")"`; + +const CODE_PARAM_DESCRIPTION = `Bash code that calls tools programmatically. Tools are available as bash functions. + +${STATELESS_WARNING} + +Each tool function accepts a JSON string as its argument. +Example: tool_name '{"key": "value"}' + +${EXAMPLES} + +${CORE_RULES}`; + +const TOOL_MANIFEST_DESCRIPTION = + 'Exact registered tool names used by the code. Required when direct-only tools are configured; ' + + 'validated before execution starts. Pass [] when the code calls no tools at all.'; + +// ============================================================================ +// Schema +// ============================================================================ + +export function createBashProgrammaticToolCallingSchema( + maxRunTimeoutMs = DEFAULT_RUN_TIMEOUT_MS +): ProgrammaticToolCallingJsonSchema { + return { + type: 'object', + properties: { + intent: { ...INTENT_PROPERTY }, + code: { + type: 'string', + minLength: 1, + description: CODE_PARAM_DESCRIPTION, + }, + tool_manifest: { + type: 'array', + items: { type: 'string' }, + uniqueItems: true, + description: TOOL_MANIFEST_DESCRIPTION, + }, + timeout: createCodeApiRunTimeoutSchema(maxRunTimeoutMs), + }, + required: ['code'], + } as const; +} + +export const BashProgrammaticToolCallingSchema = + createBashProgrammaticToolCallingSchema(); + +export const BashProgrammaticToolCallingName = + ToolNames.BASH_PROGRAMMATIC_TOOL_CALLING; + +export const BashProgrammaticToolCallingDescription = ` +Run tools via bash code. Tools are available as bash functions that accept JSON string arguments. + +${STATELESS_WARNING} + +${CORE_RULES} +${ADDITIONAL_RULES} + +When to use: shell pipelines, parallel execution (& and wait), file processing, text manipulation. + +${EXAMPLES} +`.trim(); + +export const BashProgrammaticToolCallingDefinition = { + name: BashProgrammaticToolCallingName, + description: BashProgrammaticToolCallingDescription, + schema: BashProgrammaticToolCallingSchema, +} as const; + +/** + * Composes the bash programmatic tool calling description for the selected + * execution mode: the attached-workspace variant swaps the stateless + * warning, core rules, and examples for the persistent-project wording the + * harness factory applies when the tool runs with a selected workspace. + */ +export function buildBashProgrammaticToolCallingDescription(options?: { + attachedWorkspace?: boolean; +}): string { + if (options?.attachedWorkspace !== true) { + return BashProgrammaticToolCallingDescription; + } + return BashProgrammaticToolCallingDescription.replace( + STATELESS_WARNING, + ATTACHED_WORKSPACE_WARNING + ) + .replace(CORE_RULES, ATTACHED_CORE_RULES) + .replace(EXAMPLES, ATTACHED_EXAMPLES); +} + +/** + * Builds the bash programmatic tool calling schema, applying the + * attached-workspace code-param description when requested. Mirrors the + * replacement chain the harness factory applies to the freshly created + * schema before binding the tool. + */ +export function buildBashProgrammaticToolCallingSchema(options?: { + attachedWorkspace?: boolean; + maxRunTimeoutMs?: number; +}): ProgrammaticToolCallingJsonSchema { + const schema = createBashProgrammaticToolCallingSchema( + options?.maxRunTimeoutMs + ); + if (options?.attachedWorkspace === true) { + schema.properties.code.description = CODE_PARAM_DESCRIPTION.replace( + STATELESS_WARNING, + ATTACHED_WORKSPACE_WARNING + ) + .replace(CORE_RULES, ATTACHED_CORE_RULES) + .replace(EXAMPLES, ATTACHED_EXAMPLES); + } + return schema; +} diff --git a/packages/dev-tools/src/run-tools-with-code.ts b/packages/dev-tools/src/run-tools-with-code.ts new file mode 100644 index 00000000..3e0f5319 --- /dev/null +++ b/packages/dev-tools/src/run-tools-with-code.ts @@ -0,0 +1,121 @@ +/** + * `run_tools_with_code` — the LLM-visible surface of the remote (Code API) + * Python programmatic tool calling tool: canonical name, schema (with the + * runtime-configurable timeout property), and description. Ported verbatim + * from `@librechat/agents` (`src/tools/ProgrammaticToolCalling.ts`); the + * execution client and tool-result replay stay harness-native in the agents + * SDK. + */ + +import { ToolNames } from './constants.js'; +import { CODE_ARTIFACT_PATH_GUIDANCE } from './guidance.js'; +import { INTENT_PROPERTY } from './intent.js'; +import { + createCodeApiRunTimeoutSchema, + resolveCodeApiRunTimeoutMs, +} from './timeout.js'; +import type { ProgrammaticToolCallingJsonSchema } from './timeout.js'; + +/** Default programmatic run timeout, resolved from the environment at load. */ +const DEFAULT_RUN_TIMEOUT_MS = resolveCodeApiRunTimeoutMs(); + +// Description Components (Single Source of Truth) +// ============================================================================ + +const STATELESS_WARNING = `CRITICAL - STATELESS EXECUTION: +Each call is a fresh Python interpreter. Variables, imports, and data do NOT persist between calls. +You MUST complete your entire workflow in ONE code block: query → process → output. +DO NOT split work across multiple calls expecting to reuse variables.`; + +const CORE_RULES = `Rules: +- One call: state does not persist +- Auto-wrapped async; use await, no main()/asyncio.run() +- Tools are pre-defined—DO NOT write function definitions +- Call tools with keyword args only (await tool(arg=value), never pass a dict) +- Tool results are decoded Python values (dict/list/str) +- Only print() output returns to the model +- ${CODE_ARTIFACT_PATH_GUIDANCE} +- timeout caps one sandbox run/replay iteration, not the total multi-round-trip workflow`; + +const ADDITIONAL_RULES = + '- Tool names normalized: hyphens→underscores, keywords get `_tool` suffix'; + +const EXAMPLES = `Example (Complete workflow in one call): + # Query data + data = await query_database(sql="SELECT * FROM users") + # Process it + df = pd.DataFrame(data) + summary = df.groupby('region').sum() + # Output results + await write_to_sheet(spreadsheet_id=sid, data=summary.to_dict()) + print(f"Wrote {len(summary)} rows") + +Example (Parallel calls): + sf, ny = await asyncio.gather(get_weather(city="SF"), get_weather(city="NY")) + print(f"SF: {sf}, NY: {ny}")`; + +// ============================================================================ +// Schema +// ============================================================================ + +const CODE_PARAM_DESCRIPTION = `Python code that calls tools programmatically. Tools are available as async functions. + +${STATELESS_WARNING} + +Your code is auto-wrapped in async context. Just write logic with await—no boilerplate needed. + +${EXAMPLES} + +${CORE_RULES}`; + +const TOOL_MANIFEST_DESCRIPTION = + 'Exact registered tool names used by the code. Required when direct-only tools are configured; ' + + 'validated before execution starts. Pass [] when the code calls no tools at all.'; + +export function createProgrammaticToolCallingSchema( + maxRunTimeoutMs = DEFAULT_RUN_TIMEOUT_MS +): ProgrammaticToolCallingJsonSchema { + return { + type: 'object', + properties: { + intent: { ...INTENT_PROPERTY }, + code: { + type: 'string', + minLength: 1, + description: CODE_PARAM_DESCRIPTION, + }, + tool_manifest: { + type: 'array', + items: { type: 'string' }, + uniqueItems: true, + description: TOOL_MANIFEST_DESCRIPTION, + }, + timeout: createCodeApiRunTimeoutSchema(maxRunTimeoutMs), + }, + required: ['code'], + } as const; +} + +export const ProgrammaticToolCallingSchema = + createProgrammaticToolCallingSchema(); + +export const ProgrammaticToolCallingName = ToolNames.PROGRAMMATIC_TOOL_CALLING; + +export const ProgrammaticToolCallingDescription = ` +Run tools via Python code. Auto-wrapped in async context—just use \`await\` directly. + +${STATELESS_WARNING} + +${CORE_RULES} +${ADDITIONAL_RULES} + +When to use: loops, conditionals, parallel (\`asyncio.gather\`), multi-step pipelines. + +${EXAMPLES} +`.trim(); + +export const ProgrammaticToolCallingDefinition = { + name: ProgrammaticToolCallingName, + description: ProgrammaticToolCallingDescription, + schema: ProgrammaticToolCallingSchema, +} as const; diff --git a/packages/dev-tools/src/timeout.test.ts b/packages/dev-tools/src/timeout.test.ts new file mode 100644 index 00000000..c2f475ef --- /dev/null +++ b/packages/dev-tools/src/timeout.test.ts @@ -0,0 +1,76 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { + clampCodeApiRunTimeoutMs, + createCodeApiRunTimeoutSchema, + DEFAULT_CODE_API_RUN_TIMEOUT_MS, + MAX_CODE_API_RUN_TIMEOUT_SCHEMA_MS, + MIN_CODE_API_RUN_TIMEOUT_MS, + resolveCodeApiRunTimeoutMs, +} from './timeout.js'; + +const ENV_VAR = 'CODE_API_RUN_TIMEOUT_MS'; + +test('default run timeout is 15 seconds', () => { + assert.equal(DEFAULT_CODE_API_RUN_TIMEOUT_MS, 15_000); + assert.equal(MIN_CODE_API_RUN_TIMEOUT_MS, 1_000); + assert.equal(MAX_CODE_API_RUN_TIMEOUT_SCHEMA_MS, 300_000); +}); + +test('resolveCodeApiRunTimeoutMs reads the environment override', () => { + const previous = process.env[ENV_VAR]; + try { + delete process.env[ENV_VAR]; + assert.equal( + resolveCodeApiRunTimeoutMs(), + DEFAULT_CODE_API_RUN_TIMEOUT_MS + ); + process.env[ENV_VAR] = '20000'; + assert.equal(resolveCodeApiRunTimeoutMs(), 20_000); + process.env[ENV_VAR] = 'not-a-number'; + assert.equal( + resolveCodeApiRunTimeoutMs(), + DEFAULT_CODE_API_RUN_TIMEOUT_MS + ); + process.env[ENV_VAR] = '10'; + assert.equal(resolveCodeApiRunTimeoutMs(), MIN_CODE_API_RUN_TIMEOUT_MS); + } finally { + if (previous === undefined) { + delete process.env[ENV_VAR]; + } else { + process.env[ENV_VAR] = previous; + } + } +}); + +test('run timeout schema carries bounds and a human-readable default', () => { + const schema = createCodeApiRunTimeoutSchema(60_000); + assert.equal(schema.type, 'integer'); + assert.equal(schema.minimum, MIN_CODE_API_RUN_TIMEOUT_MS); + assert.equal(schema.default, 60_000); + assert.equal(schema.maximum, MAX_CODE_API_RUN_TIMEOUT_SCHEMA_MS); + assert.match(schema.description, /Default: 60 seconds/); + assert.match(schema.description, /Schema max: 300 seconds/); +}); + +test('run timeout schema maximum grows past the cap only when configured above it', () => { + const schema = createCodeApiRunTimeoutSchema(400_000); + assert.equal(schema.default, 400_000); + assert.equal(schema.maximum, 400_000); + assert.match(schema.description, /Default: 400 seconds/); +}); + +test('clampCodeApiRunTimeoutMs caps requested timeouts at the configured max', () => { + assert.equal(clampCodeApiRunTimeoutMs(999_999, 30_000), 30_000); + assert.equal(clampCodeApiRunTimeoutMs(5_000, 30_000), 5_000); + assert.equal(clampCodeApiRunTimeoutMs(undefined, 30_000), 30_000); + assert.equal( + clampCodeApiRunTimeoutMs(500, 30_000), + MIN_CODE_API_RUN_TIMEOUT_MS + ); + assert.equal( + clampCodeApiRunTimeoutMs(undefined, 0), + MIN_CODE_API_RUN_TIMEOUT_MS + ); +}); diff --git a/packages/dev-tools/src/timeout.ts b/packages/dev-tools/src/timeout.ts new file mode 100644 index 00000000..f6ceb8c3 --- /dev/null +++ b/packages/dev-tools/src/timeout.ts @@ -0,0 +1,108 @@ +import type { JsonSchemaType } from './types.js'; +/** Environment variable overriding the default programmatic run timeout. */ +const CODE_API_RUN_TIMEOUT_MS_ENV_VAR = 'CODE_API_RUN_TIMEOUT_MS'; + +export const DEFAULT_CODE_API_RUN_TIMEOUT_MS = 15_000; +export const MIN_CODE_API_RUN_TIMEOUT_MS = 1_000; +export const MAX_CODE_API_RUN_TIMEOUT_SCHEMA_MS = 300_000; + +type TimeoutSchema = { + type: 'integer'; + minimum: number; + maximum: number; + default: number; + description: string; +}; + +export type ProgrammaticToolCallingJsonSchema = { + type: 'object'; + properties: { + intent: JsonSchemaType; + code: { + type: 'string'; + minLength: number; + description: string; + }; + tool_manifest: { + type: 'array'; + items: { type: 'string' }; + uniqueItems: true; + description: string; + }; + timeout: TimeoutSchema; + }; + required: readonly ['code']; +}; + +function normalizeTimeoutMs(value: number | undefined): number | undefined { + if (value == null || !Number.isFinite(value)) { + return undefined; + } + + return Math.max(MIN_CODE_API_RUN_TIMEOUT_MS, Math.floor(value)); +} + +function parseTimeoutMs(value: string | undefined): number | undefined { + if (value == null || value.trim() === '') { + return undefined; + } + + return normalizeTimeoutMs(Number(value)); +} + +function formatTimeout(timeoutMs: number): string { + return timeoutMs % 1000 === 0 + ? `${timeoutMs / 1000} seconds` + : `${timeoutMs} milliseconds`; +} + +export function resolveCodeApiRunTimeoutMs(override?: number): number { + return ( + normalizeTimeoutMs(override) ?? + parseTimeoutMs(process.env[CODE_API_RUN_TIMEOUT_MS_ENV_VAR]) ?? + DEFAULT_CODE_API_RUN_TIMEOUT_MS + ); +} + +export function clampCodeApiRunTimeoutMs( + timeoutMs: number | undefined, + maxRunTimeoutMs = resolveCodeApiRunTimeoutMs() +): number { + const normalizedMaxRunTimeoutMs = + normalizeTimeoutMs(maxRunTimeoutMs) ?? DEFAULT_CODE_API_RUN_TIMEOUT_MS; + const normalizedTimeoutMs = normalizeTimeoutMs(timeoutMs); + + if (normalizedTimeoutMs == null) { + return normalizedMaxRunTimeoutMs; + } + + return Math.min(normalizedTimeoutMs, normalizedMaxRunTimeoutMs); +} + +export function createCodeApiRunTimeoutSchema( + maxRunTimeoutMs = resolveCodeApiRunTimeoutMs() +): TimeoutSchema { + const normalizedMaxRunTimeoutMs = + normalizeTimeoutMs(maxRunTimeoutMs) ?? DEFAULT_CODE_API_RUN_TIMEOUT_MS; + const normalizedSchemaMaxRunTimeoutMs = Math.max( + normalizedMaxRunTimeoutMs, + MAX_CODE_API_RUN_TIMEOUT_SCHEMA_MS + ); + const formattedTimeout = formatTimeout(normalizedMaxRunTimeoutMs); + const formattedSchemaMaxTimeout = formatTimeout( + normalizedSchemaMaxRunTimeoutMs + ); + + return { + type: 'integer', + minimum: MIN_CODE_API_RUN_TIMEOUT_MS, + maximum: normalizedSchemaMaxRunTimeoutMs, + default: normalizedMaxRunTimeoutMs, + description: + 'Maximum wall-clock time in milliseconds for one sandbox run or replay iteration. ' + + 'This is not the total multi-round-trip task budget. ' + + `Default: ${formattedTimeout}. ` + + 'Accepted values above the configured cap are clamped before execution. ' + + `Schema max: ${formattedSchemaMaxTimeout}. Configured cap: ${formattedTimeout}.`, + }; +} diff --git a/packages/dev-tools/src/types.ts b/packages/dev-tools/src/types.ts new file mode 100644 index 00000000..7fbff0b5 --- /dev/null +++ b/packages/dev-tools/src/types.ts @@ -0,0 +1,65 @@ +/** + * Minimal, dependency-free types for the coding tool surface. + * + * `ToolDefinition` mirrors `LCTool` from `@librechat/agents` so definitions + * exported here register with the harness unmodified. `JsonSchemaType` is + * that SDK's schema type widened with the JSON-Schema keywords these tool + * schemas actually use (`minLength`, `minimum`, `maximum`, `default`, + * `uniqueItems`) so every schema in this package type-checks as written. + */ + +/** JSON-Schema fragment used for tool parameters and their properties. */ +export type JsonSchemaType = { + type: + | 'string' + | 'number' + | 'integer' + | 'float' + | 'boolean' + | 'array' + | 'object'; + enum?: string[]; + items?: JsonSchemaType; + properties?: Record; + required?: string[]; + description?: string; + additionalProperties?: boolean | JsonSchemaType; + minLength?: number; + minimum?: number; + maximum?: number; + default?: number | string; + uniqueItems?: boolean; +}; + +/** Specifies which contexts can invoke a tool (inspired by Anthropic's allowed_callers). */ +export type AllowedCaller = 'direct' | 'code_execution'; + +/** Response format for tool output. */ +export type ToolResponseFormat = 'content' | 'content_and_artifact'; + +/** Tool definition as registered with the harness. Mirrors `LCTool`. */ +export type ToolDefinition = { + name: string; + description?: string; + parameters?: JsonSchemaType; + /** When true, tool is not loaded into context initially (for tool search) */ + defer_loading?: boolean; + /** + * Which contexts can invoke this tool. + * Default: ['direct'] (only callable directly by LLM) + */ + allowed_callers?: AllowedCaller[]; + responseFormat?: ToolResponseFormat; + /** Server name for MCP tools */ + serverName?: string; + toolType?: 'builtin' | 'mcp' | 'action'; +}; + +/** + * In-place edit of a call's model-authored `intent` label: the first + * occurrence of `from` in the intent is replaced with `to` (case-sensitive). + */ +export type OutcomePatch = { + from: string; + to: string; +}; diff --git a/packages/dev-tools/tsconfig.json b/packages/dev-tools/tsconfig.json new file mode 100644 index 00000000..67493b60 --- /dev/null +++ b/packages/dev-tools/tsconfig.json @@ -0,0 +1,15 @@ +{ + "compilerOptions": { + "target": "ES2022", + "module": "NodeNext", + "moduleResolution": "NodeNext", + "strict": true, + "declaration": true, + "sourceMap": true, + "outDir": "dist", + "rootDir": "src", + "skipLibCheck": true, + "types": ["node"] + }, + "include": ["src/**/*.ts"] +}