diff --git a/librechat.example.yaml b/librechat.example.yaml index 280886cc3a0..96ed50f65f8 100644 --- a/librechat.example.yaml +++ b/librechat.example.yaml @@ -698,6 +698,13 @@ endpoints: # maxRecursionLimit: 100 # # (optional) Abort a run once a single streamed tool call's arguments exceed this many bytes. # # Guards against runaway malformed tool-call generation. Defaults to 65536 (64 KiB); 0 disables. + # # Host-side skill/sandbox edits, not attached-worker edits. Over-budget calls write nothing. + # hostFileEdits: + # maxEdits: 100 + # maxWorkBytes: 67108864 # cumulative matching/reconstruction bytes + # maxOccurrences: 100000 # cumulative occurrences across every edit/pass + # timeoutMs: 2000 # terminate an overlong worker job + # maxConcurrent: 2 # per API process; no waiting queue # maxToolCallArgBytes: 65536 # # (optional) Abort a run once a single model generation emits more than this many stream events. # # Defense in depth against looping provider streams. Disabled by default. diff --git a/packages/api/src/agents/edits.spec.ts b/packages/api/src/agents/edits.spec.ts index d57ab540aed..faa92c6b98e 100644 --- a/packages/api/src/agents/edits.spec.ts +++ b/packages/api/src/agents/edits.spec.ts @@ -50,3 +50,9 @@ describe('normalizeEditArgs', () => { expect(normalizeEditArgs({ edits: normalized })).toEqual(normalized); }); }); + +it('rejects oversized batches even when entries are valid or JSON-stringified', () => { + const edits = Array.from({ length: 101 }, () => ({ old_text: 'a', new_text: 'b' })); + expect(normalizeEditArgs({ edits })).toContain('limited to 100'); + expect(normalizeEditArgs({ edits: JSON.stringify(edits) })).toContain('limited to 100'); +}); diff --git a/packages/api/src/agents/edits.ts b/packages/api/src/agents/edits.ts index 23e32a060b2..f2dfbc4e93f 100644 --- a/packages/api/src/agents/edits.ts +++ b/packages/api/src/agents/edits.ts @@ -1,3 +1,5 @@ +import { HOST_FILE_EDIT_HARD_MAX_COUNT } from 'librechat-data-provider'; + export type TextEdit = { old_text: string; new_text: string; @@ -51,6 +53,9 @@ export function normalizeEditArgs(args: { if (!Array.isArray(coercedEdits) || coercedEdits.length === 0) { return 'Provide a non-empty edits array when edits is supplied.'; } + if (coercedEdits.length > HOST_FILE_EDIT_HARD_MAX_COUNT) { + return `File edits are limited to ${HOST_FILE_EDIT_HARD_MAX_COUNT} replacements per call.`; + } const edits: TextEdit[] = []; for (const rawEdit of coercedEdits) { const edit = coerceJsonValue(rawEdit); diff --git a/packages/api/src/agents/files/__fixtures__/edit-worker.cjs b/packages/api/src/agents/files/__fixtures__/edit-worker.cjs new file mode 100644 index 00000000000..d9d554b068e --- /dev/null +++ b/packages/api/src/agents/files/__fixtures__/edit-worker.cjs @@ -0,0 +1,7 @@ +const { parentPort } = require('node:worker_threads'); + +parentPort.on('message', ({ content }) => { + if (content === 'wait') return; + if (content === 'crash') throw new Error('PRIVATE-WORKER-DIAGNOSTIC'); + parentPort.postMessage({ ok: true, result: { content, strategies: [] } }); +}); diff --git a/packages/api/src/agents/files/edit-worker.ts b/packages/api/src/agents/files/edit-worker.ts new file mode 100644 index 00000000000..885a5d27d79 --- /dev/null +++ b/packages/api/src/agents/files/edit-worker.ts @@ -0,0 +1,28 @@ +import { parentPort } from 'node:worker_threads'; +import type { HostEditResult, HostEditWorkLimits } from './matching'; +import type { TextEdit } from '../edits'; +import { applyTextEdits, HostEditError } from './matching'; + +export interface HostEditJob { + content: string; + edits: TextEdit[]; + limits: HostEditWorkLimits; +} + +export type HostEditReply = { ok: true; result: HostEditResult } | { ok: false; message: string }; + +parentPort?.on('message', (job: HostEditJob) => { + let reply: HostEditReply; + try { + reply = { ok: true, result: applyTextEdits(job.content, job.edits, job.limits) }; + } catch (error) { + reply = { + ok: false, + message: + error instanceof HostEditError + ? error.message + : 'File edit processing failed. Nothing was written.', + }; + } + parentPort?.postMessage(reply); +}); diff --git a/packages/api/src/agents/files/matching.spec.ts b/packages/api/src/agents/files/matching.spec.ts new file mode 100644 index 00000000000..04e6523e9b6 --- /dev/null +++ b/packages/api/src/agents/files/matching.spec.ts @@ -0,0 +1,60 @@ +import type { HostEditWorkLimits } from './matching'; +import { applyTextEdits } from './matching'; + +const limits: HostEditWorkLimits = { + maxEdits: 100, + maxWorkBytes: 64 * 1024 * 1024, + maxOccurrences: 100000, + maxOutputBytes: 10 * 1024 * 1024, +}; + +describe('cumulative host edit accounting', () => { + it('charges input, matching, and reconstruction once for a full-context exact edit', () => { + const edits = [{ old_text: 'ax', new_text: 'bx' }]; + expect(applyTextEdits('ax', edits, { ...limits, maxWorkBytes: 12 })).toEqual({ + content: 'bx', + strategies: ['exact'], + }); + expect(() => applyTextEdits('ax', edits, { ...limits, maxWorkBytes: 11 })).toThrow( + 'budget exceeded', + ); + }); + + it('charges contraction as well as growth across a batch', () => { + const edits = [ + { old_text: 'a', new_text: 'b', replace_all: true }, + { old_text: 'b', new_text: 'a', replace_all: true }, + { old_text: 'a', new_text: '', replace_all: true }, + ]; + expect(applyTextEdits('a', edits, { ...limits, maxWorkBytes: 16 })).toEqual({ + content: '', + strategies: ['exact', 'exact', 'exact'], + }); + expect(() => applyTextEdits('a', edits, { ...limits, maxWorkBytes: 15 })).toThrow( + 'budget exceeded', + ); + }); + + it('remeasures UTF-8 sizes after each replacement and rejects oversized intermediates', () => { + const edits = [ + { old_text: 'a', new_text: 'é', replace_all: true }, + { old_text: 'é', new_text: '😀', replace_all: true }, + { old_text: '😀', new_text: '', replace_all: true }, + ]; + expect(() => applyTextEdits('aa', edits, { ...limits, maxOutputBytes: 7 })).toThrow( + 'larger than', + ); + expect(applyTextEdits('aa', edits, { ...limits, maxOutputBytes: 8 }).content).toBe(''); + }); + + it('keeps exact offsets and output size after a tolerant replacement', () => { + const edits = [ + { old_text: 'two words', new_text: 'é' }, + { old_text: 'é', new_text: '😀' }, + ]; + expect(applyTextEdits('two words', edits, { ...limits, maxOutputBytes: 10 })).toEqual({ + content: '😀', + strategies: ['whitespace-normalized', 'exact'], + }); + }); +}); diff --git a/packages/api/src/agents/files/matching.ts b/packages/api/src/agents/files/matching.ts new file mode 100644 index 00000000000..cca28ad2375 --- /dev/null +++ b/packages/api/src/agents/files/matching.ts @@ -0,0 +1,468 @@ +import type { TextEdit } from '../edits'; + +export interface HostEditWorkLimits { + maxEdits: number; + maxWorkBytes: number; + maxOccurrences: number; + maxOutputBytes: number; +} + +export interface HostEditResult { + content: string; + strategies: string[]; +} + +export class HostEditError extends Error { + constructor(message: string) { + super(message); + this.name = 'HostEditError'; + } +} + +class EditWorkBudget { + private bytes = 0; + private occurrences = 0; + + constructor(private readonly limits: HostEditWorkLimits) {} + + scan(bytes: number): void { + this.bytes += bytes; + if (this.bytes > this.limits.maxWorkBytes) { + throw new HostEditError( + 'File edit processing budget exceeded; split the batch. Nothing was written.', + ); + } + } + + occurrence(): void { + if (++this.occurrences > this.limits.maxOccurrences) { + throw new HostEditError( + 'File edit occurrence budget exceeded; narrow the replacements. Nothing was written.', + ); + } + } +} + +type MatchedRange = { index: number; length: number }; + +type MatchStatus = + | { status: 'matched'; index: number; length: number; strategy: string } + | { status: 'none' } + | { status: 'ambiguous'; strategy: string; count: number; matches: MatchedRange[] }; + +/** + * Ranges a whitespace-tolerant strategy collects before it stops looking. An + * internal memory bound, not a policy: ambiguity only needs a second match, and + * exact `replace_all` never collects ranges at all. + */ +const MAX_EDIT_MATCHES = 10_000; + +/** Pieces buffered before they are flattened into one bounded output chunk. */ +const REPLACE_ALL_FLUSH_PIECES = 1_024; +const REPLACE_ALL_FLUSH_CHARS = 16 * 1024; + +/** + * `content.split(needle).join(replacement)` without one array entry per match: pieces are + * flattened into chunks of bounded size, so memory tracks the output, not the match count. + */ +function replaceAllExact( + content: string, + needle: string, + replacement: string, + budget: EditWorkBudget, +): string { + const chunks: string[] = []; + let pieces: string[] = []; + let pendingChars = 0; + const flush = () => { + chunks.push(pieces.join('')); + pieces = []; + pendingChars = 0; + }; + let cursor = 0; + for (let index = content.indexOf(needle); index !== -1; index = content.indexOf(needle, cursor)) { + budget.occurrence(); + pieces.push(content.slice(cursor, index), replacement); + pendingChars += index - cursor + replacement.length; + cursor = index + needle.length; + if (pieces.length >= REPLACE_ALL_FLUSH_PIECES || pendingChars >= REPLACE_ALL_FLUSH_CHARS) { + flush(); + } + } + pieces.push(content.slice(cursor)); + flush(); + return chunks.join(''); +} + +/** Non-overlapping exact occurrences, counted without retaining their positions. */ +function countExactMatches(content: string, needle: string, budget: EditWorkBudget): number { + let count = 0; + for ( + let index = content.indexOf(needle); + index !== -1; + index = content.indexOf(needle, index + needle.length) + ) { + budget.occurrence(); + count++; + } + return count; +} + +function countExactOccurrences(content: string, needle: string, budget: EditWorkBudget): number[] { + const indexes: number[] = []; + let start = 0; + while (start <= content.length && indexes.length <= MAX_EDIT_MATCHES) { + const index = content.indexOf(needle, start); + if (index === -1) { + break; + } + budget.occurrence(); + indexes.push(index); + start = index + Math.max(1, needle.length); + } + return indexes; +} + +function findExactMatch(content: string, needle: string, budget: EditWorkBudget): MatchStatus { + const matches = countExactOccurrences(content, needle, budget); + if (matches.length === 1) { + return { status: 'matched', index: matches[0], length: needle.length, strategy: 'exact' }; + } + if (matches.length > 1) { + return { + status: 'ambiguous', + strategy: 'exact', + count: matches.length, + matches: matches.map((index) => ({ index, length: needle.length })), + }; + } + return { status: 'none' }; +} + +function lineStarts(content: string): number[] { + const starts = [0]; + for (let i = 0; i < content.length; i++) { + if (content[i] === '\n') { + starts.push(i + 1); + } + } + return starts; +} + +function commonIndent(lines: string[]): number { + const indents = lines + .filter((line) => line.trim().length > 0) + .map((line) => { + const match = /^(\s*)/.exec(line); + return match ? match[1].length : 0; + }); + return indents.length > 0 ? Math.min(...indents) : 0; +} + +function stripCommonIndent(text: string): string { + const lines = text.split('\n'); + const indent = commonIndent(lines); + if (indent === 0) { + return text; + } + return lines.map((line) => line.slice(Math.min(indent, line.length))).join('\n'); +} + +function findLineWindowMatch( + content: string, + needle: string, + strategy: 'line-trimmed' | 'indentation-flexible', + budget: EditWorkBudget, +): MatchStatus { + const contentLines = content.split('\n'); + const needleLines = needle.split('\n'); + if (needleLines.length > contentLines.length) { + return { status: 'none' }; + } + + const starts = lineStarts(content); + const matches: Array<{ index: number; length: number }> = []; + const addMatch = (startLine: number) => { + const index = starts[startLine]; + const endLine = startLine + needleLines.length; + const end = endLine < starts.length ? starts[endLine] - 1 : content.length; + budget.occurrence(); + matches.push({ index, length: end - index }); + }; + + if (strategy === 'line-trimmed') { + // Intern normalized lines so KMP compares integer IDs, not overlapping strings. + const ids = new Map(); + const pattern = needleLines.map((line) => { + const normalized = line.trimEnd(); + let id = ids.get(normalized); + if (id == null) { + id = ids.size; + ids.set(normalized, id); + } + return id; + }); + const prefixes = new Uint32Array(pattern.length); + let matched = 0; + for (let i = 1; i < pattern.length; i++) { + while (matched > 0 && pattern[i] !== pattern[matched]) { + matched = prefixes[matched - 1]; + } + if (pattern[i] === pattern[matched]) matched++; + prefixes[i] = matched; + } + + matched = 0; + for (let i = 0; i < contentLines.length && matches.length <= MAX_EDIT_MATCHES; i++) { + const id = ids.get(contentLines[i].trimEnd()); + while (matched > 0 && id !== pattern[matched]) { + matched = prefixes[matched - 1]; + } + if (id === pattern[matched]) matched++; + if (matched === pattern.length) { + addMatch(i - pattern.length + 1); + // Keep overlapping occurrences for ambiguity detection and replace_all. + matched = prefixes[matched - 1]; + } + } + } else { + // This fallback runs only after line-trimmed and whitespace-normalized miss. + // Two nonblank needle lines imply at least two tokens: any indentation match + // would already have matched whitespace-normalized. An all-blank needle would + // already have matched line-trimmed. Only a single nonblank line remains. + let anchor = -1; + for (let i = 0; i < needleLines.length; i++) { + if (needleLines[i].trim().length === 0) continue; + if (anchor !== -1) return { status: 'none' }; + anchor = i; + } + if (anchor === -1) return { status: 'none' }; + + const normalizedNeedle = stripCommonIndent(needle).split('\n'); + const nonblank = new Uint32Array(contentLines.length + 1); + for (let i = 0; i < contentLines.length; i++) { + nonblank[i + 1] = nonblank[i] + Number(contentLines[i].trim().length > 0); + } + for (let i = anchor; i < contentLines.length && matches.length <= MAX_EDIT_MATCHES; i++) { + if (nonblank[i + 1] === nonblank[i]) continue; + const start = i - anchor; + const end = start + needleLines.length; + if (end > contentLines.length) break; + if (nonblank[end] - nonblank[start] !== 1) continue; + const indent = contentLines[i].length - contentLines[i].trimStart().length; + let matchesNeedle = true; + // Eligible windows have one nonblank line at a fixed offset. Each file line + // can belong to at most two of them, so these comparisons stay linear too. + for (let j = 0; j < needleLines.length; j++) { + if (contentLines[start + j].slice(indent) !== normalizedNeedle[j]) { + matchesNeedle = false; + break; + } + } + if (matchesNeedle) addMatch(start); + } + } + + if (matches.length === 1) { + return { status: 'matched', ...matches[0], strategy }; + } + if (matches.length > 1) { + return { status: 'ambiguous', strategy, count: matches.length, matches }; + } + return { status: 'none' }; +} + +function escapeRegExp(value: string): string { + return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); +} + +function findWhitespaceNormalizedMatch( + content: string, + needle: string, + budget: EditWorkBudget, +): MatchStatus { + const tokens = needle.trim().split(/\s+/).filter(Boolean); + if (tokens.length < 2) { + return { status: 'none' }; + } + const pattern = tokens.map(escapeRegExp).join('\\s+'); + const regex = new RegExp(pattern, 'g'); + const matches: Array<{ index: number; length: number }> = []; + let match: RegExpExecArray | null; + while (matches.length <= MAX_EDIT_MATCHES && (match = regex.exec(content)) != null) { + budget.occurrence(); + matches.push({ index: match.index, length: match[0].length }); + if (match[0].length === 0) { + regex.lastIndex += 1; + } + } + if (matches.length === 1) { + return { status: 'matched', ...matches[0], strategy: 'whitespace-normalized' }; + } + if (matches.length > 1) { + return { + status: 'ambiguous', + strategy: 'whitespace-normalized', + count: matches.length, + matches, + }; + } + return { status: 'none' }; +} + +function findReplacementMatch( + content: string, + needle: string, + budget: EditWorkBudget, + contentBytes: number, + needleBytes: number, +): MatchStatus { + const scanBytes = contentBytes + needleBytes; + budget.scan(scanBytes); + const exact = findExactMatch(content, needle, budget); + if (exact.status !== 'none') { + return exact; + } + budget.scan(scanBytes); + const lineTrimmed = findLineWindowMatch(content, needle, 'line-trimmed', budget); + if (lineTrimmed.status !== 'none') { + return lineTrimmed; + } + budget.scan(scanBytes); + const whitespaceNormalized = findWhitespaceNormalizedMatch(content, needle, budget); + if (whitespaceNormalized.status !== 'none') { + return whitespaceNormalized; + } + budget.scan(scanBytes); + return findLineWindowMatch(content, needle, 'indentation-flexible', budget); +} + +/** Keeps the earliest of any overlapping matches so replacements never collide. */ +function nonOverlapping(matches: readonly MatchedRange[]): MatchedRange[] { + const kept: MatchedRange[] = []; + let end = -1; + for (const match of [...matches].sort((a, b) => a.index - b.index)) { + if (match.index < end) continue; + kept.push(match); + end = match.index + match.length; + } + return kept; +} + +function describeMatchCount(count: number): string { + return count > MAX_EDIT_MATCHES ? `more than ${MAX_EDIT_MATCHES}` : String(count); +} + +/** + * The size `replace_all` would produce, computed before any replacement text is + * built so an oversized result is refused without allocating it. + */ +function projectedReplaceAllBytes( + content: string, + matches: readonly MatchedRange[], + contentBytes: number, + replacementBytes: number, +): number { + let bytes = contentBytes; + for (const match of matches) { + bytes += + replacementBytes - + Buffer.byteLength(content.slice(match.index, match.index + match.length), 'utf8'); + } + return bytes; +} + +function replaceMatches(content: string, matches: readonly MatchedRange[], text: string): string { + let result = ''; + let cursor = 0; + for (const match of matches) { + result += content.slice(cursor, match.index) + text; + cursor = match.index + match.length; + } + return result + content.slice(cursor); +} + +export function applyTextEdits( + content: string, + edits: TextEdit[], + limits: HostEditWorkLimits, +): HostEditResult { + if (edits.length > limits.maxEdits) { + throw new HostEditError( + `File edits are limited to ${limits.maxEdits} replacements per call. Nothing was written.`, + ); + } + const MAX_AUTHORING_BYTES = limits.maxOutputBytes; + const budget = new EditWorkBudget(limits); + if (Buffer.byteLength(content) > MAX_AUTHORING_BYTES) { + throw new HostEditError('File exceeds the authoring size limit. Nothing was written.'); + } + let working = content; + const strategies: string[] = []; + + for (const edit of edits) { + const workingBytes = Buffer.byteLength(working); + const oldBytes = Buffer.byteLength(edit.old_text); + const newBytes = Buffer.byteLength(edit.new_text); + budget.scan(oldBytes + newBytes); + if (edit.replace_all === true) budget.scan(workingBytes + oldBytes); + const exactCount = + edit.replace_all === true ? countExactMatches(working, edit.old_text, budget) : 0; + if (exactCount > 0) { + const projectedBytes = workingBytes + exactCount * (newBytes - oldBytes); + if (projectedBytes > MAX_AUTHORING_BYTES) { + throw new HostEditError( + `replace_all would make the file larger than ${MAX_AUTHORING_BYTES} bytes; nothing was written.`, + ); + } + budget.scan(workingBytes + projectedBytes); + working = replaceAllExact(working, edit.old_text, edit.new_text, budget); + strategies.push(exactCount > 1 ? `exact x${exactCount}` : 'exact'); + continue; + } + const match = findReplacementMatch(working, edit.old_text, budget, workingBytes, oldBytes); + if (match.status === 'none') { + throw new HostEditError('old_text did not match the file content.'); + } + if (match.status === 'ambiguous' && edit.replace_all !== true) { + throw new HostEditError( + `old_text matched ${describeMatchCount(match.count)} locations with ${match.strategy}; make it unique or set replace_all before retrying.`, + ); + } + if (match.status === 'ambiguous') { + if (match.count > MAX_EDIT_MATCHES) { + throw new HostEditError( + `replace_all with whitespace-tolerant matching is limited to ${MAX_EDIT_MATCHES} locations, and old_text matched more; copy the exact text or narrow old_text before retrying.`, + ); + } + const matches = nonOverlapping(match.matches); + const projectedBytes = projectedReplaceAllBytes(working, matches, workingBytes, newBytes); + if (projectedBytes > MAX_AUTHORING_BYTES) { + throw new HostEditError( + `replace_all would make the file larger than ${MAX_AUTHORING_BYTES} bytes; nothing was written.`, + ); + } + budget.scan(workingBytes + projectedBytes); + working = replaceMatches(working, matches, edit.new_text); + strategies.push(`${match.strategy} x${matches.length}`); + continue; + } + const projectedBytes = + workingBytes - + (match.strategy === 'exact' + ? oldBytes + : Buffer.byteLength(working.slice(match.index, match.index + match.length))) + + newBytes; + if (projectedBytes > MAX_AUTHORING_BYTES) { + throw new HostEditError( + `edited content exceeds ${MAX_AUTHORING_BYTES} byte limit; nothing was written.`, + ); + } + budget.scan(workingBytes + projectedBytes); + working = + working.slice(0, match.index) + edit.new_text + working.slice(match.index + match.length); + strategies.push(match.strategy); + } + + return { content: working, strategies }; +} diff --git a/packages/api/src/agents/files/processing.spec.ts b/packages/api/src/agents/files/processing.spec.ts new file mode 100644 index 00000000000..279c8d1045b --- /dev/null +++ b/packages/api/src/agents/files/processing.spec.ts @@ -0,0 +1,273 @@ +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import type { TextEdit } from '../edits'; +import { createHostEditProcessor } from './processing'; + +const workerPath = path.join( + path.dirname(require.resolve('@librechat/api')), + 'agents/files/edit-worker.cjs', +); +const edit: TextEdit = { old_text: 'a', new_text: 'b', replace_all: true }; +const amplification = (): TextEdit[] => [ + ...Array.from({ length: 23 }, () => ({ old_text: 'a', new_text: 'aa', replace_all: true })), + ...Array.from({ length: 30 }, (_, i) => ({ + old_text: i % 2 ? 'b' : 'a', + new_text: i % 2 ? 'a' : 'b', + replace_all: true, + })), + { old_text: 'a', new_text: '', replace_all: true }, +]; + +describe('bounded host edit workers', () => { + const processor = createHostEditProcessor(workerPath); + afterAll(async () => processor.close()); + + it('rejects compact expansion and contraction while unrelated timers keep running', async () => { + let ticks = 0; + const timer = setInterval(() => { + ticks++; + }, 0); + try { + await expect(processor.apply('ax', amplification())).rejects.toThrow('budget exceeded'); + expect(ticks).toBeGreaterThan(0); + } finally { + clearInterval(timer); + } + }); + + it('bounds occurrence work independently of scanned bytes', async () => { + await expect(processor.apply('aaa', [edit], { maxOccurrences: 2 })).rejects.toThrow( + 'occurrence budget', + ); + }); + + it.each([9, 10])( + 'supports an ordinary exact edit of a %i MiB file with omitted limits', + async (mib) => { + const content = 'start\n' + 'x'.repeat(mib * 1024 * 1024 - 10) + '\nend'; + const result = await processor.apply(content, [{ old_text: 'start', new_text: 'begin' }]); + expect(Buffer.byteLength(result.content)).toBe(mib * 1024 * 1024); + expect(result.content.startsWith('begin\n')).toBe(true); + expect(result.strategies).toEqual(['exact']); + }, + ); + + it.each([ + ['9 MiB ASCII', 'a'.repeat(9 * 1024 * 1024)], + ['10 MiB ASCII', 'a'.repeat(10 * 1024 * 1024)], + ['10 MiB UTF-8', 'é'.repeat(5 * 1024 * 1024)], + ])('supports full matching context for %s with omitted limits', async (_label, content) => { + const replacement = 'b' + content.slice(1); + const result = await processor.apply(content, [{ old_text: content, new_text: replacement }]); + expect(result.content).toBe(replacement); + expect(result.strategies).toEqual(['exact']); + }); + + it('supports replace_all with full 10 MiB context and a same-size replacement', async () => { + const content = 'a'.repeat(10 * 1024 * 1024); + const replacement = 'b' + content.slice(1); + await expect( + processor.apply(content, [{ old_text: content, new_text: replacement, replace_all: true }]), + ).resolves.toEqual({ + content: replacement, + strategies: ['exact'], + }); + }); + + it.each([false, true])( + 'admits a full-file replacement into a smaller result, replace_all=%s', + async (replaceAll) => { + const content = 'a'.repeat(10 * 1024 * 1024); + await expect( + processor.apply(content, [{ old_text: content, new_text: 'b', replace_all: replaceAll }]), + ).resolves.toEqual({ + content: 'b', + strategies: ['exact'], + }); + }, + ); + + it('bounds every intermediate result, even when a later edit would shrink it', async () => { + await expect( + processor.apply( + 'a', + [ + { old_text: 'a', new_text: 'x'.repeat(10 * 1024 * 1024 + 1) }, + { old_text: 'x', new_text: '', replace_all: true }, + ], + { maxWorkBytes: 64 * 1024 * 1024 }, + ), + ).rejects.toThrow('byte limit'); + }); + + it('retains ordered edits, exact-match strategy, and UTF-16 offsets', async () => { + await expect( + processor.apply('😀 a\r\na', [edit, { old_text: 'b', new_text: 'c', replace_all: true }]), + ).resolves.toEqual({ + content: '😀 c\r\nc', + strategies: ['exact x2', 'exact x2'], + }); + }); + + it('checks server-side count limits without starting work', async () => { + await expect(processor.apply('a', [edit, edit], { maxEdits: 1 })).rejects.toThrow( + 'limited to 1', + ); + }); + + it('fails closed with a safe error on invalid configuration', async () => { + await expect(processor.apply('a', [edit], { maxConcurrent: 0 })).rejects.toThrow( + 'configuration is invalid', + ); + }); + + it('rejects pre-cancelled jobs', async () => { + const controller = new AbortController(); + controller.abort(); + await expect(processor.apply('a', [edit], undefined, controller.signal)).rejects.toMatchObject({ + name: 'AbortError', + }); + }); +}); + +describe('worker lifecycle', () => { + const fixture = path.join(__dirname, '__fixtures__', 'edit-worker.cjs'); + + it('sanitizes synchronous thread creation failures without consuming admission', async () => { + const processor = createHostEditProcessor(fixture); + const startup = jest + .spyOn( + jest.requireActual('node:worker_threads'), + 'Worker', + ) + .mockImplementationOnce(() => { + throw new Error('PRIVATE-STARTUP /operator/absolute/edit-worker.cjs'); + }); + try { + await expect(processor.apply('ready', [edit], { maxConcurrent: 1 })).rejects.toThrow( + /^File edit processing failed\. Nothing was written\.$/, + ); + startup.mockRestore(); + await expect(processor.apply('ready', [edit], { maxConcurrent: 1 })).resolves.toEqual({ + content: 'ready', + strategies: [], + }); + } finally { + startup.mockRestore(); + await processor.close(); + } + }); + + it('rejects oversized clone input before acquiring a worker', async () => { + const processor = createHostEditProcessor(fixture); + const startup = jest.spyOn( + jest.requireActual('node:worker_threads'), + 'Worker', + ); + try { + await expect( + processor.apply('a', [{ old_text: 'a'.repeat(512), new_text: 'b' }], { + maxWorkBytes: 1024, + }), + ).rejects.toThrow('budget exceeded'); + expect(startup).not.toHaveBeenCalled(); + } finally { + startup.mockRestore(); + await processor.close(); + } + }); + + it('has no queue and releases capacity only after cancellation terminates the worker', async () => { + const processor = createHostEditProcessor(fixture); + const controller = new AbortController(); + const pending = processor.apply('wait', [edit], { maxConcurrent: 1 }, controller.signal); + try { + await expect(processor.apply('ready', [edit], { maxConcurrent: 1 })).rejects.toThrow('busy'); + controller.abort(); + await expect(pending).rejects.toMatchObject({ name: 'AbortError' }); + await expect(processor.apply('ready', [edit], { maxConcurrent: 1 })).resolves.toEqual({ + content: 'ready', + strategies: [], + }); + } finally { + controller.abort(); + await processor.close(); + } + }); + + it('terminates stalled jobs at the deadline and permits later reuse', async () => { + const processor = createHostEditProcessor(fixture); + try { + await expect(processor.apply('wait', [edit], { timeoutMs: 100 })).rejects.toThrow( + 'timed out', + ); + await expect(processor.apply('ready', [edit])).resolves.toEqual({ + content: 'ready', + strategies: [], + }); + } finally { + await processor.close(); + } + }); + + it.each([10_000, 10_001])( + 'rejects a reply at monotonic time %i before the timeout callback runs', + async (replyTime) => { + const processor = createHostEditProcessor(fixture); + const persisted = jest.fn(); + const clock = jest + .spyOn(performance, 'now') + .mockReturnValueOnce(0) + .mockReturnValueOnce(0) + .mockReturnValue(replyTime); + try { + const result = processor.apply('ready', [edit], { timeoutMs: 10_000, maxConcurrent: 1 }); + await expect(result.then(persisted)).rejects.toThrow('timed out'); + expect(clock).toHaveBeenCalledTimes(3); + expect(persisted).not.toHaveBeenCalled(); + clock.mockRestore(); + await expect(processor.apply('ready', [edit], { maxConcurrent: 1 })).resolves.toEqual({ + content: 'ready', + strategies: [], + }); + } finally { + clock.mockRestore(); + await processor.close(); + } + }, + ); + + it('accepts a reply strictly before the monotonic deadline', async () => { + const processor = createHostEditProcessor(fixture); + const clock = jest + .spyOn(performance, 'now') + .mockReturnValueOnce(0) + .mockReturnValueOnce(0) + .mockReturnValue(9999); + try { + await expect(processor.apply('ready', [edit], { timeoutMs: 10_000 })).resolves.toEqual({ + content: 'ready', + strategies: [], + }); + expect(clock).toHaveBeenCalledTimes(3); + } finally { + clock.mockRestore(); + await processor.close(); + } + }); + + it('sanitizes worker crashes and recovers capacity', async () => { + const processor = createHostEditProcessor(fixture); + try { + await expect(processor.apply('crash', [edit])).rejects.toThrow( + /^File edit processing failed\. Nothing was written\.$/, + ); + await expect(processor.apply('ready', [edit])).resolves.toEqual({ + content: 'ready', + strategies: [], + }); + } finally { + await processor.close(); + } + }); +}); diff --git a/packages/api/src/agents/files/processing.ts b/packages/api/src/agents/files/processing.ts new file mode 100644 index 00000000000..327d0cd5c38 --- /dev/null +++ b/packages/api/src/agents/files/processing.ts @@ -0,0 +1,209 @@ +import path from 'node:path'; +import { Worker } from 'node:worker_threads'; +import { performance } from 'node:perf_hooks'; +import { hostFileEditLimitsSchema } from 'librechat-data-provider'; +import type { HostFileEditLimits } from 'librechat-data-provider'; +import type { HostEditJob, HostEditReply } from './edit-worker'; +import type { HostEditResult } from './matching'; +import type { TextEdit } from '../edits'; +import { HostEditError } from './matching'; + +interface EditWorkerSlot { + worker: Worker; + idleTimer?: NodeJS.Timeout; + active: boolean; +} + +/** No waiting queue: admitted jobs retain at most one input/output per bounded worker. */ +export function createHostEditProcessor(workerPath: string): { + apply: ( + content: string, + edits: TextEdit[], + configured?: Partial, + signal?: AbortSignal, + ) => Promise; + close: () => Promise; +} { + const slots = new Set(); + let active = 0; + let closed = false; + + const remove = (slot: EditWorkerSlot): void => { + clearTimeout(slot.idleTimer); + slots.delete(slot); + }; + + const acquire = (): EditWorkerSlot => { + for (const slot of slots) { + if (slot.active) continue; + clearTimeout(slot.idleTimer); + slot.active = true; + slot.worker.ref(); + return slot; + } + const worker = new Worker(workerPath, { + execArgv: [], + // The pure matcher never needs application services or an unbounded heap. + resourceLimits: { maxOldGenerationSizeMb: 128 }, + }); + const slot: EditWorkerSlot = { worker, active: true }; + slots.add(slot); + worker.on('error', () => { + if (!slot.active) remove(slot); + }); + worker.on('exit', () => remove(slot)); + return slot; + }; + + return { + async apply(content, edits, configured, signal): Promise { + signal?.throwIfAborted(); + const parsed = hostFileEditLimitsSchema.safeParse(configured ?? {}); + if (!parsed.success) { + throw new HostEditError('File edit configuration is invalid. Nothing was written.'); + } + const limits = parsed.data; + if (edits.length > limits.maxEdits) { + throw new HostEditError( + `File edits are limited to ${limits.maxEdits} replacements per call. Nothing was written.`, + ); + } + if (closed || active >= limits.maxConcurrent) { + throw new HostEditError('File edit processing is busy; retry later. Nothing was written.'); + } + const maxOutputBytes = 10 * 1024 * 1024; + if (content.length > maxOutputBytes) { + throw new HostEditError('File exceeds the authoring size limit. Nothing was written.'); + } + // Bound UTF-16 clone bytes without scanning; the worker accounts UTF-8 processing. + let inputUnits = content.length; + for (const edit of edits) { + inputUnits += edit.old_text.length + edit.new_text.length; + if (inputUnits > limits.maxWorkBytes / 2) { + throw new HostEditError( + 'File edit processing budget exceeded; split the batch. Nothing was written.', + ); + } + } + const deadline = performance.now() + limits.timeoutMs; + let slot: EditWorkerSlot; + try { + slot = acquire(); + } catch { + throw new HostEditError('File edit processing failed. Nothing was written.'); + } + active++; + const job: HostEditJob = { + content, + edits, + limits: { ...limits, maxOutputBytes }, + }; + return await new Promise((resolve, reject) => { + let settled = false; + const finish = async ( + result: HostEditResult | undefined, + error: Error | undefined, + terminate: boolean, + ): Promise => { + if (settled) return; + settled = true; + clearTimeout(timer); + signal?.removeEventListener('abort', abort); + slot.worker.off('message', message); + slot.worker.off('error', failed); + slot.worker.off('exit', exited); + if (terminate || closed) { + await slot.worker.terminate().catch(() => undefined); + remove(slot); + } else { + slot.active = false; + slot.worker.unref(); + slot.idleTimer = setTimeout(() => { + remove(slot); + void slot.worker.terminate(); + }, 30_000); + slot.idleTimer.unref(); + } + active--; + if (error) reject(error); + else if (result) resolve(result); + }; + const abort = (): void => { + void finish( + undefined, + new DOMException('File edit cancelled. Nothing was written.', 'AbortError'), + true, + ); + }; + const failed = (): void => { + void finish( + undefined, + new HostEditError('File edit processing failed. Nothing was written.'), + true, + ); + }; + const exited = (): void => failed(); + const timeout = (): void => { + void finish( + undefined, + new HostEditError('File edit processing timed out. Nothing was written.'), + true, + ); + }; + const message = (reply: HostEditReply): void => { + if (signal?.aborted) { + abort(); + return; + } + if (performance.now() >= deadline) { + timeout(); + return; + } + void finish( + reply.ok ? reply.result : undefined, + reply.ok ? undefined : new HostEditError(reply.message), + false, + ); + }; + const timer = setTimeout(timeout, Math.max(0, deadline - performance.now())); + slot.worker.once('message', message); + slot.worker.once('error', failed); + slot.worker.once('exit', exited); + signal?.addEventListener('abort', abort, { once: true }); + if (signal?.aborted) { + abort(); + return; + } + try { + slot.worker.postMessage(job); + } catch { + failed(); + } + }); + }, + async close(): Promise { + closed = true; + await Promise.all( + [...slots].map(async (slot) => { + clearTimeout(slot.idleTimer); + await slot.worker.terminate(); + remove(slot); + }), + ); + }, + }; +} + +let processor: ReturnType | undefined; + +export function applyHostTextEdits( + content: string, + edits: TextEdit[], + configured?: Partial, + signal?: AbortSignal, +): Promise { + processor ??= createHostEditProcessor( + path.join(path.dirname(require.resolve('@librechat/api')), 'agents/files/edit-worker.cjs'), + ); + return processor.apply(content, edits, configured, signal); +} diff --git a/packages/api/src/agents/handlers.spec.ts b/packages/api/src/agents/handlers.spec.ts index c9aed0897d8..1727ad610e7 100644 --- a/packages/api/src/agents/handlers.spec.ts +++ b/packages/api/src/agents/handlers.spec.ts @@ -1,3 +1,6 @@ +import { performance } from 'node:perf_hooks'; +import { applyTextEdits, HostEditError } from './files/matching'; +import * as hostEditProcessing from './files/processing'; jest.mock('./prewarm', () => ({ markSandboxReady: jest.fn(), })); @@ -5310,7 +5313,7 @@ describe('createToolExecuteHandler', () => { it('normalizes each line once on a large multiline miss', async () => { const content = ' '.repeat(19).concat('\n').repeat(20_000); const oldText = '\t'.repeat(19).concat('\n').repeat(800) + 'missing'; - const { handler, saveSkillFileContent } = makeMatchingHandler(content); + const { saveSkillFileContent } = makeMatchingHandler(content); const trimEnd = String.prototype.trimEnd; const budget = 20_001 + 801; let normalizations = 0; @@ -5320,25 +5323,21 @@ describe('createToolExecuteHandler', () => { if (++normalizations > budget) throw new Error('Repeated window normalization'); return trimEnd.call(this); }); - let result: ToolExecuteResult; + let message = ''; try { - [result] = await invokeHandler(handler, [ - { - id: 'call_matching_large_miss', - name: 'edit_file', - args: { - path: 'skills/matching-skill/references/a.md', - old_text: oldText, - new_text: 'changed', - }, - }, - ]); + applyTextEdits(content, [{ old_text: oldText, new_text: 'changed' }], { + maxEdits: 100, + maxWorkBytes: 33554432, + maxOccurrences: 100000, + maxOutputBytes: 10 * 1024 * 1024, + }); + } catch (error) { + message = error instanceof Error ? error.message : ''; } finally { spy.mockRestore(); } expect(normalizations).toBe(budget); - expect(result.status).toBe('error'); - expect(result.errorMessage).toContain('old_text did not match'); + expect(message).toContain('old_text did not match'); expect(saveSkillFileContent).not.toHaveBeenCalled(); }); @@ -5488,7 +5487,7 @@ describe('createToolExecuteHandler', () => { }, ); - it('replaces any number of exact matches when the result fits', async () => { + it('replaces more than 10000 exact matches within the work budget', async () => { const saveSkillFileContent = jest.fn(async () => ({ bytes: 10_001, relativePath: 'references/a.md', @@ -5535,6 +5534,299 @@ describe('createToolExecuteHandler', () => { ); }); + const amplificationEdits = () => [ + ...Array.from({ length: 23 }, () => ({ old_text: 'a', new_text: 'aa', replace_all: true })), + ...Array.from({ length: 30 }, (_, i) => ({ + old_text: i % 2 ? 'b' : 'a', + new_text: i % 2 ? 'a' : 'b', + replace_all: true, + })), + { old_text: 'a', new_text: '', replace_all: true }, + ]; + + it.each(['SKILL.md', 'references/a.md'])( + 'rejects compact amplification atomically for %s without starving timers', + async (file) => { + const updateSkill = jest.fn(); + const saveSkillFileContent = jest.fn(); + const handler = makeAuthoringHandler({ + getSkillByName: jest.fn(async () => ({ + _id: SKILL_ID, + name: 'bounded-skill', + body: 'ax', + fileCount: 1, + version: 1, + })), + getSkillFileByPath: jest.fn(async () => ({ + content: 'ax', + isBinary: false, + bytes: 2, + mimeType: 'text/markdown', + filepath: '/tmp/a.md', + file_id: 'revision-1', + source: 'local', + relativePath: 'references/a.md', + })), + updateSkill, + saveSkillFileContent, + }); + let timerFired = false; + const timer = setTimeout(() => { + timerFired = true; + }, 0); + const [result] = await invokeHandler(handler, [ + { + id: 'amplification', + name: 'edit_file', + args: { path: `skills/bounded-skill/${file}`, edits: amplificationEdits() }, + }, + ]); + clearTimeout(timer); + expect(result.status).toBe('error'); + expect(result.errorMessage).toContain('budget exceeded'); + expect(timerFired).toBe(true); + expect(updateSkill).not.toHaveBeenCalled(); + expect(saveSkillFileContent).not.toHaveBeenCalled(); + }, + ); + + it.each(['SKILL.md', 'references/a.md'])( + 'does not persist a late worker reply for %s', + async (file) => { + const updateSkill = jest.fn(); + const saveSkillFileContent = jest.fn(); + const handler = makeAuthoringHandler( + { + getSkillByName: jest.fn(async () => ({ + _id: SKILL_ID, + name: 'bounded-skill', + body: 'ax', + fileCount: 1, + version: 1, + })), + getSkillFileByPath: jest.fn(async () => ({ + content: 'ax', + isBinary: false, + bytes: 2, + mimeType: 'text/markdown', + filepath: '/tmp/a.md', + file_id: 'revision-1', + source: 'local', + relativePath: 'references/a.md', + })), + updateSkill, + saveSkillFileContent, + }, + { + req: { + user: { id: 'user-1' }, + config: { endpoints: { agents: { hostFileEdits: { timeoutMs: 10000 } } } }, + }, + }, + ); + const clock = jest + .spyOn(performance, 'now') + .mockReturnValueOnce(0) + .mockReturnValueOnce(0) + .mockReturnValue(10000); + try { + const [result] = await invokeHandler(handler, [ + { + id: 'late_reply', + name: 'edit_file', + args: { path: `skills/bounded-skill/${file}`, old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.status).toBe('error'); + expect(result.errorMessage).toContain('timed out'); + expect(updateSkill).not.toHaveBeenCalled(); + expect(saveSkillFileContent).not.toHaveBeenCalled(); + } finally { + clock.mockRestore(); + } + }, + ); + + it.each(['SKILL.md', 'references/a.md'])( + 'sanitizes operational edit failures before exposing %s results', + async (file) => { + const updateSkill = jest.fn(); + const saveSkillFileContent = jest.fn(); + const handler = makeAuthoringHandler({ + getSkillByName: jest.fn(async () => ({ + _id: SKILL_ID, + name: 'bounded-skill', + body: 'ax', + fileCount: 1, + version: 1, + })), + getSkillFileByPath: jest.fn(async () => ({ + content: 'ax', + isBinary: false, + bytes: 2, + mimeType: 'text/markdown', + filepath: '/tmp/a.md', + file_id: 'revision-1', + source: 'local', + relativePath: 'references/a.md', + })), + updateSkill, + saveSkillFileContent, + }); + const process = jest + .spyOn(hostEditProcessing, 'applyHostTextEdits') + .mockImplementationOnce(() => { + throw new Error('PRIVATE-STARTUP /operator/absolute/edit-worker.cjs'); + }); + try { + const [result] = await invokeHandler(handler, [ + { + id: 'failed_startup', + name: 'edit_file', + args: { path: `skills/bounded-skill/${file}`, old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.errorMessage).toBe('File edit processing failed. Nothing was written.'); + expect(JSON.stringify(result)).not.toContain('PRIVATE-STARTUP'); + expect(JSON.stringify(result)).not.toContain('/operator/absolute'); + expect(updateSkill).not.toHaveBeenCalled(); + expect(saveSkillFileContent).not.toHaveBeenCalled(); + } finally { + process.mockRestore(); + } + }, + ); + + it('honors deployment work limits across alternating edits, including contraction', async () => { + const saveSkillFileContent = jest.fn(); + const handler = makeAuthoringHandler( + { + getSkillByName: jest.fn(async () => ({ + _id: SKILL_ID, + name: 'bounded-skill', + body: '# Existing', + fileCount: 1, + version: 1, + })), + getSkillFileByPath: jest.fn(async () => ({ + content: 'a'.repeat(1000), + isBinary: false, + bytes: 1000, + mimeType: 'text/markdown', + filepath: '/tmp/a.md', + file_id: 'revision-1', + source: 'local', + relativePath: 'references/a.md', + })), + saveSkillFileContent, + }, + { + req: { + user: { id: 'user-1' }, + config: { + endpoints: { + agents: { hostFileEdits: { maxWorkBytes: 12000, maxOccurrences: 100000 } }, + }, + }, + }, + }, + ); + const [result] = await invokeHandler(handler, [ + { + id: 'alternating', + name: 'edit_file', + args: { + path: 'skills/bounded-skill/references/a.md', + edits: [ + { old_text: 'a', new_text: 'b', replace_all: true }, + { old_text: 'b', new_text: 'a', replace_all: true }, + { old_text: 'a', new_text: 'b', replace_all: true }, + { old_text: 'b', new_text: 'a', replace_all: true }, + { old_text: 'a', new_text: '', replace_all: true }, + ], + }, + }, + ]); + expect(result.errorMessage).toContain('processing budget exceeded'); + expect(saveSkillFileContent).not.toHaveBeenCalled(); + }); + + it('cancels host processing without writing an intermediate skill revision', async () => { + const controller = new AbortController(); + const saveSkillFileContent = jest.fn(); + const handler = makeAuthoringHandler({ + runSignal: controller.signal, + getSkillByName: jest.fn(async () => ({ + _id: SKILL_ID, + name: 'bounded-skill', + body: '# Existing', + fileCount: 1, + version: 1, + })), + getSkillFileByPath: jest.fn(async () => ({ + content: 'ax', + isBinary: false, + bytes: 2, + mimeType: 'text/markdown', + filepath: '/tmp/a.md', + file_id: 'revision-1', + source: 'local', + relativePath: 'references/a.md', + })), + saveSkillFileContent, + }); + const timer = setTimeout(() => controller.abort(), 0); + const [result] = await invokeHandler(handler, [ + { + id: 'cancelled', + name: 'edit_file', + args: { path: 'skills/bounded-skill/references/a.md', edits: amplificationEdits() }, + }, + ]); + clearTimeout(timer); + expect(result.status).toBe('error'); + expect(saveSkillFileContent).not.toHaveBeenCalled(); + }); + + it('checks cancellation again after final permission checks and before saving', async () => { + const controller = new AbortController(); + const saveSkillFileContent = jest.fn(); + const handler = makeAuthoringHandler({ + runSignal: controller.signal, + canEditSkill: jest.fn(async () => { + controller.abort(); + return true; + }), + getSkillByName: jest.fn(async () => ({ + _id: SKILL_ID, + name: 'bounded-skill', + body: '# Existing', + fileCount: 1, + version: 1, + })), + getSkillFileByPath: jest.fn(async () => ({ + content: 'ax', + isBinary: false, + bytes: 2, + mimeType: 'text/markdown', + filepath: '/tmp/a.md', + file_id: 'revision-1', + source: 'local', + relativePath: 'references/a.md', + })), + saveSkillFileContent, + }); + const [result] = await invokeHandler(handler, [ + { + id: 'cancel_before_save', + name: 'edit_file', + args: { path: 'skills/bounded-skill/references/a.md', old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.status).toBe('error'); + expect(saveSkillFileContent).not.toHaveBeenCalled(); + }); + it('blocks authoring hidden skills unless they were primed this turn', async () => { const updateSkill = jest.fn(); const handler = makeAuthoringHandler( @@ -5757,6 +6049,218 @@ describe('createToolExecuteHandler', () => { }); } + it('rejects exact replacement amplification before writing a non-attached sandbox file', async () => { + const writeSandboxFile = jest.fn(); + const handler = makeSandboxAuthoringHandler({ + readSandboxFile: jest.fn(async () => ({ content: 'ax' })), + writeSandboxFile, + }); + const edits = Array.from({ length: 23 }, () => ({ + old_text: 'a', + new_text: 'aa', + replace_all: true, + })); + const [result] = await invokeHandler(handler, [ + { + id: 'sandbox_amplification', + name: 'edit_file', + args: { path: '/mnt/data/a.txt', edits }, + }, + ]); + expect(result.status).toBe('error'); + expect(result.errorMessage).toContain('budget exceeded'); + expect(writeSandboxFile).not.toHaveBeenCalled(); + }); + + it('does not persist a late non-attached sandbox edit', async () => { + const writeSandboxFile = jest.fn(); + const handler = makeSandboxAuthoringHandler( + { readSandboxFile: jest.fn(async () => ({ content: 'ax' })), writeSandboxFile }, + { + req: { + user: { id: 'user-1' }, + config: { endpoints: { agents: { hostFileEdits: { timeoutMs: 10000 } } } }, + }, + }, + ); + const clock = jest + .spyOn(performance, 'now') + .mockReturnValueOnce(0) + .mockReturnValueOnce(0) + .mockReturnValue(10000); + try { + const [result] = await invokeHandler(handler, [ + { + id: 'late_sandbox_reply', + name: 'edit_file', + args: { path: '/mnt/data/a.txt', old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.status).toBe('error'); + expect(result.errorMessage).toContain('timed out'); + expect(writeSandboxFile).not.toHaveBeenCalled(); + } finally { + clock.mockRestore(); + } + }); + + it.each(['pre-write', 'in-flight'])( + 'preserves %s sandbox edit cancellation as an abort', + async (stage) => { + const controller = new AbortController(); + const throwIfAborted = controller.signal.throwIfAborted.bind(controller.signal); + let checks = 0; + const guard = jest.spyOn(controller.signal, 'throwIfAborted').mockImplementation(() => { + if (++checks === 3 && stage === 'pre-write') controller.abort(); + throwIfAborted(); + }); + const writeSandboxFile = jest.fn(async () => { + controller.abort(); + throw controller.signal.reason; + }); + const warn = jest.spyOn(logger, 'warn'); + const debug = jest.spyOn(logger, 'debug'); + const handler = makeSandboxAuthoringHandler({ + runSignal: controller.signal, + readSandboxFile: jest.fn(async () => ({ content: 'ax' })), + writeSandboxFile, + }); + try { + const [result] = await invokeHandler(handler, [ + { + id: 'cancel_at_write', + name: 'edit_file', + args: { path: '/mnt/data/a.txt', old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.status).toBe('error'); + expect(result.errorMessage).not.toContain('Error writing'); + expect(debug).toHaveBeenCalledWith( + '[ON_TOOL_EXECUTE] Tool edit_file cancelled by run abort', + expect.any(Object), + ); + expect(warn).not.toHaveBeenCalledWith( + '[file_authoring] Sandbox write failed', + expect.any(Object), + ); + expect(writeSandboxFile).toHaveBeenCalledTimes(stage === 'pre-write' ? 0 : 1); + expect(checks).toBe(3); + expect(result.artifact).toBeUndefined(); + } finally { + guard.mockRestore(); + warn.mockRestore(); + debug.mockRestore(); + } + }, + ); + + it('supports an ordinary exact sandbox edit at the authoring size limit with omitted limits', async () => { + const content = + 'start\n' + ('x'.repeat(1023) + '\n').repeat(10239) + 'x'.repeat(1014) + '\nend'; + expect(Buffer.byteLength(content)).toBe(10 * 1024 * 1024); + const writeSandboxFile = jest.fn(async (_params: { content: string }) => ({ + stdout: 'written', + })); + const handler = makeSandboxAuthoringHandler({ + readSandboxFile: jest.fn(async () => ({ content })), + writeSandboxFile, + }); + const [result] = await invokeHandler(handler, [ + { + id: 'authoring_limit', + name: 'edit_file', + args: { path: '/mnt/data/large.txt', old_text: 'start', new_text: 'begin' }, + }, + ]); + expect(result.status).toBe('success'); + expect(writeSandboxFile).toHaveBeenCalledTimes(1); + expect(writeSandboxFile.mock.calls[0][0].content.startsWith('begin\n')).toBe(true); + expect(Buffer.byteLength(writeSandboxFile.mock.calls[0][0].content)).toBe(10 * 1024 * 1024); + }); + + it('persists one permitted full-context sandbox replacement with omitted limits', async () => { + const content = + 'start\n' + ('x'.repeat(1023) + '\n').repeat(10239) + 'x'.repeat(1014) + '\nend'; + const replacement = 'begin' + content.slice(5); + const writeSandboxFile = jest.fn(async (_params: { content: string }) => ({ + stdout: 'written', + })); + const handler = makeSandboxAuthoringHandler({ + readSandboxFile: jest.fn(async () => ({ content })), + writeSandboxFile, + }); + const [result] = await invokeHandler(handler, [ + { + id: 'full_context', + name: 'edit_file', + args: { path: '/mnt/data/large.txt', old_text: content, new_text: replacement }, + }, + ]); + expect(result.status).toBe('success'); + expect(writeSandboxFile).toHaveBeenCalledTimes(1); + expect(writeSandboxFile.mock.calls[0][0].content).toBe(replacement); + }); + + it.each([false, true])( + 'sanitizes sandbox processing failures, asynchronous=%s', + async (asynchronous) => { + const writeSandboxFile = jest.fn(); + const handler = makeSandboxAuthoringHandler({ + readSandboxFile: jest.fn(async () => ({ content: 'ax' })), + writeSandboxFile, + }); + const process = jest + .spyOn(hostEditProcessing, 'applyHostTextEdits') + .mockImplementationOnce(() => { + const error = new Error('PRIVATE-STARTUP /operator/absolute/edit-worker.cjs'); + if (asynchronous) return Promise.reject(error); + throw error; + }); + try { + const [result] = await invokeHandler(handler, [ + { + id: 'failed_startup', + name: 'edit_file', + args: { path: '/mnt/data/a.txt', old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.errorMessage).toBe('File edit processing failed. Nothing was written.'); + expect(JSON.stringify(result)).not.toContain('PRIVATE-STARTUP'); + expect(writeSandboxFile).not.toHaveBeenCalled(); + } finally { + process.mockRestore(); + } + }, + ); + + it('preserves host-controlled edit diagnostics', async () => { + const writeSandboxFile = jest.fn(); + const handler = makeSandboxAuthoringHandler({ + readSandboxFile: jest.fn(async () => ({ content: 'ax' })), + writeSandboxFile, + }); + const process = jest + .spyOn(hostEditProcessing, 'applyHostTextEdits') + .mockRejectedValueOnce( + new HostEditError( + 'File edit occurrence budget exceeded; narrow the replacements. Nothing was written.', + ), + ); + try { + const [result] = await invokeHandler(handler, [ + { + id: 'safe_edit_error', + name: 'edit_file', + args: { path: '/mnt/data/a.txt', old_text: 'a', new_text: 'b' }, + }, + ]); + expect(result.errorMessage).toContain('occurrence budget exceeded'); + expect(writeSandboxFile).not.toHaveBeenCalled(); + } finally { + process.mockRestore(); + } + }); + it('creates a sandbox file when it does not already exist', async () => { const readSandboxFile = jest.fn(async () => { throw new Error('cat: /mnt/data/new.txt: No such file or directory'); diff --git a/packages/api/src/agents/handlers.ts b/packages/api/src/agents/handlers.ts index 670296a101a..845398c728a 100644 --- a/packages/api/src/agents/handlers.ts +++ b/packages/api/src/agents/handlers.ts @@ -145,9 +145,11 @@ import { mergeCodeFilesIntoContext } from './codeFilesSession'; import { toolValidationFeedback } from './validationFeedback'; import { createSkillContentDigest } from './compatibility'; import { isMissingSandboxPathError } from '~/files/code'; +import { applyHostTextEdits } from './files/processing'; import { deleteSkillWithRetry } from '~/skills/cleanup'; import { resolveDownloadPath } from '~/storage/path'; import { parseFrontmatter } from '../skills/import'; +import { HostEditError } from './files/matching'; import { cleanCodeToolOutput } from './cleanup'; import { primeSkillFiles } from './skillFiles'; import { instrumentPtcToolMap } from './ptc'; @@ -1141,13 +1143,6 @@ type ParsedSkillAuthoringPath = { displayPath: string; }; -type MatchedRange = { index: number; length: number }; - -type MatchStatus = - | { status: 'matched'; index: number; length: number; strategy: string } - | { status: 'none' } - | { status: 'ambiguous'; strategy: string; count: number; matches: MatchedRange[] }; - type LoadedSkillText = | { status: 'loaded'; content: string; bytes: number; fileId?: string } | { status: 'missing' } @@ -1731,368 +1726,6 @@ function getAuthorInfo(req: ServerRequest): { }; } -/** - * Ranges a whitespace-tolerant strategy collects before it stops looking. An - * internal memory bound, not a policy: ambiguity only needs a second match, and - * exact `replace_all` never collects ranges at all. - */ -const MAX_EDIT_MATCHES = 10_000; - -/** Pieces buffered before they are flattened into one bounded output chunk. */ -const REPLACE_ALL_FLUSH_PIECES = 1_024; -const REPLACE_ALL_FLUSH_CHARS = 16 * 1024; - -/** - * `content.split(needle).join(replacement)` without one array entry per match: pieces are - * flattened into chunks of bounded size, so memory tracks the output, not the match count. - */ -function replaceAllExact(content: string, needle: string, replacement: string): string { - const chunks: string[] = []; - let pieces: string[] = []; - let pendingChars = 0; - const flush = () => { - chunks.push(pieces.join('')); - pieces = []; - pendingChars = 0; - }; - let cursor = 0; - for (let index = content.indexOf(needle); index !== -1; index = content.indexOf(needle, cursor)) { - pieces.push(content.slice(cursor, index), replacement); - pendingChars += index - cursor + replacement.length; - cursor = index + needle.length; - if (pieces.length >= REPLACE_ALL_FLUSH_PIECES || pendingChars >= REPLACE_ALL_FLUSH_CHARS) { - flush(); - } - } - pieces.push(content.slice(cursor)); - flush(); - return chunks.join(''); -} - -/** Non-overlapping exact occurrences, counted without retaining their positions. */ -function countExactMatches(content: string, needle: string): number { - let count = 0; - for ( - let index = content.indexOf(needle); - index !== -1; - index = content.indexOf(needle, index + needle.length) - ) { - count++; - } - return count; -} - -function countExactOccurrences(content: string, needle: string): number[] { - const indexes: number[] = []; - let start = 0; - while (start <= content.length && indexes.length <= MAX_EDIT_MATCHES) { - const index = content.indexOf(needle, start); - if (index === -1) { - break; - } - indexes.push(index); - start = index + Math.max(1, needle.length); - } - return indexes; -} - -function findExactMatch(content: string, needle: string): MatchStatus { - const matches = countExactOccurrences(content, needle); - if (matches.length === 1) { - return { status: 'matched', index: matches[0], length: needle.length, strategy: 'exact' }; - } - if (matches.length > 1) { - return { - status: 'ambiguous', - strategy: 'exact', - count: matches.length, - matches: matches.map((index) => ({ index, length: needle.length })), - }; - } - return { status: 'none' }; -} - -function lineStarts(content: string): number[] { - const starts = [0]; - for (let i = 0; i < content.length; i++) { - if (content[i] === '\n') { - starts.push(i + 1); - } - } - return starts; -} - -function commonIndent(lines: string[]): number { - const indents = lines - .filter((line) => line.trim().length > 0) - .map((line) => { - const match = /^(\s*)/.exec(line); - return match ? match[1].length : 0; - }); - return indents.length > 0 ? Math.min(...indents) : 0; -} - -function stripCommonIndent(text: string): string { - const lines = text.split('\n'); - const indent = commonIndent(lines); - if (indent === 0) { - return text; - } - return lines.map((line) => line.slice(Math.min(indent, line.length))).join('\n'); -} - -function findLineWindowMatch( - content: string, - needle: string, - strategy: 'line-trimmed' | 'indentation-flexible', -): MatchStatus { - const contentLines = content.split('\n'); - const needleLines = needle.split('\n'); - if (needleLines.length > contentLines.length) { - return { status: 'none' }; - } - - const starts = lineStarts(content); - const matches: Array<{ index: number; length: number }> = []; - const addMatch = (startLine: number) => { - const index = starts[startLine]; - const endLine = startLine + needleLines.length; - const end = endLine < starts.length ? starts[endLine] - 1 : content.length; - matches.push({ index, length: end - index }); - }; - - if (strategy === 'line-trimmed') { - // Intern normalized lines so KMP compares integer IDs, not overlapping strings. - const ids = new Map(); - const pattern = needleLines.map((line) => { - const normalized = line.trimEnd(); - let id = ids.get(normalized); - if (id == null) { - id = ids.size; - ids.set(normalized, id); - } - return id; - }); - const prefixes = new Uint32Array(pattern.length); - let matched = 0; - for (let i = 1; i < pattern.length; i++) { - while (matched > 0 && pattern[i] !== pattern[matched]) { - matched = prefixes[matched - 1]; - } - if (pattern[i] === pattern[matched]) matched++; - prefixes[i] = matched; - } - - matched = 0; - for (let i = 0; i < contentLines.length && matches.length <= MAX_EDIT_MATCHES; i++) { - const id = ids.get(contentLines[i].trimEnd()); - while (matched > 0 && id !== pattern[matched]) { - matched = prefixes[matched - 1]; - } - if (id === pattern[matched]) matched++; - if (matched === pattern.length) { - addMatch(i - pattern.length + 1); - // Keep overlapping occurrences for ambiguity detection and replace_all. - matched = prefixes[matched - 1]; - } - } - } else { - // This fallback runs only after line-trimmed and whitespace-normalized miss. - // Two nonblank needle lines imply at least two tokens: any indentation match - // would already have matched whitespace-normalized. An all-blank needle would - // already have matched line-trimmed. Only a single nonblank line remains. - let anchor = -1; - for (let i = 0; i < needleLines.length; i++) { - if (needleLines[i].trim().length === 0) continue; - if (anchor !== -1) return { status: 'none' }; - anchor = i; - } - if (anchor === -1) return { status: 'none' }; - - const normalizedNeedle = stripCommonIndent(needle).split('\n'); - const nonblank = new Uint32Array(contentLines.length + 1); - for (let i = 0; i < contentLines.length; i++) { - nonblank[i + 1] = nonblank[i] + Number(contentLines[i].trim().length > 0); - } - for (let i = anchor; i < contentLines.length && matches.length <= MAX_EDIT_MATCHES; i++) { - if (nonblank[i + 1] === nonblank[i]) continue; - const start = i - anchor; - const end = start + needleLines.length; - if (end > contentLines.length) break; - if (nonblank[end] - nonblank[start] !== 1) continue; - const indent = contentLines[i].length - contentLines[i].trimStart().length; - let matchesNeedle = true; - // Eligible windows have one nonblank line at a fixed offset. Each file line - // can belong to at most two of them, so these comparisons stay linear too. - for (let j = 0; j < needleLines.length; j++) { - if (contentLines[start + j].slice(indent) !== normalizedNeedle[j]) { - matchesNeedle = false; - break; - } - } - if (matchesNeedle) addMatch(start); - } - } - - if (matches.length === 1) { - return { status: 'matched', ...matches[0], strategy }; - } - if (matches.length > 1) { - return { status: 'ambiguous', strategy, count: matches.length, matches }; - } - return { status: 'none' }; -} - -function escapeRegExp(value: string): string { - return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); -} - -function findWhitespaceNormalizedMatch(content: string, needle: string): MatchStatus { - const tokens = needle.trim().split(/\s+/).filter(Boolean); - if (tokens.length < 2) { - return { status: 'none' }; - } - const pattern = tokens.map(escapeRegExp).join('\\s+'); - const regex = new RegExp(pattern, 'g'); - const matches: Array<{ index: number; length: number }> = []; - let match: RegExpExecArray | null; - while (matches.length <= MAX_EDIT_MATCHES && (match = regex.exec(content)) != null) { - matches.push({ index: match.index, length: match[0].length }); - if (match[0].length === 0) { - regex.lastIndex += 1; - } - } - if (matches.length === 1) { - return { status: 'matched', ...matches[0], strategy: 'whitespace-normalized' }; - } - if (matches.length > 1) { - return { - status: 'ambiguous', - strategy: 'whitespace-normalized', - count: matches.length, - matches, - }; - } - return { status: 'none' }; -} - -function findReplacementMatch(content: string, needle: string): MatchStatus { - const exact = findExactMatch(content, needle); - if (exact.status !== 'none') { - return exact; - } - const lineTrimmed = findLineWindowMatch(content, needle, 'line-trimmed'); - if (lineTrimmed.status !== 'none') { - return lineTrimmed; - } - const whitespaceNormalized = findWhitespaceNormalizedMatch(content, needle); - if (whitespaceNormalized.status !== 'none') { - return whitespaceNormalized; - } - return findLineWindowMatch(content, needle, 'indentation-flexible'); -} - -/** Keeps the earliest of any overlapping matches so replacements never collide. */ -function nonOverlapping(matches: readonly MatchedRange[]): MatchedRange[] { - const kept: MatchedRange[] = []; - let end = -1; - for (const match of [...matches].sort((a, b) => a.index - b.index)) { - if (match.index < end) continue; - kept.push(match); - end = match.index + match.length; - } - return kept; -} - -function describeMatchCount(count: number): string { - return count > MAX_EDIT_MATCHES ? `more than ${MAX_EDIT_MATCHES}` : String(count); -} - -/** - * The size `replace_all` would produce, computed before any replacement text is - * built so an oversized result is refused without allocating it. - */ -function projectedReplaceAllBytes( - content: string, - matches: readonly MatchedRange[], - text: string, -): number { - const replacementBytes = Buffer.byteLength(text, 'utf8'); - let bytes = Buffer.byteLength(content, 'utf8'); - for (const match of matches) { - bytes += - replacementBytes - - Buffer.byteLength(content.slice(match.index, match.index + match.length), 'utf8'); - } - return bytes; -} - -function replaceMatches(content: string, matches: readonly MatchedRange[], text: string): string { - let result = ''; - let cursor = 0; - for (const match of matches) { - result += content.slice(cursor, match.index) + text; - cursor = match.index + match.length; - } - return result + content.slice(cursor); -} - -function applyTextEdits( - content: string, - edits: TextEdit[], -): { content: string; strategies: string[] } { - let working = content; - const strategies: string[] = []; - - for (const edit of edits) { - const exactCount = edit.replace_all === true ? countExactMatches(working, edit.old_text) : 0; - if (exactCount > 0) { - const projectedBytes = - Buffer.byteLength(working, 'utf8') + - exactCount * - (Buffer.byteLength(edit.new_text, 'utf8') - Buffer.byteLength(edit.old_text, 'utf8')); - if (projectedBytes > MAX_AUTHORING_BYTES) { - throw new Error( - `replace_all would make the file larger than ${MAX_AUTHORING_BYTES} bytes; nothing was written.`, - ); - } - working = replaceAllExact(working, edit.old_text, edit.new_text); - strategies.push(exactCount > 1 ? `exact x${exactCount}` : 'exact'); - continue; - } - const match = findReplacementMatch(working, edit.old_text); - if (match.status === 'none') { - throw new Error('old_text did not match the file content.'); - } - if (match.status === 'ambiguous' && edit.replace_all !== true) { - throw new Error( - `old_text matched ${describeMatchCount(match.count)} locations with ${match.strategy}; make it unique or set replace_all before retrying.`, - ); - } - if (match.status === 'ambiguous') { - if (match.count > MAX_EDIT_MATCHES) { - throw new Error( - `replace_all with whitespace-tolerant matching is limited to ${MAX_EDIT_MATCHES} locations, and old_text matched more; copy the exact text or narrow old_text before retrying.`, - ); - } - const matches = nonOverlapping(match.matches); - if (projectedReplaceAllBytes(working, matches, edit.new_text) > MAX_AUTHORING_BYTES) { - throw new Error( - `replace_all would make the file larger than ${MAX_AUTHORING_BYTES} bytes; nothing was written.`, - ); - } - working = replaceMatches(working, matches, edit.new_text); - strategies.push(`${match.strategy} x${matches.length}`); - continue; - } - working = - working.slice(0, match.index) + edit.new_text + working.slice(match.index + match.length); - strategies.push(match.strategy); - } - - return { content: working, strategies }; -} - function formatRange(start: number, count: number): string { return count === 1 ? String(start) : `${start},${count}`; } @@ -3204,6 +2837,7 @@ async function writeSandboxTextForAuthoring({ created, sandboxContext, codeExecutionContext, + signal, }: { tc: ToolCallRequest; options: ToolExecuteOptions; @@ -3214,6 +2848,7 @@ async function writeSandboxTextForAuthoring({ created: boolean; sandboxContext?: SandboxSessionContext; codeExecutionContext?: CodeExecutionContext; + signal?: AbortSignal; }): AuthoringResult { if (!options.writeSandboxFile) { return errorResult( @@ -3236,6 +2871,7 @@ async function writeSandboxTextForAuthoring({ } const ctx = sandboxSessionContext(tc, sandboxContext); let writeResult: Awaited>>; + signal?.throwIfAborted(); try { writeResult = await options.writeSandboxFile({ file_path: filePath, @@ -3246,6 +2882,7 @@ async function writeSandboxTextForAuthoring({ ...(req ? { req } : {}), }); } catch (error) { + if (signal?.aborted === true && isAbortError(error)) throw error; const message = getThrownValueMessage(error); logger.warn('[file_authoring] Sandbox write failed', getSafeErrorMetadata(error)); return errorResult( @@ -3887,6 +3524,7 @@ async function writeSkillMd({ skill, skillName, content, + signal, }: { tc: ToolCallRequest; options: ToolExecuteOptions; @@ -3896,6 +3534,7 @@ async function writeSkillMd({ skill: AuthoringSkill | null; skillName: string; content: string; + signal?: AbortSignal; }): AuthoringResult { const normalized = normalizeSkillMdContent(content, skillName); if (normalized.status === 'error') { @@ -4015,6 +3654,7 @@ async function writeSkillMd({ ) { diff = ''; } + signal?.throwIfAborted(); const result = await options.updateSkill({ id: skill._id.toString(), expectedVersion: skill.version, @@ -4067,6 +3707,7 @@ async function writeBundledSkillFile({ oldContent, fileId, created, + signal, }: { tc: ToolCallRequest; options: ToolExecuteOptions; @@ -4078,6 +3719,7 @@ async function writeBundledSkillFile({ oldContent?: string; fileId?: string; created: boolean; + signal?: AbortSignal; }): AuthoringResult { const editDenied = await ensureCanEditSkill(tc, options, req, skill._id); if (editDenied) { @@ -4132,6 +3774,7 @@ async function writeBundledSkillFile({ } try { + signal?.throwIfAborted(); await options.saveSkillFileContent({ req, skillId: skill._id, @@ -4531,6 +4174,12 @@ async function handleSandboxCreateFileCall({ }); } +function hostEditFailure(tc: ToolCallRequest, error: unknown): ToolExecuteResult { + if (error instanceof HostEditError) return errorResult(tc, error.message); + logger.warn('[file_authoring] Host edit processing failed', getSafeErrorMetadata(error)); + return errorResult(tc, 'File edit processing failed. Nothing was written.'); +} + async function handleSandboxEditFileCall({ tc, options, @@ -4583,10 +4232,17 @@ async function handleSandboxEditFileCall({ let edited: { content: string; strategies: string[] }; try { - edited = applyTextEdits(current.content, edits); + edited = await applyHostTextEdits( + current.content, + edits, + req?.config?.endpoints?.agents?.hostFileEdits, + signal, + ); } catch (error) { - return errorResult(tc, error instanceof Error ? error.message : 'Failed to edit file'); + if (signal?.aborted) throw error; + return hostEditFailure(tc, error); } + signal?.throwIfAborted(); if (Buffer.byteLength(edited.content, 'utf8') > MAX_AUTHORING_BYTES) { return errorResult(tc, `edited content exceeds ${MAX_AUTHORING_BYTES} byte limit`); } @@ -4598,6 +4254,7 @@ async function handleSandboxEditFileCall({ filePath, content: edited.content, oldContent: current.content, + signal, created: false, sandboxContext, codeExecutionContext, @@ -4816,10 +4473,17 @@ async function handleEditFileCall( let edited: { content: string; strategies: string[] }; try { - edited = applyTextEdits(current.content, edits); + edited = await applyHostTextEdits( + current.content, + edits, + req?.config?.endpoints?.agents?.hostFileEdits, + signal, + ); } catch (error) { - return errorResult(tc, error instanceof Error ? error.message : 'Failed to edit file'); + if (signal?.aborted) throw error; + return hostEditFailure(tc, error); } + signal?.throwIfAborted(); if (Buffer.byteLength(edited.content, 'utf8') > MAX_AUTHORING_BYTES) { return errorResult(tc, `edited content exceeds ${MAX_AUTHORING_BYTES} byte limit`); } @@ -4833,6 +4497,7 @@ async function handleEditFileCall( skill, skillName: parsed.skillName, content: edited.content, + signal, }); if (result.status === 'success') { result.artifact = { @@ -4853,6 +4518,7 @@ async function handleEditFileCall( relativePath: parsed.relativePath, displayPath: parsed.displayPath, content: edited.content, + signal, oldContent: current.content, fileId: current.fileId, created: false, diff --git a/packages/api/src/agents/tools.ts b/packages/api/src/agents/tools.ts index 598cf5747a8..66f7319fcc6 100644 --- a/packages/api/src/agents/tools.ts +++ b/packages/api/src/agents/tools.ts @@ -1,3 +1,4 @@ +import { HOST_FILE_EDIT_HARD_MAX_COUNT } from 'librechat-data-provider'; import { Constants as AgentConstants, CODE_EXECUTION_TOOLS, @@ -827,6 +828,7 @@ const SKILL_EDIT_FILE_PARAMETERS: LCTool['parameters'] = Object.freeze({ }, edits: { type: 'array', + maxItems: HOST_FILE_EDIT_HARD_MAX_COUNT, description: 'Optional batch of replacements. Each old_text must match exactly once unless its replace_all is true.', items: { @@ -864,6 +866,7 @@ const CODE_EDIT_FILE_PARAMETERS: LCTool['parameters'] = Object.freeze({ }, edits: { type: 'array', + maxItems: HOST_FILE_EDIT_HARD_MAX_COUNT, description: 'Optional batch of replacements. Each old_text must match exactly once unless its replace_all is true.', items: { diff --git a/packages/api/tsdown.config.mjs b/packages/api/tsdown.config.mjs index b815bbd6599..7865777e578 100644 --- a/packages/api/tsdown.config.mjs +++ b/packages/api/tsdown.config.mjs @@ -6,7 +6,12 @@ export default defineConfig({ // `src/telemetry/index.ts` barrel: oxc emits declarations flat into outDir keyed // by source basename, so two `index.ts` entries would collide (index.d.cts + // index2.d.cts). Distinct basenames yield stable `index.*` / `telemetry.*` output. - entry: ['src/index.ts', 'src/telemetry.ts', 'src/credentials.ts'], + entry: [ + 'src/index.ts', + 'src/telemetry.ts', + 'src/credentials.ts', + 'src/agents/files/edit-worker.ts', + ], format: ['cjs'], platform: 'node', dts: { oxc: true }, diff --git a/packages/data-provider/src/config.spec.ts b/packages/data-provider/src/config.spec.ts index 672e5a4549f..046f3bd5ff4 100644 --- a/packages/data-provider/src/config.spec.ts +++ b/packages/data-provider/src/config.spec.ts @@ -42,6 +42,32 @@ const endpointsConfig: TEndpointsConfig = { Gemini: { type: EModelEndpoint.custom, userProvide: false, order: 9999 }, }; +describe('host-side file edit limits', () => { + it('keeps configuration optional and resolves safe defaults when configured', () => { + expect(agentsEndpointSchema.parse({}).hostFileEdits).toBeUndefined(); + expect(agentsEndpointSchema.parse({ hostFileEdits: {} }).hostFileEdits).toEqual({ + maxEdits: 100, + maxWorkBytes: 67108864, + maxOccurrences: 100000, + timeoutMs: 2000, + maxConcurrent: 2, + }); + }); + it.each([ + { maxEdits: 101 }, + { maxEdits: 0 }, + { maxWorkBytes: 0 }, + { maxOccurrences: 1000001 }, + { timeoutMs: 0 }, + { timeoutMs: 10001 }, + { maxConcurrent: 9 }, + { maxConcurrent: 1.5 }, + { ignored: true }, + ])('rejects unsafe limits %p', (hostFileEdits) => { + expect(agentsEndpointSchema.safeParse({ hostFileEdits }).success).toBe(false); + }); +}); + describe('authenticated 2FA management rate limits', () => { it('accepts an account budget and defaults an empty configuration to seven requests', () => { for (const [input, expected] of [ diff --git a/packages/data-provider/src/config.ts b/packages/data-provider/src/config.ts index 23e1977960a..0125b209e0c 100644 --- a/packages/data-provider/src/config.ts +++ b/packages/data-provider/src/config.ts @@ -1392,6 +1392,31 @@ export const DEFAULT_MAX_PROVIDER_ERROR_CHARS = 2000; export const DEFAULT_AGENT_MODEL_RESPONSE_BODY_TIMEOUT_MS = 900_000; export const DEFAULT_AGENT_MODEL_RESPONSE_HEADERS_TIMEOUT_MS = 300_000; +export const HOST_FILE_EDIT_HARD_MAX_COUNT = 100; + +/** Host-side skill/sandbox edit budgets. Attached workers retain their own limits. */ +export const hostFileEditLimitsSchema = z + .object({ + maxEdits: z + .number() + .int() + .min(1) + .max(HOST_FILE_EDIT_HARD_MAX_COUNT) + .default(HOST_FILE_EDIT_HARD_MAX_COUNT), + maxWorkBytes: z + .number() + .int() + .min(1024) + .max(256 * 1024 * 1024) + .default(64 * 1024 * 1024), + maxOccurrences: z.number().int().min(1).max(1_000_000).default(100_000), + timeoutMs: z.number().int().min(100).max(10_000).default(2000), + maxConcurrent: z.number().int().min(1).max(8).default(2), + }) + .strict(); + +export type HostFileEditLimits = z.infer; + /** Server-side resource and recovery policy for ephemeral child activity. */ export const subagentActivityConfigSchema = z.object({ replayTtlMs: z.number().int().min(1_000).max(86_400_000).default(300_000), @@ -1410,6 +1435,7 @@ export const agentsEndpointSchema = baseEndpointSchema .merge( z.object({ /* agents specific */ + hostFileEdits: hostFileEditLimitsSchema.optional(), /** Maximum provider error characters retained in unprotected terminal failures. */ maxProviderErrorChars: z .number()