From 659ac8e9ff0354011afef18e1067f918b0e2e661 Mon Sep 17 00:00:00 2001 From: Rakshith Ramprakash Date: Fri, 25 Sep 2026 17:59:06 +1000 Subject: [PATCH 1/3] feat(agent): add onTermsRequired to firecrawl_agent The agent service now only calls Alexandria providers whose data terms the team has accepted, and reports the rest. Let MCP callers choose what happens ("skip", "ask" or "fail"), forwarded as exchange.onTermsRequired. Describe exchange.skippedProviders and exchange.requiresAction in the tool description, and tell calling agents they must get the user's explicit consent before calling terms/accept. Keep exchange, pendingApproval and message in firecrawl_agent_status structured content so Codex-style clients that read structuredContent do not lose them. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 4 ++ README.md | 6 ++ src/index.ts | 10 ++++ src/tool-output.ts | 7 +++ tests/mcp-smoke.test.mjs | 121 +++++++++++++++++++++++++++++++++++++++ 5 files changed, 148 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 03ecac94..4c2c383b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- `firecrawl_agent` accepts `onTermsRequired` (`"skip"`, `"ask"` or `"fail"`), forwarded as `exchange.onTermsRequired`. The agent only calls Alexandria providers whose data terms the team has accepted; `firecrawl_agent_status` now keeps `exchange` (including `skippedProviders`, `requiresAction` and `error`), `pendingApproval` and `message` in its structured content. There is no auto-accept: `terms/accept` still needs the user's explicit consent. + ### Changed - The search surface (`/v2/mcp-search`) now exposes `firecrawl_find_tools` and `firecrawl_scrape` alongside its six search tools, so agents can execute the Alexandria providers that `firecrawl_search` already returns. Both carry surface-scoped descriptions that name only tools registered on that surface, and Alexandria results there omit the `firecrawl_feedback` pointer. See docs/search-profile.md. diff --git a/README.md b/README.md index 98d268c4..c6c375a6 100644 --- a/README.md +++ b/README.md @@ -735,6 +735,12 @@ The agent performs web searches, follows links, reads pages, and gathers data au - `prompt`: Natural language description of the data you want (required, max 10,000 characters) - `urls`: Optional array of URLs to focus the agent on specific pages - `schema`: Optional JSON schema for structured output +- `onTermsRequired`: Optional. What to do when an Alexandria provider the agent would use needs data terms your team has not accepted. Gated providers are never called in any mode. + - `"skip"` (default): answer with accepted providers only. `exchange.skippedProviders` on the status result lists the gated providers that would have helped. + - `"ask"`: the same, plus `exchange.requiresAction` with the exact `terms/show` and `terms/accept` calls for each provider. + - `"fail"`: stop making calls once a gated provider is needed, and set `exchange.error` to `THIRD_PARTY_DATA_TERMS_REQUIRED`. + +**Provider terms:** there is no auto-accept mode. Only call `terms/accept` (through `firecrawl_scrape` with `alexandria`) after the user has explicitly agreed to that provider's terms; a data request is not consent. Once accepted, start `firecrawl_agent` again and the provider becomes available. **Prompt Example:** diff --git a/src/index.ts b/src/index.ts index cb03b1c4..7a975a96 100644 --- a/src/index.ts +++ b/src/index.ts @@ -3527,12 +3527,20 @@ server.addTool({ Run web research that returns structured data when the URLs are not known or the answer spans several sites. Describe the fields you need in \`prompt\`, optionally pass a JSON \`schema\` and seed \`urls\`, and the research agent searches, navigates, reads pages, and returns JSON assembled across sources. Use it to research an entity plus its fields (founders, pricing, contact details), to build lists and datasets (companies, people, products, jobs, papers), and for pages that need navigation or interaction to reach the data. This call returns only a job ID, not the research result. Read the job with \`firecrawl_agent_status\` until it reaches \`completed\` or \`failed\`; a typical research run takes one to three minutes. For one known URL use \`firecrawl_scrape\` (with formats: ["json"] for structured output); for a plain lookup that a results page answers, use \`firecrawl_search\`. + +The agent only calls Alexandria providers whose data terms the team has accepted. The status result's \`exchange.skippedProviders\` lists gated providers that would have helped, and with \`onTermsRequired\` "ask" or "fail", \`exchange.requiresAction\` holds the exact terms/show and terms/accept calls. Never call terms/accept without the user's explicit consent to that provider's terms; a data request is not consent. After they agree, run the accept call through \`firecrawl_scrape\` and start \`firecrawl_agent\` again. `, outputSchema: agentOutputSchema, parameters: z.object({ prompt: z.string().min(1).max(10000), urls: z.array(z.string().url()).optional(), schema: z.record(z.string(), z.any()).optional(), + onTermsRequired: z + .enum(['skip', 'ask', 'fail']) + .optional() + .describe( + 'What to do when a provider the agent would use needs data terms the team has not accepted. Gated providers are never called. "skip" (default): answer with accepted providers and list the rest in exchange.skippedProviders. "ask": the same, plus exchange.requiresAction with the terms/show and terms/accept calls. "fail": stop making calls once a gated provider is needed and set exchange.error (THIRD_PARTY_DATA_TERMS_REQUIRED). There is no auto-accept.' + ), }), execute: async ( args: unknown, @@ -3544,10 +3552,12 @@ This call returns only a job ID, not the research result. Read the job with \`fi prompt: (a.prompt as string).substring(0, 100), urlCount: Array.isArray(a.urls) ? a.urls.length : 0, }); + const onTermsRequired = a.onTermsRequired as 'skip' | 'ask' | 'fail' | undefined; const agentBody = removeEmptyTopLevel({ prompt: a.prompt as string, urls: a.urls as string[] | undefined, schema: (a.schema as Record) || undefined, + exchange: onTermsRequired ? { onTermsRequired } : undefined, }); const res = await (client as any).startAgent({ ...agentBody, diff --git a/src/tool-output.ts b/src/tool-output.ts index ce99a886..12c14d39 100644 --- a/src/tool-output.ts +++ b/src/tool-output.ts @@ -234,6 +234,13 @@ export const agentStatusOutputSchema = z mode: str('Agent mode the job ran in.'), threadId: str('Research thread this job belongs to.'), threadTurn: num('Turn number of this job within its thread.'), + message: unknown('The agent\'s reply, including what it could not answer.'), + exchange: unknown( + 'What the job did with Alexandria providers: onTermsRequired, paidCalls, creditsUsed, skippedProviders (gated providers that would have helped), requiresAction (terms/show and terms/accept calls; call accept only with the user\'s explicit consent) and error (THIRD_PARTY_DATA_TERMS_REQUIRED in "fail" mode).' + ), + pendingApproval: unknown( + 'Set when the job ended waiting on the caller; kind "terms" lists providers whose data terms need accepting.' + ), }) .describe('Progress or final result of a research agent job.'); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 9f85a619..e644c0a3 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -268,6 +268,65 @@ async function startFakeFirecrawlApi() { return; } + if (req.method === 'GET' && req.url === '/v2/agent/00000000-0000-4000-8000-000000000032') { + res.writeHead(200, { 'content-type': 'application/json' }); + res.end( + JSON.stringify({ + data: null, + expiresAt: '2026-10-01T00:00:00.000Z', + message: 'Apollo could add verified work emails.', + mode: 'chat', + model: 'spark-2', + status: 'completed', + success: true, + exchange: { + enabled: true, + onTermsRequired: 'ask', + paidCalls: 0, + creditsUsed: null, + skippedProviders: [ + { + provider: 'apollo', + name: 'Apollo', + capability: 'people/search', + reason: 'terms_required', + version: 'F-1.0.0', + termsUrl: 'https://www.firecrawl.dev/app/alexandria/apollo', + }, + ], + requiresAction: { + type: 'accept_terms', + approvalId: '00000000-0000-4000-8000-000000000033', + providers: [ + { + id: 'apollo', + provider: 'apollo', + name: 'Apollo', + version: 'F-1.0.0', + url: 'https://www.firecrawl.dev/app/alexandria/apollo', + show: { provider: 'firecrawl', capability: 'terms/show', options: { provider: 'apollo' } }, + accept: { + provider: 'firecrawl', + capability: 'terms/accept', + options: { provider: 'apollo', version: 'F-1.0.0', digest: null, confirmed: true }, + }, + }, + ], + }, + }, + pendingApproval: { + id: '00000000-0000-4000-8000-000000000033', + kind: 'terms', + reason: 'Apollo could add verified work emails.', + calls: [], + terms: [{ id: 'apollo', provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', url: 'https://www.firecrawl.dev/app/alexandria/apollo' }], + resolution: null, + }, + }) + ); + return; + } + if (req.method === 'POST' && req.url === '/v2/map') { res.writeHead(200, { 'content-type': 'application/json' }); res.end( @@ -3817,3 +3876,65 @@ test('every listed tool declares an output schema and returns structured content } assert.equal('id' in feedback.structuredContent, false); }); + +test('firecrawl_agent forwards onTermsRequired and status keeps the terms-required fields', async (t) => { + const fakeApi = await startFakeFirecrawlApi(); + t.after(() => fakeApi.close()); + + const child = spawnServer({ + FIRECRAWL_API_KEY: 'fc-test', + FIRECRAWL_API_URL: fakeApi.url, + }); + t.after(() => stopChild(child)); + + const client = new StdioMcpClient(child); + await client.request('initialize', { + capabilities: {}, + clientInfo: { name: 'firecrawl-mcp-terms-required', version: '0.0.0' }, + protocolVersion: '2025-06-18', + }); + client.notify('notifications/initialized'); + + const { tools } = await client.request('tools/list'); + const agentTool = tools.find((tool) => tool.name === 'firecrawl_agent'); + assert.deepEqual(agentTool.inputSchema.properties.onTermsRequired.enum, ['skip', 'ask', 'fail']); + assert.match(agentTool.description, /exchange\.skippedProviders/); + assert.match(agentTool.description, /exchange\.requiresAction/); + assert.match(agentTool.description, /Never call terms\/accept without the user's explicit consent/); + + const asked = await client.request('tools/call', { + arguments: { prompt: 'Find the key business contact at exa.ai', onTermsRequired: 'ask' }, + name: 'firecrawl_agent', + }); + assert.notEqual(asked.isError, true); + const plain = await client.request('tools/call', { + arguments: { prompt: 'Find the example domain owner' }, + name: 'firecrawl_agent', + }); + assert.notEqual(plain.isError, true); + const bodies = fakeApi.requests + .filter((request) => request.method === 'POST' && request.url === '/v2/agent') + .map((request) => request.body); + assert.deepEqual(bodies[0].exchange, { onTermsRequired: 'ask' }); + assert.equal('exchange' in bodies[1], false); + + // There is no auto-accept mode: any other value fails parameter validation. + await assert.rejects( + client.request('tools/call', { + arguments: { prompt: 'Find the key business contact at exa.ai', onTermsRequired: 'accept' }, + name: 'firecrawl_agent', + }), + /onTermsRequired/ + ); + + const status = await client.request('tools/call', { + arguments: { id: '00000000-0000-4000-8000-000000000032' }, + name: 'firecrawl_agent_status', + }); + assert.notEqual(status.isError, true); + const structured = status.structuredContent; + assert.equal(structured.exchange.skippedProviders[0].reason, 'terms_required'); + assert.equal(structured.exchange.requiresAction.providers[0].accept.capability, 'terms/accept'); + assert.equal(structured.pendingApproval.kind, 'terms'); + assert.equal(structured.message, 'Apollo could add verified work emails.'); +}); From 4a9393e0a83ead12a3f416b26af8d26b98d1033c Mon Sep 17 00:00:00 2001 From: Rakshith Ramprakash Date: Sat, 26 Sep 2026 01:47:05 +1000 Subject: [PATCH 2/3] feat(agent): cut fail mode from onTermsRequired; digest is string | null Matches the extract-v3#182 scope cut: onTermsRequired is skip or ask, and exchange.error is gone. Describe each requiresAction provider digest as string | null and always present. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- README.md | 3 +-- src/index.ts | 8 ++++---- src/tool-output.ts | 2 +- tests/mcp-smoke.test.mjs | 7 ++++--- 5 files changed, 11 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4c2c383b..88c15d09 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- `firecrawl_agent` accepts `onTermsRequired` (`"skip"`, `"ask"` or `"fail"`), forwarded as `exchange.onTermsRequired`. The agent only calls Alexandria providers whose data terms the team has accepted; `firecrawl_agent_status` now keeps `exchange` (including `skippedProviders`, `requiresAction` and `error`), `pendingApproval` and `message` in its structured content. There is no auto-accept: `terms/accept` still needs the user's explicit consent. +- `firecrawl_agent` accepts `onTermsRequired` (`"skip"` or `"ask"`), forwarded as `exchange.onTermsRequired`. The agent only calls Alexandria providers whose data terms the team has accepted; `firecrawl_agent_status` now keeps `exchange` (including `skippedProviders` and `requiresAction`, whose provider `digest` is `string | null` and always present), `pendingApproval` and `message` in its structured content. There is no auto-accept: `terms/accept` still needs the user's explicit consent. ### Changed diff --git a/README.md b/README.md index c6c375a6..c623dd41 100644 --- a/README.md +++ b/README.md @@ -737,8 +737,7 @@ The agent performs web searches, follows links, reads pages, and gathers data au - `schema`: Optional JSON schema for structured output - `onTermsRequired`: Optional. What to do when an Alexandria provider the agent would use needs data terms your team has not accepted. Gated providers are never called in any mode. - `"skip"` (default): answer with accepted providers only. `exchange.skippedProviders` on the status result lists the gated providers that would have helped. - - `"ask"`: the same, plus `exchange.requiresAction` with the exact `terms/show` and `terms/accept` calls for each provider. - - `"fail"`: stop making calls once a gated provider is needed, and set `exchange.error` to `THIRD_PARTY_DATA_TERMS_REQUIRED`. + - `"ask"`: the same, plus `exchange.requiresAction` with the exact `terms/show` and `terms/accept` calls for each provider. Each provider's `digest` is always present and is `string | null`; when it is `null`, `terms/show` returns the current digest to send. **Provider terms:** there is no auto-accept mode. Only call `terms/accept` (through `firecrawl_scrape` with `alexandria`) after the user has explicitly agreed to that provider's terms; a data request is not consent. Once accepted, start `firecrawl_agent` again and the provider becomes available. diff --git a/src/index.ts b/src/index.ts index 7a975a96..beb4ac97 100644 --- a/src/index.ts +++ b/src/index.ts @@ -3528,7 +3528,7 @@ Run web research that returns structured data when the URLs are not known or the This call returns only a job ID, not the research result. Read the job with \`firecrawl_agent_status\` until it reaches \`completed\` or \`failed\`; a typical research run takes one to three minutes. For one known URL use \`firecrawl_scrape\` (with formats: ["json"] for structured output); for a plain lookup that a results page answers, use \`firecrawl_search\`. -The agent only calls Alexandria providers whose data terms the team has accepted. The status result's \`exchange.skippedProviders\` lists gated providers that would have helped, and with \`onTermsRequired\` "ask" or "fail", \`exchange.requiresAction\` holds the exact terms/show and terms/accept calls. Never call terms/accept without the user's explicit consent to that provider's terms; a data request is not consent. After they agree, run the accept call through \`firecrawl_scrape\` and start \`firecrawl_agent\` again. +The agent only calls Alexandria providers whose data terms the team has accepted. The status result's \`exchange.skippedProviders\` lists gated providers that would have helped, and with \`onTermsRequired\` "ask", \`exchange.requiresAction\` holds the exact terms/show and terms/accept calls. Never call terms/accept without the user's explicit consent to that provider's terms; a data request is not consent. After they agree, run the accept call through \`firecrawl_scrape\` and start \`firecrawl_agent\` again. `, outputSchema: agentOutputSchema, parameters: z.object({ @@ -3536,10 +3536,10 @@ The agent only calls Alexandria providers whose data terms the team has accepted urls: z.array(z.string().url()).optional(), schema: z.record(z.string(), z.any()).optional(), onTermsRequired: z - .enum(['skip', 'ask', 'fail']) + .enum(['skip', 'ask']) .optional() .describe( - 'What to do when a provider the agent would use needs data terms the team has not accepted. Gated providers are never called. "skip" (default): answer with accepted providers and list the rest in exchange.skippedProviders. "ask": the same, plus exchange.requiresAction with the terms/show and terms/accept calls. "fail": stop making calls once a gated provider is needed and set exchange.error (THIRD_PARTY_DATA_TERMS_REQUIRED). There is no auto-accept.' + 'What to do when a provider the agent would use needs data terms the team has not accepted. Gated providers are never called. "skip" (default): answer with accepted providers and list the rest in exchange.skippedProviders. "ask": the same, plus exchange.requiresAction with the terms/show and terms/accept calls. Each provider digest is string | null and always present; when null, terms/show returns it. There is no auto-accept.' ), }), execute: async ( @@ -3552,7 +3552,7 @@ The agent only calls Alexandria providers whose data terms the team has accepted prompt: (a.prompt as string).substring(0, 100), urlCount: Array.isArray(a.urls) ? a.urls.length : 0, }); - const onTermsRequired = a.onTermsRequired as 'skip' | 'ask' | 'fail' | undefined; + const onTermsRequired = a.onTermsRequired as 'skip' | 'ask' | undefined; const agentBody = removeEmptyTopLevel({ prompt: a.prompt as string, urls: a.urls as string[] | undefined, diff --git a/src/tool-output.ts b/src/tool-output.ts index 12c14d39..663387b7 100644 --- a/src/tool-output.ts +++ b/src/tool-output.ts @@ -236,7 +236,7 @@ export const agentStatusOutputSchema = z threadTurn: num('Turn number of this job within its thread.'), message: unknown('The agent\'s reply, including what it could not answer.'), exchange: unknown( - 'What the job did with Alexandria providers: onTermsRequired, paidCalls, creditsUsed, skippedProviders (gated providers that would have helped), requiresAction (terms/show and terms/accept calls; call accept only with the user\'s explicit consent) and error (THIRD_PARTY_DATA_TERMS_REQUIRED in "fail" mode).' + 'What the job did with Alexandria providers: onTermsRequired, paidCalls, creditsUsed, skippedProviders (gated providers that would have helped), and requiresAction (terms/show and terms/accept calls, each provider digest string | null and always present; call accept only with the user\'s explicit consent).' ), pendingApproval: unknown( 'Set when the job ended waiting on the caller; kind "terms" lists providers whose data terms need accepting.' diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index e644c0a3..f99b9958 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -303,6 +303,7 @@ async function startFakeFirecrawlApi() { provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', + digest: null, url: 'https://www.firecrawl.dev/app/alexandria/apollo', show: { provider: 'firecrawl', capability: 'terms/show', options: { provider: 'apollo' } }, accept: { @@ -319,7 +320,7 @@ async function startFakeFirecrawlApi() { kind: 'terms', reason: 'Apollo could add verified work emails.', calls: [], - terms: [{ id: 'apollo', provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', url: 'https://www.firecrawl.dev/app/alexandria/apollo' }], + terms: [{ id: 'apollo', provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', digest: null, url: 'https://www.firecrawl.dev/app/alexandria/apollo' }], resolution: null, }, }) @@ -3897,7 +3898,7 @@ test('firecrawl_agent forwards onTermsRequired and status keeps the terms-requir const { tools } = await client.request('tools/list'); const agentTool = tools.find((tool) => tool.name === 'firecrawl_agent'); - assert.deepEqual(agentTool.inputSchema.properties.onTermsRequired.enum, ['skip', 'ask', 'fail']); + assert.deepEqual(agentTool.inputSchema.properties.onTermsRequired.enum, ['skip', 'ask']); assert.match(agentTool.description, /exchange\.skippedProviders/); assert.match(agentTool.description, /exchange\.requiresAction/); assert.match(agentTool.description, /Never call terms\/accept without the user's explicit consent/); @@ -3921,7 +3922,7 @@ test('firecrawl_agent forwards onTermsRequired and status keeps the terms-requir // There is no auto-accept mode: any other value fails parameter validation. await assert.rejects( client.request('tools/call', { - arguments: { prompt: 'Find the key business contact at exa.ai', onTermsRequired: 'accept' }, + arguments: { prompt: 'Find the key business contact at exa.ai', onTermsRequired: 'fail' }, name: 'firecrawl_agent', }), /onTermsRequired/ From b1bf4c3d46fe29efb8258c4410135ae69f847b20 Mon Sep 17 00:00:00 2001 From: Rakshith Ramprakash Date: Sat, 26 Sep 2026 01:51:34 +1000 Subject: [PATCH 3/3] test(agent): drop the provider id the final backend contract no longer sends Co-Authored-By: Claude Opus 5.5 --- tests/mcp-smoke.test.mjs | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index f99b9958..31ad006d 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -299,7 +299,6 @@ async function startFakeFirecrawlApi() { approvalId: '00000000-0000-4000-8000-000000000033', providers: [ { - id: 'apollo', provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', @@ -320,7 +319,7 @@ async function startFakeFirecrawlApi() { kind: 'terms', reason: 'Apollo could add verified work emails.', calls: [], - terms: [{ id: 'apollo', provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', digest: null, url: 'https://www.firecrawl.dev/app/alexandria/apollo' }], + terms: [{ provider: 'apollo', name: 'Apollo', version: 'F-1.0.0', digest: null, url: 'https://www.firecrawl.dev/app/alexandria/apollo' }], resolution: null, }, })