From dce087e9f18a6b25fc4aa939801b216a7ea7fa54 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 15:13:20 -0500 Subject: [PATCH 01/25] feat(mcp): expose keyless job feedback --- src/index.ts | 104 +++++++++++++----- tests/mcp-smoke.test.mjs | 224 ++++++++++++++++++++++++++++++++++++++- 2 files changed, 297 insertions(+), 31 deletions(-) diff --git a/src/index.ts b/src/index.ts index 3aba8f15..1362e86f 100644 --- a/src/index.ts +++ b/src/index.ts @@ -954,7 +954,7 @@ const openAiAppsChallengeToken = normalizeHeader( ); const FULL_PROFILE_INSTRUCTIONS = `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question — code behaviour, a library or framework, an API contract, an error message, or a known bug — firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of repositories, GitHub issues, merged pull requests, READMEs, and curated documentation sites. Provide only the required inputs and account for stated network or external side effects.`; -const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse processes supported local files through its two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, and firecrawl_research_* for paper-index and repository research.`; +const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_feedback accepts optional evidence about those jobs without consuming operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse processes supported local files through its two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, and firecrawl_research_* for paper-index and repository research.`; // The search surface exposes web/developer/research search only. Its instructions // and tool copy describe just those tools and stay neutral about how a client @@ -1071,6 +1071,7 @@ const KEYLESS_TOOL_NAMES = new Set([ 'firecrawl_scrape', 'firecrawl_search', 'firecrawl_parse', + 'firecrawl_feedback', ]); function isHostedKeylessSession(session?: SessionData): boolean { @@ -1082,7 +1083,7 @@ function isHostedKeylessSession(session?: SessionData): boolean { } // A stdio client without a cloud credential can use only the keyless tools. -// Do this at registration time so unsupported feedback tools are not advertised. +// Apply the keyless tool set at registration time for local sessions. function isLocalKeylessStartup(): boolean { return ( process.env.CLOUD_SERVICE !== 'true' && @@ -2105,6 +2106,8 @@ server.addTool({ destructiveHint: false, // Does not modify, delete, or write to external websites. }, description: ` +Optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. + Retrieve and extract content from one supplied URL through Firecrawl. Use this when the request identifies a page and needs its content or defined fields. It can return markdown, HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema; JSON is useful when the requested result has defined fields, while markdown preserves readable page content. This tool operates on a known page. For a set of pages use \`firecrawl_crawl\`, and to discover page URLs use \`firecrawl_map\` or \`firecrawl_search\`. Options include JavaScript render delay, cache age, main-content filtering, PII redaction, and lockdown cache-only retrieval. Browser actions may change the live page when interactive actions are enabled. @@ -2195,13 +2198,15 @@ server.addTool({ destructiveHint: false, // Query-only; no destructive side effects on external entities. }, description: ` +Optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. + Search web, news, or image sources and return ranked results. Operators include quoted phrases, \`-term\`, \`site:host\`, \`inurl:term\`, \`intitle:term\`, and \`related:host\`; the set is non-exhaustive. \`includeDomains\` and \`excludeDomains\` are mutually exclusive hostname filters; categories limit results to GitHub, research, PDF, or developer sources. For a programming question, add \`categories: ["developer"]\`. It searches an index of repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites, and returns the hits in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts these web results to research-affiliated websites and returns page snippets. The \`firecrawl_research_*\` tools are a separate surface that searches paper abstracts and full text across biomedical (PubMed, bioRxiv, medRxiv) and arXiv literature. -Each web result is a title, URL, and description, not the page. Add \`scrapeOptions\` to attach page content in the same call; those fetches ignore \`maxAge\`, so use \`firecrawl_scrape\` when you need a live fetch. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. +Each web result is a title, URL, and description, not the page. Add \`scrapeOptions\` to attach page content in the same call; those fetches ignore \`maxAge\`, so use \`firecrawl_scrape\` when you need a live fetch. Returns source-type result groups and usage metadata. Responses can include an \`id\` for optional feedback through \`firecrawl_feedback\`. `, parameters: z .object({ @@ -2241,11 +2246,7 @@ Each web result is a title, URL, and description, not the page. Add \`scrapeOpti }; if (isKeylessMode(session)) { const json = await keylessPost('/v2/search', searchBody, session); - // Search feedback requires an authenticated account. Do not expose its - // identifier to keyless clients, where it would invite an unusable call. - const keylessResponse = { ...(json ?? {}) }; - delete keylessResponse.id; - return asText(keylessResponse); + return asText(json); } // Call /v2/search through the SDK's HTTP layer (auth + retries) instead // of `client.search()` so we preserve the full response envelope. The @@ -2332,6 +2333,24 @@ function isKeylessMode(session?: SessionData): boolean { return !process.env.FIRECRAWL_API_URL; } +function filterFeedbackInvitation(json: any): any { + if (!ENDPOINT_FEEDBACK_DISABLED || !json || typeof json !== 'object') + return json; + const omitInvitation = (metadata: any) => { + if (!metadata || typeof metadata !== 'object') return metadata; + const rest = { ...metadata }; + delete rest.feedback; + return rest; + }; + return { + ...json, + ...(json.metadata ? { metadata: omitInvitation(json.metadata) } : {}), + ...(json.data?.metadata + ? { data: { ...json.data, metadata: omitInvitation(json.data.metadata) } } + : {}), + }; +} + async function keylessPost( path: string, body: Record, @@ -2368,7 +2387,9 @@ async function keylessPost( headers, body: JSON.stringify(body), }); - const json: any = await response.json().catch(() => ({})); + const json: any = filterFeedbackInvitation( + await response.json().catch(() => ({})) + ); if (!response.ok) { if (isKeylessMode(session) && response.status === 429) { // The API normally supplies requests|credits. Preserve a structured, @@ -2385,6 +2406,9 @@ async function keylessPost( }); throw new UserError(String(payload.message), payload); } + if (json?.metadata?.jobId) { + throw new UserError(asText(json), json); + } throw new Error( json?.error || `Firecrawl request failed (HTTP ${response.status})` ); @@ -2657,7 +2681,7 @@ if (ENDPOINT_FEEDBACK_DISABLED) { ); } -if (!ENDPOINT_FEEDBACK_DISABLED && !isLocalKeylessStartup()) { +if (!ENDPOINT_FEEDBACK_DISABLED) { server.addTool({ name: 'firecrawl_feedback', annotations: { @@ -2667,14 +2691,23 @@ if (!ENDPOINT_FEEDBACK_DISABLED && !isLocalKeylessStartup()) { destructiveHint: false, // Additive only; submits ratings and notes, does not delete jobs or external content. }, description: ` -Submit concise quality feedback for a completed search, scrape, parse, or map job. Provide the endpoint, job ID, rating, and relevant issue codes or small contextual fields; omit large page contents and raw outputs. +Submit optional quality feedback for a search, scrape, parse, or map job. Authenticated callers can provide issue codes or small contextual fields with the endpoint, job ID, and rating. Omit large page contents and raw outputs. + +Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. Search kinds are useful or irrelevant with source (web, images, news) and one-based position within that delivered group, or missing with topic and optional knownSources URLs. Scrape kinds are correct, missing, incorrect, or failure, with optional location and already-observed retryOutcome. Parse kinds are correct, text, table, layout, or completeness, with optional location. -Returns submission status, feedback ID, and accounting fields. +Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. A failed scrape can be reported with its jobId and observed failure. One accepted submission per keyless identity, category, and UTC day, shared across clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, parameters: z.object({ endpoint: z.enum(['search', 'scrape', 'parse', 'map']), jobId: z.string().uuid('jobId must be the UUID returned by Firecrawl'), rating: z.enum(['good', 'bad', 'partial']), + task: z.string().min(10).max(2000).optional(), + assessment: z.string().min(10).max(2000).optional(), + observations: z + .array(z.record(z.string(), z.unknown())) + .min(1) + .max(20) + .optional(), issues: z.array(feedbackIssueSchema).max(20).optional(), tags: z.array(feedbackIssueSchema).max(20).optional(), note: z.string().max(4000).optional(), @@ -2690,6 +2723,9 @@ Returns submission status, feedback ID, and accounting fields. endpoint, jobId, rating, + task, + assessment, + observations, issues, tags, note, @@ -2703,6 +2739,9 @@ Returns submission status, feedback ID, and accounting fields. endpoint: 'search' | 'scrape' | 'parse' | 'map'; jobId: string; rating: 'good' | 'bad' | 'partial'; + task?: string; + assessment?: string; + observations?: Record[]; issues?: string[]; tags?: string[]; note?: string; @@ -2722,14 +2761,26 @@ Returns submission status, feedback ID, and accounting fields. const credential = credentialForOutboundRequest(session); if (credential) { headers['Authorization'] = `Bearer ${credential}`; - } else if (process.env.CLOUD_SERVICE === 'true') { - throw new Error('Unauthorized: missing API key for feedback.'); + } else if (isHostedKeylessSession(session)) { + if (!session?.keylessClientIp || !process.env.KEYLESS_PROXY_SECRET) { + return asText({ + success: false, + error: 'Feedback requires a trusted client identity.', + retryable: false, + }); + } + headers['x-firecrawl-keyless-ip'] = session.keylessClientIp; + headers['x-firecrawl-keyless-secret'] = + process.env.KEYLESS_PROXY_SECRET; } const body = removeEmptyTopLevel({ endpoint, jobId, rating, + task, + assessment, + observations, issues, tags, note, @@ -3126,6 +3177,8 @@ server.addTool({ destructiveHint: false, // Read-only parsing; no deletion or writes to the source file. }, description: ` +Optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. + Parse one supported document into markdown, HTML, links, summary, targeted answers, or JSON matching a schema. Supported inputs include common HTML, PDF, Word, RTF, OpenDocument, and spreadsheet files; PDF parsing can be bounded with \`pdfOptions.maxPages\`. Local MCP reads \`filePath\` from the server filesystem. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. @@ -3138,12 +3191,7 @@ Set \`redactPII\` to request redaction of personally identifiable information in return executeHostedParse(args as ParseToolArgs, session, log); } - const apiUrl = process.env.FIRECRAWL_API_URL; - if (!apiUrl) { - throw new Error( - 'firecrawl_parse requires FIRECRAWL_API_URL to be set to a self-hosted Firecrawl API instance.' - ); - } + const apiUrl = resolveApiBaseUrl(); const { filePath, @@ -3193,17 +3241,21 @@ Set \`redactPII\` to request redaction of personally identifiable information in }); const responseText = await response.text(); + let result: any; + try { + result = filterFeedbackInvitation(JSON.parse(responseText)); + } catch { + result = undefined; + } if (!response.ok) { + if (result?.metadata?.jobId) { + throw new UserError(asText(result), result); + } throw new Error( `Parse request failed with status ${response.status}: ${responseText}` ); } - - try { - return asText(JSON.parse(responseText)); - } catch { - return responseText; - } + return result === undefined ? responseText : asText(result); }, }); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 2a40cee8..c079fefd 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -2,6 +2,9 @@ import assert from 'node:assert/strict'; import { spawn } from 'node:child_process'; import { createServer } from 'node:http'; import net from 'node:net'; +import { mkdtemp, writeFile, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; import test from 'node:test'; import { setTimeout as delay } from 'node:timers/promises'; import { assertAgentMetadataPolicy } from '../scripts/agent-metadata-policy.mjs'; @@ -331,6 +334,8 @@ async function startFakeFirecrawlBackend(options = {}) { keylessEligible = false, keylessEligibilityResponse, searchResponse, + feedbackResponse, + parseResponse, } = options; const requests = []; const server = createServer(async (req, res) => { @@ -406,6 +411,20 @@ async function startFakeFirecrawlBackend(options = {}) { return; } + if (req.method === 'POST' && req.url === '/v2/feedback') { + const response = feedbackResponse ?? { + status: 200, + body: { + success: true, + feedbackId: '00000000-0000-4000-8000-000000000001', + creditsRefunded: 0, + }, + }; + res.writeHead(response.status, { 'content-type': 'application/json' }); + res.end(JSON.stringify(response.body)); + return; + } + if (req.method === 'POST' && req.url === '/v2/search') { if (searchResponse) { res.writeHead(searchResponse.status, { 'content-type': 'application/json' }); @@ -443,6 +462,11 @@ async function startFakeFirecrawlBackend(options = {}) { } if (req.method === 'POST' && req.url === '/v2/parse') { + if (parseResponse) { + res.writeHead(parseResponse.status, { 'content-type': 'application/json' }); + res.end(JSON.stringify(parseResponse.body)); + return; + } res.writeHead(200, { 'content-type': 'application/json' }); res.end( JSON.stringify({ @@ -543,7 +567,7 @@ test('HTTP cloud keyless transport preserves app challenge without advertising O const anonymousTools = parseSseJson(await unauthenticated.text()).result.tools; assert.deepEqual( anonymousTools.map((tool) => tool.name).sort(), - ['firecrawl_parse', 'firecrawl_scrape', 'firecrawl_search'] + ['firecrawl_feedback', 'firecrawl_parse', 'firecrawl_scrape', 'firecrawl_search'] ); const anonymousParse = anonymousTools.find( (tool) => tool.name === 'firecrawl_parse' @@ -970,7 +994,7 @@ test('stdio transport initializes and lists Firecrawl tools', async (t) => { assert.equal(stderr.includes('TypeError'), false, stderr); }); -test('local keyless stdio keeps profile guidance keyless-scoped and omits feedback tools', async (t) => { +test('local keyless stdio keeps profile guidance keyless-scoped and exposes shared feedback', async (t) => { const child = spawnServer({ FIRECRAWL_API_KEY: '', FIRECRAWL_API_URL: '', @@ -1024,12 +1048,12 @@ test('local keyless stdio keeps profile guidance keyless-scoped and omits feedba const tools = await client.request('tools/list'); const toolNames = tools.tools.map((tool) => tool.name); assert.equal(toolNames.includes('firecrawl_search_feedback'), false); - assert.equal(toolNames.includes('firecrawl_feedback'), false); + assert.equal(toolNames.includes('firecrawl_feedback'), true); const search = tools.tools.find((tool) => tool.name === 'firecrawl_search'); assert.ok(search); assert.match( search.description, - /authenticated responses can include an `id` for optional search feedback/i + /responses can include an `id` for optional feedback/i ); }); @@ -1489,7 +1513,7 @@ test('HTTP cloud transport serves an eligible keyless client and forwards its IP const message = parseSseJson(await toolCall.text()); assert.notEqual(message.result.isError, true); const keylessSearchPayload = JSON.parse(message.result.content[0].text); - assert.equal('id' in keylessSearchPayload, false); + assert.equal(keylessSearchPayload.id, '00000000-0000-4000-8000-000000000000'); const eligibilityCalls = backend.requests.filter( (r) => r.url === '/v2/keyless/eligibility' @@ -3011,3 +3035,193 @@ test('account OAuth tokens cannot replay on keyless and invalid keys get correct .map((request) => request.body.token); assert.deepEqual(introspectedTokens, ['fco_account']); }); + + +test('hosted keyless feedback bypasses exhausted operation allowance and preserves rejection details', async (t) => { + for (const status of [200, 400, 429, 503]) { + await t.test(`HTTP ${status}`, async (t) => { + const body = + status === 200 + ? { + success: true, + feedbackId: '00000000-0000-4000-8000-000000000001', + creditsRefunded: 0, + } + : { + success: false, + error: 'Feedback rejected', + feedbackErrorCode: + status === 429 ? 'DAILY_LIMIT_REACHED' : 'INVALID_BODY', + }; + const backend = await startFakeFirecrawlBackend({ + keylessEligible: false, + feedbackResponse: { status, body }, + }); + t.after(() => backend.close()); + const port = await getFreePort(); + const child = spawnServer({ + CLOUD_SERVICE: 'true', + FASTMCP_ENDPOINT: '/v2/mcp', + FIRECRAWL_API_URL: backend.url, + FIRECRAWL_OAUTH_ISSUER: backend.url, + HTTP_STREAMABLE_SERVER: 'true', + PORT: String(port), + KEYLESS_PROXY_SECRET: 'feedback-test-secret', + }); + t.after(() => stopChild(child)); + await waitForHealth(port, child); + const response = await httpToolCall(port, { + id: `feedback-${status}`, + headers: { 'x-forwarded-for': '203.0.113.71' }, + params: { + name: 'firecrawl_feedback', + arguments: { + endpoint: 'search', + jobId: '00000000-0000-4000-8000-000000000000', + rating: 'partial', + task: 'Read the API retry reference', + assessment: 'The reference explains supported retry intervals.', + observations: [ + { + kind: 'useful', + source: 'web', + position: 1, + basis: 'output', + detail: 'The reference answered the retry question.', + }, + ], + }, + }, + }); + assert.equal(response.status, 200); + const result = JSON.parse( + parseSseJson(await response.text()).result.content[0].text + ); + assert.equal(result.success, status === 200); + if (status !== 200) { + assert.equal(result.feedbackErrorCode, body.feedbackErrorCode); + assert.equal(result.retryable, status >= 500); + } + const submission = backend.requests.find( + (req) => req.url === '/v2/feedback' + ); + assert.ok(submission); + assert.equal(submission.headers.authorization, undefined); + assert.equal( + submission.headers['x-firecrawl-keyless-ip'], + '203.0.113.71' + ); + assert.equal( + submission.headers['x-firecrawl-keyless-secret'], + 'feedback-test-secret' + ); + assert.equal(submission.body.observations[0].position, 1); + assert.equal( + backend.requests.some((req) => req.url === '/v2/keyless/eligibility'), + false + ); + }); + } +}); + +test('keyless Search preserves optional feedback invitations in its tool result', async (t) => { + const metadata = { + jobId: '00000000-0000-4000-8000-000000000000', + feedback: { + endpoint: 'search', + message: 'Optional feedback is available.', + }, + }; + const backend = await startFakeFirecrawlBackend({ + keylessEligible: true, + searchResponse: { + status: 200, + body: { success: true, id: metadata.jobId, data: { web: [] }, metadata }, + }, + }); + t.after(() => backend.close()); + const port = await getFreePort(); + const child = spawnServer({ + CLOUD_SERVICE: 'true', + FASTMCP_ENDPOINT: '/v2/mcp', + FIRECRAWL_API_URL: backend.url, + FIRECRAWL_OAUTH_ISSUER: backend.url, + HTTP_STREAMABLE_SERVER: 'true', + PORT: String(port), + KEYLESS_PROXY_SECRET: 'feedback-test-secret', + }); + t.after(() => stopChild(child)); + await waitForHealth(port, child); + const response = await httpToolCall(port, { + id: 'feedback-invitation', + headers: { 'x-forwarded-for': '203.0.113.71' }, + params: { + name: 'firecrawl_search', + arguments: { query: 'retry behavior' }, + }, + }); + assert.deepEqual( + JSON.parse(parseSseJson(await response.text()).result.content[0].text) + .metadata, + metadata + ); +}); + +test('local Parse preserves job evidence on success and failure and respects invitation opt-out', async (t) => { + const directory = await mkdtemp(join(tmpdir(), 'feedback-parse-')); + t.after(() => rm(directory, { recursive: true, force: true })); + const filePath = join(directory, 'fixture.html'); + await writeFile(filePath, '

Parsed fixture

'); + for (const status of [200, 500]) { + for (const disabled of [false, true]) { + await t.test(`HTTP ${status}, disabled ${disabled}`, async (t) => { + const metadata = { + jobId: '00000000-0000-4000-8000-000000000000', + feedback: { + endpoint: 'parse', + message: 'Optional feedback is available.', + }, + }; + const body = + status === 200 + ? { + success: true, + data: { markdown: '# Parsed fixture', metadata }, + } + : { success: false, error: 'Parsing failed', metadata }; + const backend = await startFakeFirecrawlBackend({ + parseResponse: { status, body }, + }); + t.after(() => backend.close()); + const child = spawnServer({ + FIRECRAWL_API_KEY: '', + FIRECRAWL_API_URL: backend.url, + CLOUD_SERVICE: '', + FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK: disabled ? 'true' : '', + }); + t.after(() => stopChild(child)); + const client = new StdioMcpClient(child); + await client.request('initialize', { + capabilities: {}, + clientInfo: { name: 'parse-feedback-test', version: '0.0.0' }, + protocolVersion: '2025-06-18', + }); + client.notify('notifications/initialized'); + const result = await client.request('tools/call', { + name: 'firecrawl_parse', + arguments: { filePath }, + }); + const payload = + result.structuredContent ?? JSON.parse(result.content[0].text); + const returnedMetadata = payload.data?.metadata ?? payload.metadata; + assert.equal(returnedMetadata.jobId, metadata.jobId); + assert.equal(Boolean(returnedMetadata.feedback), !disabled); + assert.equal(result.isError === true, status !== 200); + const call = backend.requests.find((req) => req.url === '/v2/parse'); + assert.equal(call.headers.authorization, undefined); + assert.match(call.headers['content-type'], /multipart\/form-data/); + assert.match(call.raw, /Parsed fixture/); + }); + } + } +}); From 64e4a0dea78489d47d00ef8d6392594552250573 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 15:23:56 -0500 Subject: [PATCH 02/25] docs(mcp): clarify feedback contracts by authentication --- README.md | 33 +++++++++++++++++++++++++-------- 1 file changed, 25 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 12821fc7..003ae3ea 100644 --- a/README.md +++ b/README.md @@ -498,7 +498,7 @@ For scientific papers, see [Research Tools](#12-research-tools-firecrawl_researc **Returns:** -- Array of search results (with optional scraped content), plus an `id` field. Pass that `id` to `firecrawl_search_feedback` after you've used the results to refund 1 credit (search costs 2) and improve search quality. +- Array of search results with optional scraped content, plus an `id` field. Keyless callers can use the returned job reference and optional invitation with `firecrawl_feedback`. Authenticated callers can continue using `firecrawl_search_feedback` with its existing fields and policy. **Prompt Example:** @@ -506,9 +506,9 @@ For scientific papers, see [Research Tools](#12-research-tools-firecrawl_researc ### 3b. Search Feedback Tool (`firecrawl_search_feedback`) -Sends structured feedback on a previous `firecrawl_search` result. The first feedback per search id refunds 1 credit and improves Firecrawl's search quality. Idempotent per search id. +This authenticated tool sends structured feedback on a previous `firecrawl_search` result. Its request fields, eligibility, and refund policy are unchanged. It is idempotent per search ID. Keyless callers must use `firecrawl_feedback`; this legacy tool does not accept keyless jobs. -**Call this after every search you actually use** (or that didn't help). Bad/partial feedback with `missingContent` is just as valuable as good feedback. +Feedback is optional. If submitting, use only observations already available after processing the results; no user interview or additional investigation is required. **Opt out:** set `FIRECRAWL_NO_SEARCH_FEEDBACK=1` (or `FIRECRAWL_DISABLE_SEARCH_FEEDBACK=1`) in the environment when starting the MCP server. The `firecrawl_search_feedback` tool will not be registered, so agents can't call it. Team admins can also disable feedback server-side; in that case the tool is registered but always returns `feedbackErrorCode: "TEAM_OPTED_OUT"`. @@ -548,17 +548,34 @@ Sends structured feedback on a previous `firecrawl_search` result. The first fee ### 3c. Generic Feedback Tool (`firecrawl_feedback`) -Sends structured feedback for a completed v2 endpoint job through `/v2/feedback`. -Use this for endpoint-level feedback on `scrape`, `parse`, `map`, or `search` -jobs. For search-result quality specifically, prefer -`firecrawl_search_feedback` because it includes search-specific guidance. +Sends optional evidence through `/v2/feedback`. Keyless Search, Scrape, and Parse +jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 +`observations`. Each observation has `kind`, `detail`, and `basis`: `output`, +`source_comparison`, or `expectation`. Source comparisons also require +`comparison: {reference, detail}`. + +For Search, report useful or irrelevant results by source group and one-based +position, or missing information with a topic and any already-known source URLs. +For Scrape, report correct, missing, incorrect, or failed output. For Parse, +report correct output or text, table, layout, or completeness issues. The tool +description lists the complete fields. Use only available evidence and keep +unverified expectations distinct from source comparisons. + +Keyless submissions are limited to one per identity, category, and UTC day across +clients, with references valid for 24 hours. Feedback remains available after +operation allowance is exhausted and does not consume or restore that allowance. + +Authenticated callers retain the existing issue/note fields for Search, Scrape, +Parse, and Map. For authenticated Search-specific feedback, continue using +`firecrawl_search_feedback`. Choose the contract that matches the originating +job's authentication; adding credentials does not convert a keyless job. Keep feedback concise: use issue codes, tags, short notes, URLs, page numbers, and small metadata objects. Do not include raw scrape/parse outputs. **Opt out:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) in the environment when starting the MCP server. The `firecrawl_feedback` tool will not be registered, so agents cannot call it. -**Usage Example:** +**Authenticated usage example:** ```json { From 1acece410119d830bcd4460f1b24e182837e296c Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 15:35:28 -0500 Subject: [PATCH 03/25] fix(mcp): preserve explicit local parse API configuration --- src/index.ts | 7 ++++++- tests/mcp-smoke.test.mjs | 23 +++++++++++++++++++++++ 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/src/index.ts b/src/index.ts index 1362e86f..79aad483 100644 --- a/src/index.ts +++ b/src/index.ts @@ -3191,7 +3191,12 @@ Set \`redactPII\` to request redaction of personally identifiable information in return executeHostedParse(args as ParseToolArgs, session, log); } - const apiUrl = resolveApiBaseUrl(); + const apiUrl = process.env.FIRECRAWL_API_URL; + if (!apiUrl) { + throw new Error( + 'firecrawl_parse requires FIRECRAWL_API_URL to be set to a self-hosted Firecrawl API instance.' + ); + } const { filePath, diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index c079fefd..7e0cd0e9 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -3225,3 +3225,26 @@ test('local Parse preserves job evidence on success and failure and respects inv } } }); + + +test('local Parse still requires an explicit API URL', async (t) => { + const child = spawnServer({ + FIRECRAWL_API_KEY: '', + FIRECRAWL_API_URL: '', + CLOUD_SERVICE: '', + }); + t.after(() => stopChild(child)); + const client = new StdioMcpClient(child); + await client.request('initialize', { + capabilities: {}, + clientInfo: { name: 'parse-configuration-test', version: '0.0.0' }, + protocolVersion: '2025-06-18', + }); + client.notify('notifications/initialized'); + const result = await client.request('tools/call', { + name: 'firecrawl_parse', + arguments: { filePath: '/not-read-without-an-api-url.html' }, + }); + assert.equal(result.isError, true); + assert.match(result.content[0].text, /requires FIRECRAWL_API_URL/); +}); From b935f42c47b116dd19f94fff45754884c8fe65cf Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 15:44:00 -0500 Subject: [PATCH 04/25] fix(mcp): align local parse with API URL defaults --- README.md | 4 +- src/index.ts | 22 ++++------ tests/mcp-smoke.test.mjs | 91 ++++++++++++++++++++++++++++++---------- 3 files changed, 80 insertions(+), 37 deletions(-) diff --git a/README.md b/README.md index 003ae3ea..0dfc979f 100644 --- a/README.md +++ b/README.md @@ -666,11 +666,11 @@ Check the status and results of an existing crawl job by ID. Parse local files or hosted upload references with Firecrawl's `/v2/parse` endpoint. -**Best for:** PDFs, Word documents, spreadsheets, HTML files, and other documents that need markdown or structured JSON output. Hosted MCP supports a two-step upload-ref flow; local direct file reads require a self-hosted `FIRECRAWL_API_URL`. +**Best for:** PDFs, Word documents, spreadsheets, HTML files, and other documents that need markdown or structured JSON output. Hosted MCP supports a two-step upload-ref flow; local MCP reads the requested file and uploads it to the configured API, defaulting to the Firecrawl cloud API. **Not recommended for:** Remote URLs (use scrape), multiple files in one call (call parse once per file), or browser-only actions such as screenshots and clicks. -**Hosted MCP flow:** Hosted MCP cannot read the caller's filesystem directly. Call `firecrawl_parse` with `filePath` to receive a short-lived upload command and `nextToolCall`, upload the file locally, then call `firecrawl_parse` again with the returned `uploadRef`. Minting the hosted upload URL requires Firecrawl auth or keyless eligibility. In local `npx firecrawl-mcp` mode, direct file parsing currently requires `FIRECRAWL_API_URL` pointing to a self-hosted Firecrawl API; a plain cloud API-key-only local server cannot read and upload files through this tool. +**Hosted MCP flow:** Hosted MCP cannot read the caller's filesystem directly. Call `firecrawl_parse` with `filePath` to receive a short-lived upload command and `nextToolCall`, upload the file locally, then call `firecrawl_parse` again with the returned `uploadRef`. Minting the hosted upload URL requires Firecrawl auth or keyless eligibility. In local `npx firecrawl-mcp` mode, the server uploads `filePath` to `FIRECRAWL_API_URL` when configured, or to `https://api.firecrawl.dev` when unset, matching Search and Scrape. Both authenticated and eligible keyless calls are supported. Running MCP locally does not perform parsing on the local machine; configure your self-hosted API URL if that is where files should be processed. **Usage Example:** diff --git a/src/index.ts b/src/index.ts index 79aad483..1b6453c6 100644 --- a/src/index.ts +++ b/src/index.ts @@ -657,13 +657,12 @@ async function authenticateRequest( !process.env.FIRECRAWL_API_KEY && !process.env.FIRECRAWL_API_URL ) { - // No credential and no self-hosted URL: run in keyless mode. scrape and - // search work for free (rate-limited per IP) against the Firecrawl cloud; - // every other tool needs an API key and will return Unauthorized. + // Without credentials or a custom API URL, use the cloud keyless tools. console.error( - 'No FIRECRAWL_API_KEY or FIRECRAWL_API_URL set — running in keyless mode. ' + - 'firecrawl_scrape and firecrawl_search are free (rate-limited per IP) against the Firecrawl cloud; ' + - 'other tools require an API key (get one free at https://firecrawl.dev).' + 'No FIRECRAWL_API_KEY or FIRECRAWL_API_URL set. Running in keyless mode. ' + + 'firecrawl_scrape, firecrawl_search, and firecrawl_parse use the Firecrawl cloud with usage limits. ' + + 'Local Parse uploads the requested file to the API. Optional firecrawl_feedback is available when enabled. ' + + 'Other tools require an API key (get one free at https://firecrawl.dev).' ); } @@ -954,7 +953,7 @@ const openAiAppsChallengeToken = normalizeHeader( ); const FULL_PROFILE_INSTRUCTIONS = `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question — code behaviour, a library or framework, an API contract, an error message, or a known bug — firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of repositories, GitHub issues, merged pull requests, READMEs, and curated documentation sites. Provide only the required inputs and account for stated network or external side effects.`; -const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_feedback accepts optional evidence about those jobs without consuming operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse processes supported local files through its two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, and firecrawl_research_* for paper-index and repository research.`; +const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_feedback accepts optional evidence about those jobs without consuming operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse uploads local file bytes to the API in local MCP and uses a two-phase upload flow in hosted MCP. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, and firecrawl_research_* for paper-index and repository research.`; // The search surface exposes web/developer/research search only. Its instructions // and tool copy describe just those tools and stay neutral about how a client @@ -3181,7 +3180,7 @@ Optional feedback is available through \`firecrawl_feedback\` using the returned Parse one supported document into markdown, HTML, links, summary, targeted answers, or JSON matching a schema. Supported inputs include common HTML, PDF, Word, RTF, OpenDocument, and spreadsheet files; PDF parsing can be bounded with \`pdfOptions.maxPages\`. -Local MCP reads \`filePath\` from the server filesystem. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. +Local MCP reads \`filePath\` from the server filesystem and uploads it to the configured \`FIRECRAWL_API_URL\`, or the Firecrawl cloud API when unset. Parsing happens on that API server. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. Set \`redactPII\` to request redaction of personally identifiable information in the returned content. \`zeroDataRetention\` requires an eligible authenticated account; omit it for anonymous keyless use. Returns upload instructions for hosted phase one or parsed document content for the final call. `, @@ -3191,12 +3190,7 @@ Set \`redactPII\` to request redaction of personally identifiable information in return executeHostedParse(args as ParseToolArgs, session, log); } - const apiUrl = process.env.FIRECRAWL_API_URL; - if (!apiUrl) { - throw new Error( - 'firecrawl_parse requires FIRECRAWL_API_URL to be set to a self-hosted Firecrawl API instance.' - ); - } + const apiUrl = resolveApiBaseUrl(); const { filePath, diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 7e0cd0e9..8b401a65 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1015,7 +1015,7 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar const apiKeyGuidance = init.instructions.slice(apiKeyBoundaryIndex); assert.match( keylessGuidance, - /Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits/i + /Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits/i ); assert.match( keylessGuidance, @@ -3227,24 +3227,73 @@ test('local Parse preserves job evidence on success and failure and respects inv }); -test('local Parse still requires an explicit API URL', async (t) => { - const child = spawnServer({ - FIRECRAWL_API_KEY: '', - FIRECRAWL_API_URL: '', - CLOUD_SERVICE: '', - }); - t.after(() => stopChild(child)); - const client = new StdioMcpClient(child); - await client.request('initialize', { - capabilities: {}, - clientInfo: { name: 'parse-configuration-test', version: '0.0.0' }, - protocolVersion: '2025-06-18', - }); - client.notify('notifications/initialized'); - const result = await client.request('tools/call', { - name: 'firecrawl_parse', - arguments: { filePath: '/not-read-without-an-api-url.html' }, - }); - assert.equal(result.isError, true); - assert.match(result.content[0].text, /requires FIRECRAWL_API_URL/); +test('local Parse defaults to the cloud API with and without credentials', async (t) => { + const directory = await mkdtemp(join(tmpdir(), 'parse-default-api-')); + t.after(() => rm(directory, { recursive: true, force: true })); + const filePath = join(directory, 'fixture.html'); + await writeFile(filePath, '

Parsed fixture

'); + for (const authenticated of [false, true]) { + await t.test(`authenticated ${authenticated}`, async (t) => { + const backend = await startFakeFirecrawlBackend(); + t.after(() => backend.close()); + const preloadPath = join(directory, `fetch-${authenticated}.mjs`); + await writeFile( + preloadPath, + ` + const originalFetch = globalThis.fetch; + globalThis.fetch = (url, init) => { + if (url !== 'https://api.firecrawl.dev/v2/parse') { + throw new Error('Unexpected outbound URL'); + } + return originalFetch(${JSON.stringify(`${backend.url}/v2/parse`)}, { + ...init, + headers: { ...init.headers, 'x-test-original-url': url }, + }); + }; + ` + ); + const child = spawnServer({ + FIRECRAWL_API_KEY: authenticated ? 'fc-parse-test' : '', + FIRECRAWL_OAUTH_TOKEN: '', + FIRECRAWL_API_URL: '', + CLOUD_SERVICE: '', + NODE_OPTIONS: `--import=${preloadPath}`, + }); + t.after(() => stopChild(child)); + const client = new StdioMcpClient(child); + await client.request('initialize', { + capabilities: {}, + clientInfo: { name: 'parse-default-api-test', version: '0.0.0' }, + protocolVersion: '2025-06-18', + }); + client.notify('notifications/initialized'); + const tools = await client.request('tools/list'); + const parse = tools.tools.find((tool) => tool.name === 'firecrawl_parse'); + assert.ok(parse); + assert.match( + parse.description, + /uploads it.*Firecrawl cloud API when unset/ + ); + const result = await client.request('tools/call', { + name: 'firecrawl_parse', + arguments: { filePath, formats: ['markdown'] }, + }); + assert.notEqual(result.isError, true); + assert.equal( + JSON.parse(result.content[0].text).data.markdown, + '# Parsed fixture' + ); + assert.equal(backend.requests.length, 1); + const call = backend.requests[0]; + assert.equal( + call.headers['x-test-original-url'], + 'https://api.firecrawl.dev/v2/parse' + ); + assert.equal( + call.headers.authorization, + authenticated ? 'Bearer fc-parse-test' : undefined + ); + assert.match(call.raw, /Parsed fixture/); + }); + } }); From c597f30c39b8b6b45bf568685c7d6128ed343249 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 15:45:15 -0500 Subject: [PATCH 05/25] test(mcp): verify local and hosted parse guidance --- tests/mcp-smoke.test.mjs | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 8b401a65..7ebe5b87 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1026,7 +1026,10 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar /firecrawl_search with categories: \["research"\].*research-affiliated websites/i ); assert.match(keylessGuidance, /firecrawl_scrape retrieves one supplied page/i); - assert.match(keylessGuidance, /firecrawl_parse processes supported local files/i); + assert.match( + keylessGuidance, + /firecrawl_parse uploads local file bytes to the API in local MCP and uses a two-phase upload flow in hosted MCP/i + ); assert.doesNotMatch( keylessGuidance, /firecrawl_(?:map|agent|agent_status|research_)/i From 5d1507acd81294356fe25c3edcc412dd03396608 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 16:09:33 -0500 Subject: [PATCH 06/25] fix(mcp): forward feedback invitation preferences --- README.md | 4 +- src/index.ts | 12 ++++- tests/mcp-smoke.test.mjs | 94 ++++++++++++++++++++++------------------ 3 files changed, 66 insertions(+), 44 deletions(-) diff --git a/README.md b/README.md index 0dfc979f..b2dfd752 100644 --- a/README.md +++ b/README.md @@ -35,7 +35,7 @@ A Model Context Protocol (MCP) server that brings [Firecrawl](https://github.com - Use the `firecrawl_monitor_*` tools when the same page needs to be checked on a recurring schedule with diffs and change alerts, rather than fetched once. - Consider something else when you need to hold a browser session open across many of your own steps with your own retry and termination logic: each `firecrawl_interact` call runs one `prompt` or `code` turn to completion and returns control — the session can persist across calls via `scrapeId` and ends with `firecrawl_interact_stop`, but you cannot drive it interactively step-by-step from the client side within a single call. -This server lists 25 tools when the full profile registers with default settings (feedback tools included, not running in local-keyless mode). Setting `FIRECRAWL_NO_SEARCH_FEEDBACK=1` and/or `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` removes the corresponding feedback tools and reduces this count, as does local keyless startup. For clients with a tool-slot limit: the hosted keyless endpoint (`https://mcp.firecrawl.dev/v2/mcp`, no API key) exposes only 3 — `firecrawl_scrape`, `firecrawl_search`, `firecrawl_parse` — and the dedicated [search-only endpoint](#search-only-endpoint) (`https://mcp.firecrawl.dev/v2/mcp-search`) exposes a fixed 6 read-only tools. +This server lists 25 tools when the full profile registers with default settings (feedback tools included, not running in local-keyless mode). Setting `FIRECRAWL_NO_SEARCH_FEEDBACK=1` and/or `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` removes the corresponding feedback tools and reduces this count, as does local keyless startup. For clients with a tool-slot limit: the hosted keyless endpoint (`https://mcp.firecrawl.dev/v2/mcp`, no API key) exposes 4 tools: `firecrawl_scrape`, `firecrawl_search`, `firecrawl_parse`, and `firecrawl_feedback`. The dedicated [search-only endpoint](#search-only-endpoint) (`https://mcp.firecrawl.dev/v2/mcp-search`) exposes a fixed 6 read-only tools. ## Installation @@ -573,7 +573,7 @@ job's authentication; adding credentials does not convert a keyless job. Keep feedback concise: use issue codes, tags, short notes, URLs, page numbers, and small metadata objects. Do not include raw scrape/parse outputs. -**Opt out:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) in the environment when starting the MCP server. The `firecrawl_feedback` tool will not be registered, so agents cannot call it. +**Opt out:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) in the environment when starting the MCP server. The `firecrawl_feedback` tool will not be registered, so agents cannot call it. Operation requests also send `x-firecrawl-no-feedback: 1` so the API does not issue or count invitations that this server suppresses. **Authenticated usage example:** diff --git a/src/index.ts b/src/index.ts index 1b6453c6..e6b18df7 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2374,6 +2374,7 @@ async function keylessPost( const headers: Record = { ...ORIGIN_HEADERS, 'Content-Type': 'application/json', + ...feedbackPreferenceHeaders(), }; // Forward the real client IP (secret-authenticated) when proxying keyless // requests through the hosted MCP, so the API rate-limits per real IP. @@ -2532,6 +2533,12 @@ const ENDPOINT_FEEDBACK_DISABLED = feedbackEnvEnabled( 'FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK' ); +function feedbackPreferenceHeaders(): Record { + return ENDPOINT_FEEDBACK_DISABLED + ? { 'x-firecrawl-no-feedback': '1' } + : {}; +} + if (SEARCH_FEEDBACK_DISABLED) { console.error( '[firecrawl-mcp] Search feedback tool disabled by FIRECRAWL_NO_SEARCH_FEEDBACK; firecrawl_search_feedback will not be registered.' @@ -3220,7 +3227,10 @@ Set \`redactPII\` to request redaction of personally identifiable information in form.append('file', blob, filename); form.append('options', JSON.stringify(optionsPayload)); - const headers: Record = { ...ORIGIN_HEADERS }; + const headers: Record = { + ...ORIGIN_HEADERS, + ...feedbackPreferenceHeaders(), + }; const credential = credentialForOutboundRequest(session); if (credential) { headers['Authorization'] = `Bearer ${credential}`; diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 7ebe5b87..7a1c8f6e 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -3127,48 +3127,56 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv } }); -test('keyless Search preserves optional feedback invitations in its tool result', async (t) => { - const metadata = { - jobId: '00000000-0000-4000-8000-000000000000', - feedback: { - endpoint: 'search', - message: 'Optional feedback is available.', - }, - }; - const backend = await startFakeFirecrawlBackend({ - keylessEligible: true, - searchResponse: { - status: 200, - body: { success: true, id: metadata.jobId, data: { web: [] }, metadata }, - }, - }); - t.after(() => backend.close()); - const port = await getFreePort(); - const child = spawnServer({ - CLOUD_SERVICE: 'true', - FASTMCP_ENDPOINT: '/v2/mcp', - FIRECRAWL_API_URL: backend.url, - FIRECRAWL_OAUTH_ISSUER: backend.url, - HTTP_STREAMABLE_SERVER: 'true', - PORT: String(port), - KEYLESS_PROXY_SECRET: 'feedback-test-secret', - }); - t.after(() => stopChild(child)); - await waitForHealth(port, child); - const response = await httpToolCall(port, { - id: 'feedback-invitation', - headers: { 'x-forwarded-for': '203.0.113.71' }, - params: { - name: 'firecrawl_search', - arguments: { query: 'retry behavior' }, - }, +for (const disabled of [false, true]) { + test(`keyless Search preserves references and forwards invitation opt-out: ${disabled}`, async (t) => { + const metadata = { + jobId: '00000000-0000-4000-8000-000000000000', + feedback: { + endpoint: 'search', + message: 'Optional feedback is available.', + }, + }; + const backend = await startFakeFirecrawlBackend({ + keylessEligible: true, + searchResponse: { + status: 200, + body: { success: true, id: metadata.jobId, data: { web: [] }, metadata }, + }, + }); + t.after(() => backend.close()); + const port = await getFreePort(); + const child = spawnServer({ + CLOUD_SERVICE: 'true', + FIRECRAWL_NO_ENDPOINT_FEEDBACK: disabled ? 'true' : '', + FASTMCP_ENDPOINT: '/v2/mcp', + FIRECRAWL_API_URL: backend.url, + FIRECRAWL_OAUTH_ISSUER: backend.url, + HTTP_STREAMABLE_SERVER: 'true', + PORT: String(port), + KEYLESS_PROXY_SECRET: 'feedback-test-secret', + }); + t.after(() => stopChild(child)); + await waitForHealth(port, child); + const response = await httpToolCall(port, { + id: 'feedback-invitation', + headers: { 'x-forwarded-for': '203.0.113.71' }, + params: { + name: 'firecrawl_search', + arguments: { query: 'retry behavior' }, + }, + }); + assert.deepEqual( + JSON.parse(parseSseJson(await response.text()).result.content[0].text) + .metadata, + disabled ? { jobId: metadata.jobId } : metadata + ); + const call = backend.requests.find((req) => req.url === '/v2/search'); + assert.equal( + call.headers['x-firecrawl-no-feedback'], + disabled ? '1' : undefined + ); }); - assert.deepEqual( - JSON.parse(parseSseJson(await response.text()).result.content[0].text) - .metadata, - metadata - ); -}); +} test('local Parse preserves job evidence on success and failure and respects invitation opt-out', async (t) => { const directory = await mkdtemp(join(tmpdir(), 'feedback-parse-')); @@ -3222,6 +3230,10 @@ test('local Parse preserves job evidence on success and failure and respects inv assert.equal(result.isError === true, status !== 200); const call = backend.requests.find((req) => req.url === '/v2/parse'); assert.equal(call.headers.authorization, undefined); + assert.equal( + call.headers['x-firecrawl-no-feedback'], + disabled ? '1' : undefined + ); assert.match(call.headers['content-type'], /multipart\/form-data/); assert.match(call.raw, /Parsed fixture/); }); From 126b4843f8ff40968024de33b2f6c09b4c581db7 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 17:53:11 -0500 Subject: [PATCH 07/25] docs(mcp): clarify shared keyless feedback daily limit --- README.md | 7 ++++--- src/index.ts | 2 +- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index b2dfd752..3c72942e 100644 --- a/README.md +++ b/README.md @@ -561,9 +561,10 @@ report correct output or text, table, layout, or completeness issues. The tool description lists the complete fields. Use only available evidence and keep unverified expectations distinct from source comparisons. -Keyless submissions are limited to one per identity, category, and UTC day across -clients, with references valid for 24 hours. Feedback remains available after -operation allowance is exhausted and does not consume or restore that allowance. +Keyless submissions are limited to one per identity per UTC day across Search, +Scrape, Parse, and all clients, with references valid for 24 hours. Feedback remains +available after operation allowance is exhausted and does not consume or restore +that allowance. Authenticated callers retain the existing issue/note fields for Search, Scrape, Parse, and Map. For authenticated Search-specific feedback, continue using diff --git a/src/index.ts b/src/index.ts index e6b18df7..30975bb2 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2701,7 +2701,7 @@ Submit optional quality feedback for a search, scrape, parse, or map job. Authen Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. Search kinds are useful or irrelevant with source (web, images, news) and one-based position within that delivered group, or missing with topic and optional knownSources URLs. Scrape kinds are correct, missing, incorrect, or failure, with optional location and already-observed retryOutcome. Parse kinds are correct, text, table, layout, or completeness, with optional location. -Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. A failed scrape can be reported with its jobId and observed failure. One accepted submission per keyless identity, category, and UTC day, shared across clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. +Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. A failed scrape can be reported with its jobId and observed failure. One accepted submission per keyless identity per UTC day, shared across Search, Scrape, Parse, and all clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, parameters: z.object({ endpoint: z.enum(['search', 'scrape', 'parse', 'map']), From ff71c5760c95bdeb6028c356fdfe7b4099e7a6f0 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 18:15:35 -0500 Subject: [PATCH 08/25] fix(mcp): honor feedback opt-outs across authenticated tools --- src/index.ts | 26 ++++++++++++++++------ tests/mcp-search-profile.test.mjs | 16 +++++++++++++ tests/mcp-smoke.test.mjs | 37 +++++++++++++++++++++++++++++++ 3 files changed, 72 insertions(+), 7 deletions(-) diff --git a/src/index.ts b/src/index.ts index 30975bb2..026a7d77 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1483,7 +1483,12 @@ function createClient(apiKey?: string): FirecrawlApp { config.apiKey = apiKey; } - return new FirecrawlApp(config); + const client = new FirecrawlApp(config); + Object.assign( + (client as any).http.instance.defaults.headers.common, + feedbackPreferenceHeaders() + ); + return client; } const ORIGIN = 'mcp-fastmcp'; @@ -1912,6 +1917,7 @@ async function apiPostJson( headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${apiKey}`, + ...feedbackPreferenceHeaders(), }, body: JSON.stringify(body), }); @@ -1930,7 +1936,7 @@ async function apiPostJson( response.status ); } - return parsed; + return filterFeedbackInvitation(parsed); } async function apiPostJsonForSession( @@ -2147,7 +2153,7 @@ Returns the selected content formats and page metadata. ...cleaned, origin: ORIGIN, } as any); - return asText(res); + return asText(filterFeedbackInvitation(res)); }, }); @@ -2253,7 +2259,7 @@ Each web result is a title, URL, and description, not the page. Add \`scrapeOpti // supports the optional authenticated `firecrawl_search_feedback` workflow. const client = getClient(session); const httpRes = await (client as any).http.post('/v2/search', searchBody); - return asText(httpRes?.data ?? {}); + return asText(filterFeedbackInvitation(httpRes?.data ?? {})); }, }); @@ -2407,7 +2413,10 @@ async function keylessPost( throw new UserError(String(payload.message), payload); } if (json?.metadata?.jobId) { - throw new UserError(asText(json), json); + throw new UserError( + json.error || `Firecrawl request failed (HTTP ${response.status})`, + json + ); } throw new Error( json?.error || `Firecrawl request failed (HTTP ${response.status})` @@ -3258,7 +3267,10 @@ Set \`redactPII\` to request redaction of personally identifiable information in } if (!response.ok) { if (result?.metadata?.jobId) { - throw new UserError(asText(result), result); + throw new UserError( + result.error || `Parse request failed (HTTP ${response.status})`, + result + ); } throw new Error( `Parse request failed with status ${response.status}: ${responseText}` @@ -3351,7 +3363,7 @@ Returns \`{ success, data, id, creditsUsed }\`, with source arrays in \`data\`. log.info('Searching', { query: searchQuery }); const client = getClientFn(session); const httpRes = await (client as any).http.post('/v2/search', searchBody); - return asText(httpRes?.data ?? {}); + return asText(filterFeedbackInvitation(httpRes?.data ?? {})); }, }); } diff --git a/tests/mcp-search-profile.test.mjs b/tests/mcp-search-profile.test.mjs index b3780c31..c6a01558 100644 --- a/tests/mcp-search-profile.test.mjs +++ b/tests/mcp-search-profile.test.mjs @@ -111,6 +111,7 @@ async function startFakeBackend(options = {}) { const { apiKeyFromIntrospection = 'fc-from-introspection', introspectionAud, + searchMetadata, } = options; const requests = []; const server = createServer(async (req, res) => { @@ -200,6 +201,7 @@ async function startFakeBackend(options = {}) { : [{ title: 'Example Domain', url: 'https://example.com/' }], }, id: '00000000-0000-4000-8000-000000000000', + ...(searchMetadata ? { metadata: searchMetadata } : {}), success: true, }) ); @@ -1085,3 +1087,17 @@ test('companion telemetry follows credential precedence without resolving API ke ); assert.doesNotMatch(getStdout(), /fc-primary-credential|fco_secondary-credential/); }); + +test('search profile honors feedback opt-out and preserves its job reference', async (t) => { + const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { message: 'Optional feedback.' } }; + const backend = await startFakeBackend({ searchMetadata: metadata }); + t.after(() => backend.close()); + const { searchPort } = await startHostedServer(t, { FIRECRAWL_API_URL: backend.url, FIRECRAWL_NO_ENDPOINT_FEEDBACK: '1' }); + const response = await jsonRpc(searchPort, SEARCH_ENDPOINT, { id: 1, method: 'tools/call', headers: { 'x-api-key': 'fc-test' }, + params: { name: 'firecrawl_search', arguments: { query: 'retry behavior' } } }); + const message = parseSseJson(await response.text()); + assert.notEqual(message.result.isError, true); + assert.deepEqual(JSON.parse(message.result.content[0].text).metadata, { jobId: metadata.jobId }); + const call = backend.requests.find(req => req.url === '/v2/search'); + assert.equal(call.headers['x-firecrawl-no-feedback'], '1'); +}); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 7a1c8f6e..91d6ebba 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -334,6 +334,7 @@ async function startFakeFirecrawlBackend(options = {}) { keylessEligible = false, keylessEligibilityResponse, searchResponse, + scrapeResponse, feedbackResponse, parseResponse, } = options; @@ -443,6 +444,12 @@ async function startFakeFirecrawlBackend(options = {}) { return; } + if (req.method === 'POST' && req.url === '/v2/scrape' && scrapeResponse) { + res.writeHead(scrapeResponse.status, { 'content-type': 'application/json' }); + res.end(JSON.stringify(scrapeResponse.body)); + return; + } + if (req.method === 'POST' && req.url === '/v2/parse/upload-url') { res.writeHead(200, { 'content-type': 'application/json' }); res.end( @@ -3228,6 +3235,7 @@ test('local Parse preserves job evidence on success and failure and respects inv assert.equal(returnedMetadata.jobId, metadata.jobId); assert.equal(Boolean(returnedMetadata.feedback), !disabled); assert.equal(result.isError === true, status !== 200); + if (status !== 200) assert.equal(result.content[0].text, 'Parsing failed'); const call = backend.requests.find((req) => req.url === '/v2/parse'); assert.equal(call.headers.authorization, undefined); assert.equal( @@ -3312,3 +3320,32 @@ test('local Parse defaults to the cloud API with and without credentials', async }); } }); + +for (const endpoint of ['search', 'scrape', 'parse']) { + for (const disabled of [false, true]) { + test(`authenticated ${endpoint} preserves references and honors feedback opt-out: ${disabled}`, async (t) => { + const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { endpoint, message: 'Optional feedback.' } }; + const body = endpoint === 'search' + ? { success: true, id: metadata.jobId, data: { web: [] }, metadata } + : { success: true, data: { markdown: 'Observed content', metadata } }; + const backend = await startFakeFirecrawlBackend({ [`${endpoint}Response`]: { status: 200, body } }); + t.after(() => backend.close()); + const port = await getFreePort(); + const child = spawnServer({ CLOUD_SERVICE: 'true', FIRECRAWL_NO_ENDPOINT_FEEDBACK: disabled ? 'true' : '', + FASTMCP_ENDPOINT: '/v2/mcp', FIRECRAWL_API_URL: backend.url, FIRECRAWL_OAUTH_ISSUER: backend.url, + HTTP_STREAMABLE_SERVER: 'true', PORT: String(port), KEYLESS_PROXY_SECRET: 'feedback-test-secret' }); + t.after(() => stopChild(child)); + await waitForHealth(port, child); + const response = await httpToolCall(port, { id: 'authenticated-feedback-preference', headers: { Authorization: 'Bearer fc-feedback-test' }, params: { + name: `firecrawl_${endpoint}`, arguments: endpoint === 'search' ? { query: 'retry behavior' } : endpoint === 'scrape' ? { url: 'https://example.com/' } : { uploadRef: 'test-upload-ref' }, + } }); + const result = parseSseJson(await response.text()).result; + assert.notEqual(result.isError, true); + const payload = JSON.parse(result.content[0].text); + assert.deepEqual(payload.metadata ?? payload.data?.metadata, disabled ? { jobId: metadata.jobId } : metadata); + const call = backend.requests.find(req => req.url === `/v2/${endpoint}`); + assert.equal(call.headers.authorization, 'Bearer fc-feedback-test'); + assert.equal(call.headers['x-firecrawl-no-feedback'], disabled ? '1' : undefined); + }); + } +} From f247bf65afd784a2b1bcb196a9335db28ee51bb5 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 11 Sep 2026 21:53:33 -0500 Subject: [PATCH 09/25] fix(mcp): retain keyless feedback regardless of client preferences --- README.md | 4 +- src/index.ts | 71 +++++++++++-------------------- tests/mcp-search-profile.test.mjs | 6 +-- tests/mcp-smoke.test.mjs | 46 ++++++++++++++++---- 4 files changed, 66 insertions(+), 61 deletions(-) diff --git a/README.md b/README.md index 3c72942e..c5abf006 100644 --- a/README.md +++ b/README.md @@ -35,7 +35,7 @@ A Model Context Protocol (MCP) server that brings [Firecrawl](https://github.com - Use the `firecrawl_monitor_*` tools when the same page needs to be checked on a recurring schedule with diffs and change alerts, rather than fetched once. - Consider something else when you need to hold a browser session open across many of your own steps with your own retry and termination logic: each `firecrawl_interact` call runs one `prompt` or `code` turn to completion and returns control — the session can persist across calls via `scrapeId` and ends with `firecrawl_interact_stop`, but you cannot drive it interactively step-by-step from the client side within a single call. -This server lists 25 tools when the full profile registers with default settings (feedback tools included, not running in local-keyless mode). Setting `FIRECRAWL_NO_SEARCH_FEEDBACK=1` and/or `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` removes the corresponding feedback tools and reduces this count, as does local keyless startup. For clients with a tool-slot limit: the hosted keyless endpoint (`https://mcp.firecrawl.dev/v2/mcp`, no API key) exposes 4 tools: `firecrawl_scrape`, `firecrawl_search`, `firecrawl_parse`, and `firecrawl_feedback`. The dedicated [search-only endpoint](#search-only-endpoint) (`https://mcp.firecrawl.dev/v2/mcp-search`) exposes a fixed 6 read-only tools. +This server lists 25 tools when the full profile registers with default settings (feedback tools included, not running in local-keyless mode). Setting `FIRECRAWL_NO_SEARCH_FEEDBACK=1` and/or `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` removes the corresponding authenticated feedback tools and reduces this count, as does local keyless startup. For clients with a tool-slot limit: the hosted keyless endpoint (`https://mcp.firecrawl.dev/v2/mcp`, no API key) exposes 4 tools: `firecrawl_scrape`, `firecrawl_search`, `firecrawl_parse`, and `firecrawl_feedback`. The dedicated [search-only endpoint](#search-only-endpoint) (`https://mcp.firecrawl.dev/v2/mcp-search`) exposes a fixed 6 read-only tools. ## Installation @@ -574,7 +574,7 @@ job's authentication; adding credentials does not convert a keyless job. Keep feedback concise: use issue codes, tags, short notes, URLs, page numbers, and small metadata objects. Do not include raw scrape/parse outputs. -**Opt out:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) in the environment when starting the MCP server. The `firecrawl_feedback` tool will not be registered, so agents cannot call it. Operation requests also send `x-firecrawl-no-feedback: 1` so the API does not issue or count invitations that this server suppresses. +**Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API controls invitation frequency and eligibility. Submitting feedback remains optional and is never required for continued keyless access. **Authenticated usage example:** diff --git a/src/index.ts b/src/index.ts index 026a7d77..f9405a31 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1483,12 +1483,7 @@ function createClient(apiKey?: string): FirecrawlApp { config.apiKey = apiKey; } - const client = new FirecrawlApp(config); - Object.assign( - (client as any).http.instance.defaults.headers.common, - feedbackPreferenceHeaders() - ); - return client; + return new FirecrawlApp(config); } const ORIGIN = 'mcp-fastmcp'; @@ -1917,7 +1912,6 @@ async function apiPostJson( headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${apiKey}`, - ...feedbackPreferenceHeaders(), }, body: JSON.stringify(body), }); @@ -1936,7 +1930,7 @@ async function apiPostJson( response.status ); } - return filterFeedbackInvitation(parsed); + return parsed; } async function apiPostJsonForSession( @@ -2111,7 +2105,7 @@ server.addTool({ destructiveHint: false, // Does not modify, delete, or write to external websites. }, description: ` -Optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. +For keyless jobs, optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. Retrieve and extract content from one supplied URL through Firecrawl. Use this when the request identifies a page and needs its content or defined fields. It can return markdown, HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema; JSON is useful when the requested result has defined fields, while markdown preserves readable page content. @@ -2153,7 +2147,7 @@ Returns the selected content formats and page metadata. ...cleaned, origin: ORIGIN, } as any); - return asText(filterFeedbackInvitation(res)); + return asText(res); }, }); @@ -2203,7 +2197,7 @@ server.addTool({ destructiveHint: false, // Query-only; no destructive side effects on external entities. }, description: ` -Optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. +For keyless jobs, optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. Search web, news, or image sources and return ranked results. Operators include quoted phrases, \`-term\`, \`site:host\`, \`inurl:term\`, \`intitle:term\`, and \`related:host\`; the set is non-exhaustive. \`includeDomains\` and \`excludeDomains\` are mutually exclusive hostname filters; categories limit results to GitHub, research, PDF, or developer sources. @@ -2211,7 +2205,7 @@ For a programming question, add \`categories: ["developer"]\`. It searches an in \`categories: ["research"]\` restricts these web results to research-affiliated websites and returns page snippets. The \`firecrawl_research_*\` tools are a separate surface that searches paper abstracts and full text across biomedical (PubMed, bioRxiv, medRxiv) and arXiv literature. -Each web result is a title, URL, and description, not the page. Add \`scrapeOptions\` to attach page content in the same call; those fetches ignore \`maxAge\`, so use \`firecrawl_scrape\` when you need a live fetch. Returns source-type result groups and usage metadata. Responses can include an \`id\` for optional feedback through \`firecrawl_feedback\`. +Each web result is a title, URL, and description, not the page. Add \`scrapeOptions\` to attach page content in the same call; those fetches ignore \`maxAge\`, so use \`firecrawl_scrape\` when you need a live fetch. Returns source-type result groups and usage metadata. Keyless responses can include an \`id\` for optional feedback through \`firecrawl_feedback\`. Authenticated Search feedback continues to use \`firecrawl_search_feedback\`. `, parameters: z .object({ @@ -2259,7 +2253,7 @@ Each web result is a title, URL, and description, not the page. Add \`scrapeOpti // supports the optional authenticated `firecrawl_search_feedback` workflow. const client = getClient(session); const httpRes = await (client as any).http.post('/v2/search', searchBody); - return asText(filterFeedbackInvitation(httpRes?.data ?? {})); + return asText(httpRes?.data ?? {}); }, }); @@ -2338,24 +2332,6 @@ function isKeylessMode(session?: SessionData): boolean { return !process.env.FIRECRAWL_API_URL; } -function filterFeedbackInvitation(json: any): any { - if (!ENDPOINT_FEEDBACK_DISABLED || !json || typeof json !== 'object') - return json; - const omitInvitation = (metadata: any) => { - if (!metadata || typeof metadata !== 'object') return metadata; - const rest = { ...metadata }; - delete rest.feedback; - return rest; - }; - return { - ...json, - ...(json.metadata ? { metadata: omitInvitation(json.metadata) } : {}), - ...(json.data?.metadata - ? { data: { ...json.data, metadata: omitInvitation(json.data.metadata) } } - : {}), - }; -} - async function keylessPost( path: string, body: Record, @@ -2380,7 +2356,6 @@ async function keylessPost( const headers: Record = { ...ORIGIN_HEADERS, 'Content-Type': 'application/json', - ...feedbackPreferenceHeaders(), }; // Forward the real client IP (secret-authenticated) when proxying keyless // requests through the hosted MCP, so the API rate-limits per real IP. @@ -2393,9 +2368,7 @@ async function keylessPost( headers, body: JSON.stringify(body), }); - const json: any = filterFeedbackInvitation( - await response.json().catch(() => ({})) - ); + const json: any = await response.json().catch(() => ({})); if (!response.ok) { if (isKeylessMode(session) && response.status === 429) { // The API normally supplies requests|credits. Preserve a structured, @@ -2542,12 +2515,6 @@ const ENDPOINT_FEEDBACK_DISABLED = feedbackEnvEnabled( 'FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK' ); -function feedbackPreferenceHeaders(): Record { - return ENDPOINT_FEEDBACK_DISABLED - ? { 'x-firecrawl-no-feedback': '1' } - : {}; -} - if (SEARCH_FEEDBACK_DISABLED) { console.error( '[firecrawl-mcp] Search feedback tool disabled by FIRECRAWL_NO_SEARCH_FEEDBACK; firecrawl_search_feedback will not be registered.' @@ -2692,13 +2659,19 @@ Eligibility is limited to successful searches within the feedback age window. Th if (ENDPOINT_FEEDBACK_DISABLED) { console.error( - '[firecrawl-mcp] Endpoint feedback tool disabled by FIRECRAWL_NO_ENDPOINT_FEEDBACK; firecrawl_feedback will not be registered.' + '[firecrawl-mcp] Authenticated endpoint feedback tool disabled by FIRECRAWL_NO_ENDPOINT_FEEDBACK. Keyless feedback remains available.' ); } -if (!ENDPOINT_FEEDBACK_DISABLED) { +if ( + !ENDPOINT_FEEDBACK_DISABLED || + isLocalKeylessStartup() || + process.env.CLOUD_SERVICE === 'true' +) { server.addTool({ name: 'firecrawl_feedback', + canList: (session: SessionData) => + !ENDPOINT_FEEDBACK_DISABLED || isKeylessMode(session), annotations: { title: 'Send feedback on a Firecrawl job', readOnlyHint: false, // POSTs structured feedback for a completed job to /v2/feedback. @@ -2734,6 +2707,11 @@ Use only evidence already available. Do not guess missing content, diagnose caus metadata: z.record(z.string(), z.unknown()).optional(), }), execute: async (args: unknown, { session, log }): Promise => { + if (ENDPOINT_FEEDBACK_DISABLED && !isKeylessMode(session)) { + throw new UserError( + 'Endpoint feedback is disabled for authenticated sessions.' + ); + } const { endpoint, jobId, @@ -3192,7 +3170,7 @@ server.addTool({ destructiveHint: false, // Read-only parsing; no deletion or writes to the source file. }, description: ` -Optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. +For keyless jobs, optional feedback is available through \`firecrawl_feedback\` using the returned jobId or Search id. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. Parse one supported document into markdown, HTML, links, summary, targeted answers, or JSON matching a schema. Supported inputs include common HTML, PDF, Word, RTF, OpenDocument, and spreadsheet files; PDF parsing can be bounded with \`pdfOptions.maxPages\`. @@ -3238,7 +3216,6 @@ Set \`redactPII\` to request redaction of personally identifiable information in const headers: Record = { ...ORIGIN_HEADERS, - ...feedbackPreferenceHeaders(), }; const credential = credentialForOutboundRequest(session); if (credential) { @@ -3261,7 +3238,7 @@ Set \`redactPII\` to request redaction of personally identifiable information in const responseText = await response.text(); let result: any; try { - result = filterFeedbackInvitation(JSON.parse(responseText)); + result = JSON.parse(responseText); } catch { result = undefined; } @@ -3363,7 +3340,7 @@ Returns \`{ success, data, id, creditsUsed }\`, with source arrays in \`data\`. log.info('Searching', { query: searchQuery }); const client = getClientFn(session); const httpRes = await (client as any).http.post('/v2/search', searchBody); - return asText(filterFeedbackInvitation(httpRes?.data ?? {})); + return asText(httpRes?.data ?? {}); }, }); } diff --git a/tests/mcp-search-profile.test.mjs b/tests/mcp-search-profile.test.mjs index c6a01558..a0371459 100644 --- a/tests/mcp-search-profile.test.mjs +++ b/tests/mcp-search-profile.test.mjs @@ -1088,7 +1088,7 @@ test('companion telemetry follows credential precedence without resolving API ke assert.doesNotMatch(getStdout(), /fc-primary-credential|fco_secondary-credential/); }); -test('search profile honors feedback opt-out and preserves its job reference', async (t) => { +test('search profile leaves operation requests and metadata unchanged by feedback flags', async (t) => { const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { message: 'Optional feedback.' } }; const backend = await startFakeBackend({ searchMetadata: metadata }); t.after(() => backend.close()); @@ -1097,7 +1097,7 @@ test('search profile honors feedback opt-out and preserves its job reference', a params: { name: 'firecrawl_search', arguments: { query: 'retry behavior' } } }); const message = parseSseJson(await response.text()); assert.notEqual(message.result.isError, true); - assert.deepEqual(JSON.parse(message.result.content[0].text).metadata, { jobId: metadata.jobId }); + assert.deepEqual(JSON.parse(message.result.content[0].text).metadata, metadata); const call = backend.requests.find(req => req.url === '/v2/search'); - assert.equal(call.headers['x-firecrawl-no-feedback'], '1'); + assert.equal(call.headers['x-firecrawl-no-feedback'], undefined); }); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 91d6ebba..32b03fc3 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1003,6 +1003,7 @@ test('stdio transport initializes and lists Firecrawl tools', async (t) => { test('local keyless stdio keeps profile guidance keyless-scoped and exposes shared feedback', async (t) => { const child = spawnServer({ + FIRECRAWL_NO_ENDPOINT_FEEDBACK: '1', FIRECRAWL_API_KEY: '', FIRECRAWL_API_URL: '', FIRECRAWL_OAUTH_TOKEN: '', @@ -3070,6 +3071,7 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv t.after(() => backend.close()); const port = await getFreePort(); const child = spawnServer({ + FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK: '1', CLOUD_SERVICE: 'true', FASTMCP_ENDPOINT: '/v2/mcp', FIRECRAWL_API_URL: backend.url, @@ -3135,7 +3137,7 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv }); for (const disabled of [false, true]) { - test(`keyless Search preserves references and forwards invitation opt-out: ${disabled}`, async (t) => { + test(`keyless Search preserves references and retains invitations regardless of client feedback flags: ${disabled}`, async (t) => { const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { @@ -3164,6 +3166,32 @@ for (const disabled of [false, true]) { }); t.after(() => stopChild(child)); await waitForHealth(port, child); + for (const authenticated of [false, true]) { + const listed = await fetch(`http://127.0.0.1:${port}/v2/mcp`, { + method: 'POST', + headers: { + accept: 'application/json, text/event-stream', + 'content-type': 'application/json', + 'x-forwarded-for': '203.0.113.71', + ...(authenticated ? { Authorization: 'Bearer fc-feedback-test' } : {}), + }, + body: JSON.stringify({ id: 'feedback-tools', jsonrpc: '2.0', method: 'tools/list', params: {} }), + }); + const names = parseSseJson(await listed.text()).result.tools.map(tool => tool.name); + assert.equal(names.includes('firecrawl_feedback'), !authenticated || !disabled); + assert.equal(names.includes('firecrawl_search_feedback'), authenticated); + } + if (disabled) { + const blocked = await httpToolCall(port, { + id: 'disabled-authenticated-feedback', + headers: { Authorization: 'Bearer fc-feedback-test' }, + params: { name: 'firecrawl_feedback', arguments: { + endpoint: 'search', jobId: metadata.jobId, rating: 'good', + } }, + }); + assert.equal(parseSseJson(await blocked.text()).result.isError, true); + assert.equal(backend.requests.some(req => req.url === '/v2/feedback'), false); + } const response = await httpToolCall(port, { id: 'feedback-invitation', headers: { 'x-forwarded-for': '203.0.113.71' }, @@ -3175,17 +3203,17 @@ for (const disabled of [false, true]) { assert.deepEqual( JSON.parse(parseSseJson(await response.text()).result.content[0].text) .metadata, - disabled ? { jobId: metadata.jobId } : metadata + metadata ); const call = backend.requests.find((req) => req.url === '/v2/search'); assert.equal( call.headers['x-firecrawl-no-feedback'], - disabled ? '1' : undefined + undefined ); }); } -test('local Parse preserves job evidence on success and failure and respects invitation opt-out', async (t) => { +test('local Parse preserves job evidence on success and failure and retains invitations regardless of client feedback flags', async (t) => { const directory = await mkdtemp(join(tmpdir(), 'feedback-parse-')); t.after(() => rm(directory, { recursive: true, force: true })); const filePath = join(directory, 'fixture.html'); @@ -3233,14 +3261,14 @@ test('local Parse preserves job evidence on success and failure and respects inv result.structuredContent ?? JSON.parse(result.content[0].text); const returnedMetadata = payload.data?.metadata ?? payload.metadata; assert.equal(returnedMetadata.jobId, metadata.jobId); - assert.equal(Boolean(returnedMetadata.feedback), !disabled); + assert.equal(Boolean(returnedMetadata.feedback), true); assert.equal(result.isError === true, status !== 200); if (status !== 200) assert.equal(result.content[0].text, 'Parsing failed'); const call = backend.requests.find((req) => req.url === '/v2/parse'); assert.equal(call.headers.authorization, undefined); assert.equal( call.headers['x-firecrawl-no-feedback'], - disabled ? '1' : undefined + undefined ); assert.match(call.headers['content-type'], /multipart\/form-data/); assert.match(call.raw, /Parsed fixture/); @@ -3323,7 +3351,7 @@ test('local Parse defaults to the cloud API with and without credentials', async for (const endpoint of ['search', 'scrape', 'parse']) { for (const disabled of [false, true]) { - test(`authenticated ${endpoint} preserves references and honors feedback opt-out: ${disabled}`, async (t) => { + test(`authenticated ${endpoint} preserves references and leaves response metadata unchanged by feedback flags: ${disabled}`, async (t) => { const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { endpoint, message: 'Optional feedback.' } }; const body = endpoint === 'search' ? { success: true, id: metadata.jobId, data: { web: [] }, metadata } @@ -3342,10 +3370,10 @@ for (const endpoint of ['search', 'scrape', 'parse']) { const result = parseSseJson(await response.text()).result; assert.notEqual(result.isError, true); const payload = JSON.parse(result.content[0].text); - assert.deepEqual(payload.metadata ?? payload.data?.metadata, disabled ? { jobId: metadata.jobId } : metadata); + assert.deepEqual(payload.metadata ?? payload.data?.metadata, metadata); const call = backend.requests.find(req => req.url === `/v2/${endpoint}`); assert.equal(call.headers.authorization, 'Bearer fc-feedback-test'); - assert.equal(call.headers['x-firecrawl-no-feedback'], disabled ? '1' : undefined); + assert.equal(call.headers['x-firecrawl-no-feedback'], undefined); }); } } From e0533f5c4983e30861852b00c139da2044e79ae9 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Sat, 12 Sep 2026 12:52:37 -0500 Subject: [PATCH 10/25] fix(mcp): require explicit API configuration for local parse --- README.md | 4 +-- src/index.ts | 9 +++-- tests/mcp-smoke.test.mjs | 72 +++++++++++++++++----------------------- 3 files changed, 39 insertions(+), 46 deletions(-) diff --git a/README.md b/README.md index c5abf006..e0d5ec16 100644 --- a/README.md +++ b/README.md @@ -667,11 +667,11 @@ Check the status and results of an existing crawl job by ID. Parse local files or hosted upload references with Firecrawl's `/v2/parse` endpoint. -**Best for:** PDFs, Word documents, spreadsheets, HTML files, and other documents that need markdown or structured JSON output. Hosted MCP supports a two-step upload-ref flow; local MCP reads the requested file and uploads it to the configured API, defaulting to the Firecrawl cloud API. +**Best for:** PDFs, Word documents, spreadsheets, HTML files, and other documents that need markdown or structured JSON output. Hosted MCP supports a two-step upload-ref flow; local MCP requires an explicit API URL before reading and uploading the requested file. **Not recommended for:** Remote URLs (use scrape), multiple files in one call (call parse once per file), or browser-only actions such as screenshots and clicks. -**Hosted MCP flow:** Hosted MCP cannot read the caller's filesystem directly. Call `firecrawl_parse` with `filePath` to receive a short-lived upload command and `nextToolCall`, upload the file locally, then call `firecrawl_parse` again with the returned `uploadRef`. Minting the hosted upload URL requires Firecrawl auth or keyless eligibility. In local `npx firecrawl-mcp` mode, the server uploads `filePath` to `FIRECRAWL_API_URL` when configured, or to `https://api.firecrawl.dev` when unset, matching Search and Scrape. Both authenticated and eligible keyless calls are supported. Running MCP locally does not perform parsing on the local machine; configure your self-hosted API URL if that is where files should be processed. +**Hosted MCP flow:** Hosted MCP cannot read the caller's filesystem directly. Call `firecrawl_parse` with `filePath` to receive a short-lived upload command and `nextToolCall`, upload the file locally, then call `firecrawl_parse` again with the returned `uploadRef`. Minting the hosted upload URL requires Firecrawl auth or keyless eligibility. In local `npx firecrawl-mcp` mode, `FIRECRAWL_API_URL` must be explicitly configured before the server reads or uploads `filePath`. There is no default upload destination for local Parse. Both authenticated and eligible keyless calls are supported by the selected API. Running MCP locally does not perform parsing on the local machine. This configuration requirement selects the destination; it does not restrict which files the process can read. **Usage Example:** diff --git a/src/index.ts b/src/index.ts index f9405a31..3d8ac83d 100644 --- a/src/index.ts +++ b/src/index.ts @@ -3174,7 +3174,7 @@ For keyless jobs, optional feedback is available through \`firecrawl_feedback\` Parse one supported document into markdown, HTML, links, summary, targeted answers, or JSON matching a schema. Supported inputs include common HTML, PDF, Word, RTF, OpenDocument, and spreadsheet files; PDF parsing can be bounded with \`pdfOptions.maxPages\`. -Local MCP reads \`filePath\` from the server filesystem and uploads it to the configured \`FIRECRAWL_API_URL\`, or the Firecrawl cloud API when unset. Parsing happens on that API server. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. +Local MCP requires an explicitly configured \`FIRECRAWL_API_URL\` before reading \`filePath\` from the server filesystem and uploading it. Parsing happens on that API server. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. Set \`redactPII\` to request redaction of personally identifiable information in the returned content. \`zeroDataRetention\` requires an eligible authenticated account; omit it for anonymous keyless use. Returns upload instructions for hosted phase one or parsed document content for the final call. `, @@ -3184,7 +3184,12 @@ Set \`redactPII\` to request redaction of personally identifiable information in return executeHostedParse(args as ParseToolArgs, session, log); } - const apiUrl = resolveApiBaseUrl(); + const apiUrl = process.env.FIRECRAWL_API_URL; + if (!apiUrl) { + throw new UserError( + 'Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files.' + ); + } const { filePath, diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 32b03fc3..78e24d0e 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -4,7 +4,7 @@ import { createServer } from 'node:http'; import net from 'node:net'; import { mkdtemp, writeFile, rm } from 'node:fs/promises'; import { tmpdir } from 'node:os'; -import { join } from 'node:path'; +import { join, relative } from 'node:path'; import test from 'node:test'; import { setTimeout as delay } from 'node:timers/promises'; import { assertAgentMetadataPolicy } from '../scripts/agent-metadata-policy.mjs'; @@ -3278,31 +3278,36 @@ test('local Parse preserves job evidence on success and failure and retains invi }); -test('local Parse defaults to the cloud API with and without credentials', async (t) => { - const directory = await mkdtemp(join(tmpdir(), 'parse-default-api-')); +test('local Parse requires an explicit API URL before reading or uploading files', async (t) => { + const directory = await mkdtemp(join(tmpdir(), 'parse-api-configuration-')); t.after(() => rm(directory, { recursive: true, force: true })); const filePath = join(directory, 'fixture.html'); - await writeFile(filePath, '

Parsed fixture

'); + await writeFile(filePath, '

Controlled parse fixture

'); for (const authenticated of [false, true]) { await t.test(`authenticated ${authenticated}`, async (t) => { const backend = await startFakeFirecrawlBackend(); t.after(() => backend.close()); - const preloadPath = join(directory, `fetch-${authenticated}.mjs`); - await writeFile( - preloadPath, - ` + const preloadPath = join(directory, `guard-${authenticated}.mjs`); + await writeFile(preloadPath, ` + import fs from 'node:fs/promises'; + import { syncBuiltinESMExports } from 'node:module'; + import { resolve } from 'node:path'; + const originalReadFile = fs.readFile; + fs.readFile = (...args) => { + if (typeof args[0] === 'string' && resolve(args[0]) === ${JSON.stringify(filePath)}) { + throw new Error('Test detected an unconfigured local file read'); + } + return originalReadFile(...args); + }; + syncBuiltinESMExports(); const originalFetch = globalThis.fetch; globalThis.fetch = (url, init) => { if (url !== 'https://api.firecrawl.dev/v2/parse') { throw new Error('Unexpected outbound URL'); } - return originalFetch(${JSON.stringify(`${backend.url}/v2/parse`)}, { - ...init, - headers: { ...init.headers, 'x-test-original-url': url }, - }); + return originalFetch(${JSON.stringify(`${backend.url}/v2/parse`)}, init); }; - ` - ); + `); const child = spawnServer({ FIRECRAWL_API_KEY: authenticated ? 'fc-parse-test' : '', FIRECRAWL_OAUTH_TOKEN: '', @@ -3314,37 +3319,20 @@ test('local Parse defaults to the cloud API with and without credentials', async const client = new StdioMcpClient(child); await client.request('initialize', { capabilities: {}, - clientInfo: { name: 'parse-default-api-test', version: '0.0.0' }, + clientInfo: { name: 'parse-api-configuration-test', version: '0.0.0' }, protocolVersion: '2025-06-18', }); client.notify('notifications/initialized'); - const tools = await client.request('tools/list'); - const parse = tools.tools.find((tool) => tool.name === 'firecrawl_parse'); - assert.ok(parse); - assert.match( - parse.description, - /uploads it.*Firecrawl cloud API when unset/ - ); - const result = await client.request('tools/call', { - name: 'firecrawl_parse', - arguments: { filePath, formats: ['markdown'] }, - }); - assert.notEqual(result.isError, true); - assert.equal( - JSON.parse(result.content[0].text).data.markdown, - '# Parsed fixture' - ); - assert.equal(backend.requests.length, 1); - const call = backend.requests[0]; - assert.equal( - call.headers['x-test-original-url'], - 'https://api.firecrawl.dev/v2/parse' - ); - assert.equal( - call.headers.authorization, - authenticated ? 'Bearer fc-parse-test' : undefined - ); - assert.match(call.raw, /Parsed fixture/); + for (const candidate of [filePath, relative(process.cwd(), filePath), join(directory, 'missing.html')]) { + const result = await client.request('tools/call', { + name: 'firecrawl_parse', + arguments: { filePath: candidate, formats: ['markdown'] }, + }); + assert.equal(result.isError, true); + assert.match(result.content[0].text, /requires.*FIRECRAWL_API_URL/); + assert.doesNotMatch(result.content[0].text, /unconfigured local file read|ENOENT/); + } + assert.equal(backend.requests.length, 0); }); } }); From 3f4fba7a5a0194f12904de548a54ecc1895b7d56 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Sat, 12 Sep 2026 13:49:07 -0500 Subject: [PATCH 11/25] fix(mcp): clarify local parse configuration in keyless guidance --- src/index.ts | 7 ++++--- tests/mcp-smoke.test.mjs | 2 +- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/src/index.ts b/src/index.ts index 3d8ac83d..815b2638 100644 --- a/src/index.ts +++ b/src/index.ts @@ -660,8 +660,9 @@ async function authenticateRequest( // Without credentials or a custom API URL, use the cloud keyless tools. console.error( 'No FIRECRAWL_API_KEY or FIRECRAWL_API_URL set. Running in keyless mode. ' + - 'firecrawl_scrape, firecrawl_search, and firecrawl_parse use the Firecrawl cloud with usage limits. ' + - 'Local Parse uploads the requested file to the API. Optional firecrawl_feedback is available when enabled. ' + + 'firecrawl_scrape and firecrawl_search use the Firecrawl cloud with usage limits. ' + + 'Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. ' + + 'Optional firecrawl_feedback is available when enabled. ' + 'Other tools require an API key (get one free at https://firecrawl.dev).' ); } @@ -953,7 +954,7 @@ const openAiAppsChallengeToken = normalizeHeader( ); const FULL_PROFILE_INSTRUCTIONS = `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question — code behaviour, a library or framework, an API contract, an error message, or a known bug — firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of repositories, GitHub issues, merged pull requests, READMEs, and curated documentation sites. Provide only the required inputs and account for stated network or external side effects.`; -const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_feedback accepts optional evidence about those jobs without consuming operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse uploads local file bytes to the API in local MCP and uses a two-phase upload flow in hosted MCP. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, and firecrawl_research_* for paper-index and repository research.`; +const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_feedback accepts optional evidence about those jobs without consuming operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. In local MCP, firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files; parsing happens on that API server. Hosted MCP uses a two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, and firecrawl_research_* for paper-index and repository research.`; // The search surface exposes web/developer/research search only. Its instructions // and tool copy describe just those tools and stay neutral about how a client diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 78e24d0e..759089d9 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1036,7 +1036,7 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar assert.match(keylessGuidance, /firecrawl_scrape retrieves one supplied page/i); assert.match( keylessGuidance, - /firecrawl_parse uploads local file bytes to the API in local MCP and uses a two-phase upload flow in hosted MCP/i + /In local MCP, firecrawl_parse requires FIRECRAWL_API_URL.*before reading or uploading files.*Hosted MCP uses a two-phase upload flow/i ); assert.doesNotMatch( keylessGuidance, From fd3696e858a386341df140916d537d63e71ef26b Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Sat, 12 Sep 2026 17:40:32 -0500 Subject: [PATCH 12/25] feat(mcp): align keyless feedback observation guidance --- README.md | 18 ++++++++++++------ src/index.ts | 19 +++++++++++++++++-- tests/mcp-smoke.test.mjs | 21 +++++++++++++++++++++ 3 files changed, 50 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index e0d5ec16..0b8f38b1 100644 --- a/README.md +++ b/README.md @@ -554,12 +554,8 @@ jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 `source_comparison`, or `expectation`. Source comparisons also require `comparison: {reference, detail}`. -For Search, report useful or irrelevant results by source group and one-based -position, or missing information with a topic and any already-known source URLs. -For Scrape, report correct, missing, incorrect, or failed output. For Parse, -report correct output or text, table, layout, or completeness issues. The tool -description lists the complete fields. Use only available evidence and keep -unverified expectations distinct from source comparisons. +The tool description lists the category fields below. Use only available evidence +and keep unverified expectations distinct from source comparisons. Keyless submissions are limited to one per identity per UTC day across Search, Scrape, Parse, and all clients, with references valid for 24 hours. Feedback remains @@ -574,6 +570,16 @@ job's authentication; adding credentials does not convert a keyless job. Keep feedback concise: use issue codes, tags, short notes, URLs, page numbers, and small metadata objects. Do not include raw scrape/parse outputs. +**Keyless observation fields** + +Search: useful and irrelevant require a one-based position within the delivered group. source is web, images, or news; required for jobs requesting multiple sources, otherwise defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). + +Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. + +Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. + +Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. + **Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API controls invitation frequency and eligibility. Submitting feedback remains optional and is never required for continued keyless access. **Authenticated usage example:** diff --git a/src/index.ts b/src/index.ts index 815b2638..1b5374ce 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2682,9 +2682,17 @@ if ( description: ` Submit optional quality feedback for a search, scrape, parse, or map job. Authenticated callers can provide issue codes or small contextual fields with the endpoint, job ID, and rating. Omit large page contents and raw outputs. -Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. Search kinds are useful or irrelevant with source (web, images, news) and one-based position within that delivered group, or missing with topic and optional knownSources URLs. Scrape kinds are correct, missing, incorrect, or failure, with optional location and already-observed retryOutcome. Parse kinds are correct, text, table, layout, or completeness, with optional location. +Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. -Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. A failed scrape can be reported with its jobId and observed failure. One accepted submission per keyless identity per UTC day, shared across Search, Scrape, Parse, and all clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. +Search: useful and irrelevant require a one-based position within the delivered group. source is web, images, or news; required for jobs requesting multiple sources, otherwise defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). + +Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. + +Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. + +Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. + +Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. One accepted submission per keyless identity per UTC day, shared across Search, Scrape, Parse, and all clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, parameters: z.object({ endpoint: z.enum(['search', 'scrape', 'parse', 'map']), @@ -2692,6 +2700,10 @@ Use only evidence already available. Do not guess missing content, diagnose caus rating: z.enum(['good', 'bad', 'partial']), task: z.string().min(10).max(2000).optional(), assessment: z.string().min(10).max(2000).optional(), + docClass: z + .enum(['born_digital', 'scanned', 'mixed', 'unknown']) + .optional() + .describe('Required once per keyless Parse submission.'), observations: z .array(z.record(z.string(), z.unknown())) .min(1) @@ -2719,6 +2731,7 @@ Use only evidence already available. Do not guess missing content, diagnose caus rating, task, assessment, + docClass, observations, issues, tags, @@ -2735,6 +2748,7 @@ Use only evidence already available. Do not guess missing content, diagnose caus rating: 'good' | 'bad' | 'partial'; task?: string; assessment?: string; + docClass?: 'born_digital' | 'scanned' | 'mixed' | 'unknown'; observations?: Record[]; issues?: string[]; tags?: string[]; @@ -2774,6 +2788,7 @@ Use only evidence already available. Do not guess missing content, diagnose caus rating, task, assessment, + docClass, observations, issues, tags, diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 759089d9..fc016df9 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -3128,6 +3128,27 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv 'feedback-test-secret' ); assert.equal(submission.body.observations[0].position, 1); + if (status === 200) { + for (const endpoint of ['scrape', 'parse']) { + const args = { + endpoint, + jobId: '00000000-0000-4000-8000-000000000000', + rating: 'partial', + task: 'Extract the documented retry interval', + assessment: 'The structured output omitted the retry interval.', + ...(endpoint === 'parse' ? {docClass: 'born_digital'} : {}), + observations: [{kind: 'incorrect', reason: 'missing_fields', format: 'json', + basis: 'output', detail: 'The returned object has no retry interval.', + ...(endpoint === 'parse' ? {page: 2} : {location: 'Retry section'})}], + }; + const call = await httpToolCall(port, {id: `feedback-${endpoint}`, headers: {'x-forwarded-for': '203.0.113.71'}, params: {name: 'firecrawl_feedback', arguments: args}}); + assert.equal(JSON.parse(parseSseJson(await call.text()).result.content[0].text).success, true); + const posted = backend.requests.filter(req => req.url === '/v2/feedback').at(-1).body; + assert.deepEqual(posted.observations, args.observations); + assert.equal(posted.docClass, args.docClass); + } + } + assert.equal( backend.requests.some((req) => req.url === '/v2/keyless/eligibility'), false From 5e69c4b3c43faeb9e19e17ef9d29e26fd8fd5eef Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Sat, 12 Sep 2026 17:45:00 -0500 Subject: [PATCH 13/25] docs(mcp): clarify search feedback source attribution --- README.md | 2 +- src/index.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 0b8f38b1..a9e46a5f 100644 --- a/README.md +++ b/README.md @@ -572,7 +572,7 @@ and small metadata objects. Do not include raw scrape/parse outputs. **Keyless observation fields** -Search: useful and irrelevant require a one-based position within the delivered group. source is web, images, or news; required for jobs requesting multiple sources, otherwise defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). +Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). Do not submit engine attribution; it comes from the stored category tag at that position. Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. diff --git a/src/index.ts b/src/index.ts index 1b5374ce..6edf895f 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2684,7 +2684,7 @@ Submit optional quality feedback for a search, scrape, parse, or map job. Authen Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. -Search: useful and irrelevant require a one-based position within the delivered group. source is web, images, or news; required for jobs requesting multiple sources, otherwise defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). +Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). Do not submit engine attribution; it comes from the stored category tag at that position. Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. From b630d32647bef8cd00d147670e2f4753029587be Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Sun, 13 Sep 2026 22:00:22 -0500 Subject: [PATCH 14/25] docs(mcp): align keyless feedback evidence guidance --- README.md | 8 ++++---- src/index.ts | 8 ++++---- tests/mcp-smoke.test.mjs | 10 ++++++++-- 3 files changed, 16 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index a9e46a5f..1018cdb3 100644 --- a/README.md +++ b/README.md @@ -572,13 +572,13 @@ and small metadata objects. Do not include raw scrape/parse outputs. **Keyless observation fields** -Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). Do not submit engine attribution; it comes from the stored category tag at that position. +Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution; it comes from the stored category tag at that position. -Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. +Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. -Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. +Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. incorrect applies to json and summary outputs. For text_ocr and table, include the correct text or cell values in comparison.detail when already known. Parse feedback does not automatically retain the document, extracted output, page images, or layout blocks; submitted observations and corrections are retained. -Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. +Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. **Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API controls invitation frequency and eligibility. Submitting feedback remains optional and is never required for continued keyless access. diff --git a/src/index.ts b/src/index.ts index 6edf895f..b09c3e85 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2684,13 +2684,13 @@ Submit optional quality feedback for a search, scrape, parse, or map job. Authen Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. -Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters) and knownSources (up to 20 HTTP(S) URLs). Do not submit engine attribution; it comes from the stored category tag at that position. +Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution; it comes from the stored category tag at that position. -Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. +Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. -Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. +Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. incorrect applies to json and summary outputs. For text_ocr and table, include the correct text or cell values in comparison.detail when already known. Parse feedback does not automatically retain the document, extracted output, page images, or layout blocks; submitted observations and corrections are retained. -Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. +Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. One accepted submission per keyless identity per UTC day, shared across Search, Scrape, Parse, and all clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index fc016df9..847b3ec5 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -3095,11 +3095,14 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv assessment: 'The reference explains supported retry intervals.', observations: [ { - kind: 'useful', + kind: 'irrelevant', + reason: 'aggregator_over_official', + knownSources: ['https://example.com/official'], source: 'web', position: 1, basis: 'output', - detail: 'The reference answered the retry question.', + detail: + 'The official reference should rank before this aggregator.', }, ], }, @@ -3128,6 +3131,9 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv 'feedback-test-secret' ); assert.equal(submission.body.observations[0].position, 1); + assert.deepEqual(submission.body.observations[0].knownSources, [ + 'https://example.com/official', + ]); if (status === 200) { for (const endpoint of ['scrape', 'parse']) { const args = { From 60f7e288a67b4df74665670e04a02b38c1d45228 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 17 Sep 2026 16:37:08 -0500 Subject: [PATCH 15/25] fix: align keyless feedback discovery and reason guidance --- README.md | 56 ++++++++++++++++++++++++++++++++++----- src/index.ts | 57 +++++++++++++++++++++++++++++++++------- tests/mcp-smoke.test.mjs | 30 +++++++++++++++++++++ 3 files changed, 127 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 1018cdb3..7d39e451 100644 --- a/README.md +++ b/README.md @@ -557,8 +557,8 @@ jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 The tool description lists the category fields below. Use only available evidence and keep unverified expectations distinct from source comparisons. -Keyless submissions are limited to one per identity per UTC day across Search, -Scrape, Parse, and all clients, with references valid for 24 hours. Feedback remains +By default, keyless submissions are limited to one per caller IP per UTC day across Search, +Scrape, Parse, and all clients, with references valid for 24 hours. The server invitation states the deployment allowance. Feedback remains available after operation allowance is exhausted and does not consume or restore that allowance. @@ -572,15 +572,57 @@ and small metadata objects. Do not include raw scrape/parse outputs. **Keyless observation fields** -Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution; it comes from the stored category tag at that position. +Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required for multi-source jobs. Omission defaults to web, so images-only and news-only jobs must explicitly name their source. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution. -Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. +Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. incorrect applies to json and summary outputs. For text_ocr and table, include the correct text or cell values in comparison.detail when already known. Parse feedback does not automatically retain the document, extracted output, page images, or layout blocks; submitted observations and corrections are retained. -Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. - -**Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API controls invitation frequency and eligibility. Submitting feedback remains optional and is never required for continued keyless access. +Scrape and Parse observations other than failure: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. + +Failed Search, Scrape, or Parse jobs: use kind failure with reason timeout, transport_error, proxy_error, or other. Accepted only for a failed job. Include detail and basis; do not supply position, source, format, location, or page. Parse still requires docClass (unknown is allowed). + +The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. By default, one accepted submission per caller IP per UTC day is shared across Search, Scrape, Parse, and all clients; the server invitation states the deployment allowance. Attempts, including rejected requests, are limited to 30 per minute. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. + +If the saved Search response is unavailable, otherwise valid observations are accepted and stored with metadata.unverified: true because their positions could not be checked. Job ownership and requested sources are still checked. Available results must contain every referenced position. + +Reason definitions: + +- aggregator_over_official: An intermediary was returned where the task needed an available official or primary source. +- off_topic: The result addresses a different topic from the task. +- stale: The content is outdated for the time or version the task requires. +- wrong_content_type: The destination has the wrong content type for the task, such as a discussion instead of a reference. +- snippet_misleading: The returned description misrepresents source content already inspected. +- blocked_or_paywalled: Access to the destination was observed to be blocked or require a subscription; do not infer this from its URL or snippet. +- blocked_shell: The successful response contains a bot challenge or access-blocking shell instead of the requested content. +- login_required: The successful response contains a login requirement instead of the requested content. +- paywall: The successful response contains a subscription barrier instead of the requested content. +- empty: The successful response contains no meaningful requested content. +- wrong_page: The successful response contains a different page or resource. +- wrong_locale: The response uses the wrong language or region for the task. +- partial_content: Only part of the expected content was returned, without a more specific known cause. +- dynamic_content: Content loaded by client-side rendering or interaction is missing. +- pagination: Expected content on additional pages is missing. +- main_content_stripped: Content filtering removed requested primary content. +- format_lost: Text is present, but meaningful structure such as headings, lists, or code formatting was lost. +- wrong: Returned facts or values conflict with the inspected source. +- hallucinated: The output asserts content unsupported by the inspected source. +- missing_fields: Requested fields are absent from the structured output. +- misread_chars: Characters were recognized incorrectly. +- garbled: Extracted text is corrupted or unreadable. +- missing_text: Visible source text was omitted. +- structure: Table rows, columns, or header relationships were reconstructed incorrectly. +- cells_glued: Distinct table cells were merged. +- digits: Numeric table values were recognized incorrectly. +- pages_missing: Source pages are absent from the output. +- truncated_at_max_pages: Extraction ended at the configured page limit; this does not by itself imply a parser error. +- sections_dropped: Sections within processed pages were omitted. +- timeout: The operation explicitly reported a timeout. +- transport_error: The operation explicitly reported a network, connection, or TLS failure. +- proxy_error: The operation explicitly reported a proxy failure. +- other: Another operation failure was reported; describe the returned error without guessing its cause. + +**Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API includes a pointer on every eligible keyless job response. Submitting feedback remains optional and is never required for continued keyless access. **Authenticated usage example:** diff --git a/src/index.ts b/src/index.ts index b09c3e85..a25cba7e 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2666,16 +2666,16 @@ if (ENDPOINT_FEEDBACK_DISABLED) { if ( !ENDPOINT_FEEDBACK_DISABLED || - isLocalKeylessStartup() || + !resolveCredentialFromEnv() || process.env.CLOUD_SERVICE === 'true' ) { server.addTool({ name: 'firecrawl_feedback', canList: (session: SessionData) => - !ENDPOINT_FEEDBACK_DISABLED || isKeylessMode(session), + !ENDPOINT_FEEDBACK_DISABLED || !hasCredential(session), annotations: { title: 'Send feedback on a Firecrawl job', - readOnlyHint: false, // POSTs structured feedback for a completed job to /v2/feedback. + readOnlyHint: false, // POSTs structured feedback for a job to /v2/feedback. openWorldHint: true, // Feedback is tied to jobs that processed open-web URLs. destructiveHint: false, // Additive only; submits ratings and notes, does not delete jobs or external content. }, @@ -2684,15 +2684,54 @@ Submit optional quality feedback for a search, scrape, parse, or map job. Authen Keyless Search, Scrape, and Parse feedback requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. -Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required only when the job requested multiple sources; otherwise it defaults to web. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution; it comes from the stored category tag at that position. +Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required for multi-source jobs. Omission defaults to web, so images-only and news-only jobs must explicitly name their source. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution. -Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. Hard-failed Scrape jobs receive no feedback invitation. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. +Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. incorrect applies to json and summary outputs. For text_ocr and table, include the correct text or cell values in comparison.detail when already known. Parse feedback does not automatically retain the document, extracted output, page images, or layout blocks; submitted observations and corrections are retained. -Scrape and Parse: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. - -Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. One accepted submission per keyless identity per UTC day, shared across Search, Scrape, Parse, and all clients. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. +Scrape and Parse observations other than failure: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. + +Failed Search, Scrape, or Parse jobs: use kind failure with reason timeout, transport_error, proxy_error, or other. Accepted only for a failed job. Include detail and basis; do not supply position, source, format, location, or page. Parse still requires docClass (unknown is allowed). + +If the saved Search response is unavailable, otherwise valid observations are accepted and stored with metadata.unverified: true because their positions could not be checked. Job ownership and requested sources are still checked. Available results must contain every referenced position. + +Reason definitions: +- aggregator_over_official: An intermediary was returned where the task needed an available official or primary source. +- off_topic: The result addresses a different topic from the task. +- stale: The content is outdated for the time or version the task requires. +- wrong_content_type: The destination has the wrong content type for the task, such as a discussion instead of a reference. +- snippet_misleading: The returned description misrepresents source content already inspected. +- blocked_or_paywalled: Access to the destination was observed to be blocked or require a subscription; do not infer this from its URL or snippet. +- blocked_shell: The successful response contains a bot challenge or access-blocking shell instead of the requested content. +- login_required: The successful response contains a login requirement instead of the requested content. +- paywall: The successful response contains a subscription barrier instead of the requested content. +- empty: The successful response contains no meaningful requested content. +- wrong_page: The successful response contains a different page or resource. +- wrong_locale: The response uses the wrong language or region for the task. +- partial_content: Only part of the expected content was returned, without a more specific known cause. +- dynamic_content: Content loaded by client-side rendering or interaction is missing. +- pagination: Expected content on additional pages is missing. +- main_content_stripped: Content filtering removed requested primary content. +- format_lost: Text is present, but meaningful structure such as headings, lists, or code formatting was lost. +- wrong: Returned facts or values conflict with the inspected source. +- hallucinated: The output asserts content unsupported by the inspected source. +- missing_fields: Requested fields are absent from the structured output. +- misread_chars: Characters were recognized incorrectly. +- garbled: Extracted text is corrupted or unreadable. +- missing_text: Visible source text was omitted. +- structure: Table rows, columns, or header relationships were reconstructed incorrectly. +- cells_glued: Distinct table cells were merged. +- digits: Numeric table values were recognized incorrectly. +- pages_missing: Source pages are absent from the output. +- truncated_at_max_pages: Extraction ended at the configured page limit; this does not by itself imply a parser error. +- sections_dropped: Sections within processed pages were omitted. +- timeout: The operation explicitly reported a timeout. +- transport_error: The operation explicitly reported a network, connection, or TLS failure. +- proxy_error: The operation explicitly reported a proxy failure. +- other: Another operation failure was reported; describe the returned error without guessing its cause. + +Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. By default, one accepted submission per caller IP per UTC day is shared across Search, Scrape, Parse, and all clients; the server invitation states the deployment allowance. Attempts, including rejected requests, are limited to 30 per minute. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, parameters: z.object({ endpoint: z.enum(['search', 'scrape', 'parse', 'map']), @@ -2720,7 +2759,7 @@ Use only evidence already available. Do not guess missing content, diagnose caus metadata: z.record(z.string(), z.unknown()).optional(), }), execute: async (args: unknown, { session, log }): Promise => { - if (ENDPOINT_FEEDBACK_DISABLED && !isKeylessMode(session)) { + if (ENDPOINT_FEEDBACK_DISABLED && hasCredential(session)) { throw new UserError( 'Endpoint feedback is disabled for authenticated sessions.' ); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 847b3ec5..d8ab7d98 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -3280,6 +3280,8 @@ test('local Parse preserves job evidence on success and failure and retains invi protocolVersion: '2025-06-18', }); client.notify('notifications/initialized'); + const listed = await client.request('tools/list', {}); + assert.equal(listed.tools.some(tool => tool.name === 'firecrawl_feedback'), true); const result = await client.request('tools/call', { name: 'firecrawl_parse', arguments: { filePath }, @@ -3392,3 +3394,31 @@ for (const endpoint of ['search', 'scrape', 'parse']) { }); } } + + +test('keyless Search failure preserves the feedback reference', async (t) => { + const metadata = { + jobId: '00000000-0000-4000-8000-000000000000', + feedback: { endpoint: 'search', docs: 'https://docs.firecrawl.dev/api-reference/endpoint/feedback' }, + }; + const backend = await startFakeFirecrawlBackend({ + keylessEligible: true, + searchResponse: { status: 500, body: { success: false, error: 'Search transport failed', metadata } }, + }); + t.after(() => backend.close()); + const port = await getFreePort(); + const child = spawnServer({ + CLOUD_SERVICE: 'true', FASTMCP_ENDPOINT: '/v2/mcp', + FIRECRAWL_API_URL: backend.url, FIRECRAWL_OAUTH_ISSUER: backend.url, + HTTP_STREAMABLE_SERVER: 'true', PORT: String(port), KEYLESS_PROXY_SECRET: 'feedback-test-secret', + }); + t.after(() => stopChild(child)); + await waitForHealth(port, child); + const response = await httpToolCall(port, { + id: 'failed-search-feedback', headers: { 'x-forwarded-for': '203.0.113.71' }, + params: { name: 'firecrawl_search', arguments: { query: 'retry reference' } }, + }); + const result = parseSseJson(await response.text()).result; + assert.equal(result.isError, true); + assert.deepEqual(result.structuredContent.metadata, metadata); +}); From 75a6910832034c35a5f76130e4c713e2db1e494d Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Wed, 23 Sep 2026 10:29:58 -0500 Subject: [PATCH 16/25] fix(mcp): describe keyless feedback as one submission per job Remove the daily submission rule and the exact attempt rate from the feedback tool guidance and README. Each keyless job accepts one submission, retries return the original feedback ID, and attempts are described as rate limited. Model the smoke test's 429 on the API attempt throttle response. --- README.md | 9 ++++----- src/index.ts | 2 +- tests/mcp-smoke.test.mjs | 17 +++++++++++------ 3 files changed, 16 insertions(+), 12 deletions(-) diff --git a/README.md b/README.md index 7d39e451..ced8541a 100644 --- a/README.md +++ b/README.md @@ -557,10 +557,9 @@ jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 The tool description lists the category fields below. Use only available evidence and keep unverified expectations distinct from source comparisons. -By default, keyless submissions are limited to one per caller IP per UTC day across Search, -Scrape, Parse, and all clients, with references valid for 24 hours. The server invitation states the deployment allowance. Feedback remains -available after operation allowance is exhausted and does not consume or restore -that allowance. +Keyless job references are valid for 24 hours. Each job accepts one submission; +retrying it returns the original feedback ID. Feedback remains available after +operation allowance is exhausted and does not consume or restore that allowance. Authenticated callers retain the existing issue/note fields for Search, Scrape, Parse, and Map. For authenticated Search-specific feedback, continue using @@ -582,7 +581,7 @@ Scrape and Parse observations other than failure: format must be a format type t Failed Search, Scrape, or Parse jobs: use kind failure with reason timeout, transport_error, proxy_error, or other. Accepted only for a failed job. Include detail and basis; do not supply position, source, format, location, or page. Parse still requires docClass (unknown is allowed). -The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. By default, one accepted submission per caller IP per UTC day is shared across Search, Scrape, Parse, and all clients; the server invitation states the deployment allowance. Attempts, including rejected requests, are limited to 30 per minute. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. +The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. Each job accepts one submission, and retrying returns the original feedback ID. Submission attempts are rate limited. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. If the saved Search response is unavailable, otherwise valid observations are accepted and stored with metadata.unverified: true because their positions could not be checked. Job ownership and requested sources are still checked. Available results must contain every referenced position. diff --git a/src/index.ts b/src/index.ts index fe7678f6..5692da81 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2775,7 +2775,7 @@ Reason definitions: - proxy_error: The operation explicitly reported a proxy failure. - other: Another operation failure was reported; describe the returned error without guessing its cause. -Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. By default, one accepted submission per caller IP per UTC day is shared across Search, Scrape, Parse, and all clients; the server invitation states the deployment allowance. Attempts, including rejected requests, are limited to 30 per minute. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. +Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. Each job accepts one submission, and retrying returns the original feedback ID. Submission attempts are rate limited. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, parameters: z.object({ endpoint: z.enum(['search', 'scrape', 'parse', 'map']), diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 5b21e3f2..cc03f4bc 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -3077,12 +3077,17 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv feedbackId: '00000000-0000-4000-8000-000000000001', creditsRefunded: 0, } - : { - success: false, - error: 'Feedback rejected', - feedbackErrorCode: - status === 429 ? 'DAILY_LIMIT_REACHED' : 'INVALID_BODY', - }; + : status === 429 + ? { + success: false, + error: 'Too many feedback attempts. Retry in one minute.', + retry_after_seconds: 60, + } + : { + success: false, + error: 'Feedback rejected', + feedbackErrorCode: 'INVALID_BODY', + }; const backend = await startFakeFirecrawlBackend({ keylessEligible: false, feedbackResponse: { status, body }, From a76d7a4e90d423902118d837093c6f37f39bfec8 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Wed, 23 Sep 2026 10:51:15 -0500 Subject: [PATCH 17/25] fix(mcp): describe keyless feedback as requested in startup and README copy Replace the remaining optional keyless feedback wording in the local keyless startup message and the Search tool README entry. --- README.md | 2 +- src/index.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 72fe7347..3cb887ef 100644 --- a/README.md +++ b/README.md @@ -504,7 +504,7 @@ For scientific papers, see [Research Tools](#12-research-tools-firecrawl_researc **Returns:** -- Array of search results with optional scraped content, plus an `id` field. Keyless callers can use the returned job reference and optional invitation with `firecrawl_feedback`. Authenticated callers can continue using `firecrawl_search_feedback` with its existing fields and policy. +- Array of search results with optional scraped content, plus an `id` field. Keyless callers can use the returned job reference and feedback invitation with `firecrawl_feedback`. Authenticated callers can continue using `firecrawl_search_feedback` with its existing fields and policy. **Prompt Example:** diff --git a/src/index.ts b/src/index.ts index 8c649685..1e886f8b 100644 --- a/src/index.ts +++ b/src/index.ts @@ -699,7 +699,7 @@ async function authenticateRequest( 'No FIRECRAWL_API_KEY or FIRECRAWL_API_URL set. Running in keyless mode. ' + 'firecrawl_scrape and firecrawl_search use the Firecrawl cloud with usage limits. ' + 'Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. ' + - 'Optional firecrawl_feedback is available when enabled. ' + + 'firecrawl_feedback requests evidence about keyless jobs in exchange for free keyless use. ' + 'Other tools require an API key (get one free at https://firecrawl.dev).' ); } From 47625b895b59ade91b11fd1825ec9221cd0f5a58 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Wed, 23 Sep 2026 10:58:23 -0500 Subject: [PATCH 18/25] fix(mcp): keep the authenticated Search feedback sentence Restore the authenticated Search id sentence from main next to the keyless feedback sentence so both contracts stay described. --- src/index.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/index.ts b/src/index.ts index 1e886f8b..de28a73d 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2554,7 +2554,7 @@ ${ALEXANDRIA_SEARCH_LEAD} On an authenticated session, tool matches are discovery, not executed data: execute one through \`firecrawl_scrape\` with an \`alexandria\` body, or read its full contract with \`firecrawl_find_tools\`; after an Alexandria task, call firecrawl_feedback once per website with endpoint "alexandria" (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. -For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Keyless responses include an \`id\`; free keyless use requests feedback on it through \`firecrawl_feedback\`. Authenticated Search feedback continues to use \`firecrawl_search_feedback\`. +For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. Keyless responses include an \`id\`; free keyless use requests feedback on it through \`firecrawl_feedback\`. `, parameters: z .object({ From a9574f261f0559d57658f4e925e8181fd82e8356 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Mon, 28 Sep 2026 17:09:17 -0500 Subject: [PATCH 19/25] feat(feedback): invite optional keyless feedback when a result goes wrong Replace the in-exchange framing with an optional ask tied to a trigger: if a result is wrong, incomplete, blocked, or an error. This covers the Search, Scrape and Parse descriptions, the feedback tool, the keyless instructions, the startup message and the README. The descriptions say "via" so Search stays inside the 2,048-character description cap. --- README.md | 2 +- src/index.ts | 12 ++++++------ tests/mcp-smoke.test.mjs | 2 +- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 3cb887ef..71fa3190 100644 --- a/README.md +++ b/README.md @@ -555,7 +555,7 @@ Feedback is optional. If submitting, use only observations already available aft ### 3c. Generic Feedback Tool (`firecrawl_feedback`) Sends evidence through `/v2/feedback`. Feedback on keyless Search, Scrape, and -Parse jobs is requested in exchange for free keyless access. These jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 +Parse jobs is optional and helps improve Firecrawl when a result is wrong, incomplete, blocked, or an error. These jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 `observations`. Each observation has `kind`, `detail`, and `basis`: `output`, `source_comparison`, or `expectation`. Source comparisons also require `comparison: {reference, detail}`. diff --git a/src/index.ts b/src/index.ts index de28a73d..f69a4268 100644 --- a/src/index.ts +++ b/src/index.ts @@ -699,7 +699,7 @@ async function authenticateRequest( 'No FIRECRAWL_API_KEY or FIRECRAWL_API_URL set. Running in keyless mode. ' + 'firecrawl_scrape and firecrawl_search use the Firecrawl cloud with usage limits. ' + 'Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. ' + - 'firecrawl_feedback requests evidence about keyless jobs in exchange for free keyless use. ' + + 'If a keyless result is wrong, incomplete, blocked, or an error, optional feedback via firecrawl_feedback helps improve Firecrawl. ' + 'Other tools require an API key (get one free at https://firecrawl.dev).' ); } @@ -1199,7 +1199,7 @@ const openAiAppsChallengeToken = normalizeHeader( const FULL_PROFILE_INSTRUCTIONS = `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} When a provider matches, prefer it over scraping pages if the task needs the same fields across several entities, provenance, exact figures or timestamps, or a large set of records; execute it through firecrawl_scrape with the returned contract. If web results already answer the question, use them. Before scraping more than one page for the same fields, spend one free firecrawl_find_tools call to check for a provider. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question — code behaviour, a library or framework, an API contract, an error message, or a known bug — firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Execution with firecrawl_scrape alexandria uses requestId; reuse the returned ID for retries of the same payload, never a new ID to bypass a pending or uncertain 409. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Follow the returned terms/show and terms/accept calls through firecrawl_scrape; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. Otherwise direct an organization admin to the dashboard URL. Do not repeat successful provider calls just because the client could not display their output. Provide only the required inputs and account for stated network or external side effects. ${ALEXANDRIA_FEEDBACK_GUIDANCE}`; -const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_feedback collects evidence about those jobs in exchange for free keyless access and does not consume operation quota; submit it when a response includes a feedback invitation. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. In local MCP, firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files; parsing happens on that API server. Hosted MCP uses a two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; +const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. If a result is wrong, incomplete, blocked, or an error, optional feedback via firecrawl_feedback helps improve Firecrawl; it does not consume operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. In local MCP, firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files; parsing happens on that API server. Hosted MCP uses a two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; // The search surface exposes web/developer/research search plus the two Alexandria // tools (catalogue lookup and provider execution). Its instructions @@ -2438,7 +2438,7 @@ const scrapeTool: RegisteredTool = { description: ` Scrape one URL and return its content: markdown by default, or HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema. Use it when the request identifies a page and needs its content or defined fields. Use \`firecrawl_search\` when additional web sources are needed; on an authenticated session, \`firecrawl_map\` lists a site's URLs and \`firecrawl_crawl\` collects a set of pages. -Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. Keyless responses include \`metadata.jobId\`; free keyless use requests feedback on it through \`firecrawl_feedback\`. +Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. Keyless responses include \`metadata.jobId\`; if a result is wrong, incomplete, blocked, or an error, optional feedback via \`firecrawl_feedback\` helps improve Firecrawl. On an authenticated session with Alexandria access, if you are about to scrape the same fields from several pages, first run \`firecrawl_search\` with \`sources\` unset (or \`firecrawl_find_tools\`): a matching Alexandria provider returns those fields as typed records in one call. Keyless sessions have no provider matches; scrape directly. @@ -2554,7 +2554,7 @@ ${ALEXANDRIA_SEARCH_LEAD} On an authenticated session, tool matches are discovery, not executed data: execute one through \`firecrawl_scrape\` with an \`alexandria\` body, or read its full contract with \`firecrawl_find_tools\`; after an Alexandria task, call firecrawl_feedback once per website with endpoint "alexandria" (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. -For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. Keyless responses include an \`id\`; free keyless use requests feedback on it through \`firecrawl_feedback\`. +For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. Keyless responses include an \`id\`; if a result is wrong, incomplete, blocked, or an error, optional feedback via \`firecrawl_feedback\` helps improve Firecrawl. `, parameters: z .object({ @@ -3248,7 +3248,7 @@ Submit quality feedback for a search, scrape, parse, or map job. Authenticated c For an Alexandria session, set endpoint to \`alexandria\`, omit jobId, and provide requestedWebsite (url and requestedFunctionality), rationale, and rating. Optional providerFeedback and capabilityFeedback describe gaps or errors. Capability issues: new_capability_request (requires requestedFunctionality), missing_capability, insufficient_functionality, incorrect_result, execution_error, other. Alexandria feedback has no job-age deadline and no credit refund. -Keyless Search, Scrape, and Parse feedback is requested in exchange for free keyless access and requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. +Keyless Search, Scrape, and Parse feedback is optional and helps improve Firecrawl when a result is wrong, incomplete, blocked, or an error. It requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. Each job accepts one submission, and retrying returns the original feedback ID. Submission attempts are rate limited. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. `, @@ -3788,7 +3788,7 @@ server.addTool({ destructiveHint: false, // Read-only parsing; no deletion or writes to the source file. }, description: ` -For keyless jobs, Firecrawl requests feedback through \`firecrawl_feedback\` using the returned jobId or Search id, in exchange for free keyless access. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. +Keyless responses include \`metadata.jobId\`; if a result is wrong, incomplete, blocked, or an error, optional feedback via \`firecrawl_feedback\` helps improve Firecrawl. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. Parse one supported document into markdown, HTML, links, summary, targeted answers, or JSON matching a schema. Supported inputs include common HTML, PDF, Word, RTF, OpenDocument, and spreadsheet files; PDF parsing can be bounded with \`pdfOptions.maxPages\`. diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index d02e764b..45aaab60 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1430,7 +1430,7 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar assert.ok(search); assert.match( search.description, - /Keyless responses include an `id`; free keyless use requests feedback on it through `firecrawl_feedback`/i + /Keyless responses include an `id`; if a result is wrong, incomplete, blocked, or an error, optional feedback via `firecrawl_feedback` helps improve Firecrawl\./i ); }); From c0cfa2bac851b4a85379123610b279be7a10d707 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Mon, 28 Sep 2026 17:30:23 -0500 Subject: [PATCH 20/25] refactor(feedback): keep keyless feedback out of shared tool descriptions Tool descriptions are read by every session, so authenticated agents should not carry keyless-only guidance. Keyless agents still get the ask in each keyless result's invitation, the feedback tool's description and the keyless instructions. --- src/index.ts | 6 ++---- tests/mcp-smoke.test.mjs | 7 +++---- 2 files changed, 5 insertions(+), 8 deletions(-) diff --git a/src/index.ts b/src/index.ts index f69a4268..44297ec7 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2438,7 +2438,7 @@ const scrapeTool: RegisteredTool = { description: ` Scrape one URL and return its content: markdown by default, or HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema. Use it when the request identifies a page and needs its content or defined fields. Use \`firecrawl_search\` when additional web sources are needed; on an authenticated session, \`firecrawl_map\` lists a site's URLs and \`firecrawl_crawl\` collects a set of pages. -Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. Keyless responses include \`metadata.jobId\`; if a result is wrong, incomplete, blocked, or an error, optional feedback via \`firecrawl_feedback\` helps improve Firecrawl. +Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. On an authenticated session with Alexandria access, if you are about to scrape the same fields from several pages, first run \`firecrawl_search\` with \`sources\` unset (or \`firecrawl_find_tools\`): a matching Alexandria provider returns those fields as typed records in one call. Keyless sessions have no provider matches; scrape directly. @@ -2554,7 +2554,7 @@ ${ALEXANDRIA_SEARCH_LEAD} On an authenticated session, tool matches are discovery, not executed data: execute one through \`firecrawl_scrape\` with an \`alexandria\` body, or read its full contract with \`firecrawl_find_tools\`; after an Alexandria task, call firecrawl_feedback once per website with endpoint "alexandria" (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. -For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. Keyless responses include an \`id\`; if a result is wrong, incomplete, blocked, or an error, optional feedback via \`firecrawl_feedback\` helps improve Firecrawl. +For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. `, parameters: z .object({ @@ -3788,8 +3788,6 @@ server.addTool({ destructiveHint: false, // Read-only parsing; no deletion or writes to the source file. }, description: ` -Keyless responses include \`metadata.jobId\`; if a result is wrong, incomplete, blocked, or an error, optional feedback via \`firecrawl_feedback\` helps improve Firecrawl. Invitations appear in result metadata. Report only observations already available; feedback never requires additional investigation. - Parse one supported document into markdown, HTML, links, summary, targeted answers, or JSON matching a schema. Supported inputs include common HTML, PDF, Word, RTF, OpenDocument, and spreadsheet files; PDF parsing can be bounded with \`pdfOptions.maxPages\`. Local MCP requires an explicitly configured \`FIRECRAWL_API_URL\` before reading \`filePath\` from the server filesystem and uploading it. Parsing happens on that API server. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 45aaab60..7f909cbe 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1428,10 +1428,9 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar assert.equal(toolNames.includes('firecrawl_feedback'), true); const search = tools.tools.find((tool) => tool.name === 'firecrawl_search'); assert.ok(search); - assert.match( - search.description, - /Keyless responses include an `id`; if a result is wrong, incomplete, blocked, or an error, optional feedback via `firecrawl_feedback` helps improve Firecrawl\./i - ); + // Tool descriptions are shared by every session; keyless feedback is + // pointed to in keyless results and the feedback tool instead. + assert.doesNotMatch(search.description, /Keyless responses include/i); }); test('monitor create gives queries precedence over page targets', async (t) => { From ac6553e94fe24ded2bba732f006cfaf25a87807e Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 1 Oct 2026 11:07:52 -0500 Subject: [PATCH 21/25] fix(feedback): preserve keyless invitations and optional guidance --- README.md | 8 +++--- src/index.ts | 26 +++++++++++--------- src/tool-output.ts | 5 +++- tests/mcp-alexandria-feedback-hints.test.mjs | 2 +- tests/mcp-smoke.test.mjs | 18 ++++++++------ 5 files changed, 34 insertions(+), 25 deletions(-) diff --git a/README.md b/README.md index a076aa70..839a4167 100644 --- a/README.md +++ b/README.md @@ -515,9 +515,9 @@ For scientific papers, see [Research Tools](#12-research-tools-firecrawl_researc ### 3b. Search Feedback Tool (`firecrawl_search_feedback`) -This authenticated tool sends structured feedback on a previous `firecrawl_search` result. Its request fields, eligibility, and refund policy are unchanged. It is idempotent per search ID. Keyless callers must use `firecrawl_feedback`; this legacy tool does not accept keyless jobs. +Sends structured feedback on a previous `firecrawl_search` result. The first feedback per search id refunds 1 credit and improves Firecrawl's search quality. Idempotent per search id. -Feedback is optional. If submitting, use only observations already available after processing the results; no user interview or additional investigation is required. +**Call this after every search you actually use** (or that didn't help). Bad/partial feedback with `missingContent` is just as valuable as good feedback. **Opt out:** set `FIRECRAWL_NO_SEARCH_FEEDBACK=1` (or `FIRECRAWL_DISABLE_SEARCH_FEEDBACK=1`) in the environment when starting the MCP server. The `firecrawl_search_feedback` tool will not be registered, so agents can't call it. Team admins can also disable feedback server-side; in that case the tool is registered but always returns `feedbackErrorCode: "TEAM_OPTED_OUT"`. @@ -558,7 +558,7 @@ Feedback is optional. If submitting, use only observations already available aft ### 3c. Generic Feedback Tool (`firecrawl_feedback`) Sends evidence through `/v2/feedback`. Feedback on keyless Search, Scrape, and -Parse jobs is optional and helps improve Firecrawl when a result is wrong, incomplete, blocked, or an error. These jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 +Parse jobs is optional. Consider submitting feedback through `firecrawl_feedback`, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. These jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 `observations`. Each observation has `kind`, `detail`, and `basis`: `output`, `source_comparison`, or `expectation`. Source comparisons also require `comparison: {reference, detail}`. @@ -630,7 +630,7 @@ Reason definitions: - proxy_error: The operation explicitly reported a proxy failure. - other: Another operation failure was reported; describe the returned error without guessing its cause. -**Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API includes a pointer on every eligible keyless job response. Feedback is requested in exchange for free keyless access; continued keyless access does not depend on it. +**Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API includes a pointer on every eligible keyless job response. Feedback is optional; continued keyless access does not depend on it. **Authenticated usage example:** diff --git a/src/index.ts b/src/index.ts index ee59c972..6b3874eb 100644 --- a/src/index.ts +++ b/src/index.ts @@ -787,7 +787,7 @@ async function authenticateRequest( 'No FIRECRAWL_API_KEY or FIRECRAWL_API_URL set. Running in keyless mode. ' + 'firecrawl_scrape and firecrawl_search use the Firecrawl cloud with usage limits. ' + 'Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. ' + - 'If a keyless result is wrong, incomplete, blocked, or an error, optional feedback via firecrawl_feedback helps improve Firecrawl. ' + + 'Consider submitting feedback through firecrawl_feedback, especially if a keyless result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. ' + 'Other tools require an API key (get one free at https://firecrawl.dev).' ); } @@ -2706,7 +2706,7 @@ const scrapeTool: RegisteredTool = { description: ` Scrape one URL and return its content: markdown by default, or HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema. Use it when the request identifies a page and needs its content or defined fields. Use \`firecrawl_search\` when additional web sources are needed; on an authenticated session, \`firecrawl_map\` lists a site's URLs and \`firecrawl_crawl\` collects a set of pages. -Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. +Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. Keyless results include a job reference. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. On an authenticated session with Alexandria access, \`firecrawl_search\` with \`sources\` unset and \`firecrawl_find_tools\` can discover providers for the same fields across several pages; a matching provider returns typed records in one call. Keyless sessions have no provider matches. @@ -2834,7 +2834,7 @@ ${ALEXANDRIA_SEARCH_LEAD} On an authenticated session, tool matches describe available capabilities; \`firecrawl_find_tools\` returns their contracts and \`firecrawl_scrape\` with an \`alexandria\` body executes a selected capability. Keyless sessions get no Alexandria matches in data.tools. -For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. +For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. Keyless results include a job reference. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. `, outputSchema: searchOutputSchema, parameters: z @@ -3563,13 +3563,13 @@ if ( destructiveHint: false, // Additive only; submits ratings and notes, does not delete jobs or external content. }, description: ` -Submit quality feedback for a search, scrape, parse, or map job. Authenticated callers can provide issue codes or small contextual fields with the endpoint, job ID, and rating. Omit large page contents and raw outputs. +Submit job quality feedback through /v2/feedback. Authenticated jobs use existing issue/note fields. Omit full page contents and raw outputs. For an Alexandria session, set endpoint to \`alexandria\`, omit jobId, and provide requestedWebsite (url and requestedFunctionality), objective, rationale, and rating. objective is the underlying goal behind the session: what you or your user were ultimately trying to accomplish (for example, "shortlist federal IT contracts to bid on this quarter"), not only what was needed from this website. Optional providerFeedback and capabilityFeedback describe gaps or errors. Capability issues: new_capability_request (requires requestedFunctionality), missing_capability, insufficient_functionality, incorrect_result, execution_error, other. Alexandria feedback has no job-age deadline and no credit refund. -Keyless Search, Scrape, and Parse feedback is optional and helps improve Firecrawl when a result is wrong, incomplete, blocked, or an error. It requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. +Consider submitting keyless Search, Scrape, and Parse feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. It requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. -Use only evidence already available. Do not guess missing content, diagnose causes, or investigate further. The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. Each job accepts one submission, and retrying returns the original feedback ID. Submission attempts are rate limited. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. Feedback remains available after operation quota exhaustion and does not restore quota. Returns submission status and feedback ID. Authenticated feedback retains its existing fields. +Use evidence already available; no extra investigation. The stored submission must fit within 8 KiB, including server defaults. Submit within 24 hours from the same caller IP. One submission per job; retries return the original feedback ID. Attempts are rate limited. Feedback remains available after operation quota exhaustion and does not consume or restore quota. Contract: https://docs.firecrawl.dev/api-reference/endpoint/feedback. `, outputSchema: feedbackOutputSchema, parameters: z.object({ @@ -3577,8 +3577,8 @@ Use only evidence already available. Do not guess missing content, diagnose caus jobId: z.string().uuid('jobId must be the UUID returned by Firecrawl').optional(), ...alexandriaFeedbackFields, rating: z.enum(['good', 'bad', 'partial']), - task: z.string().min(10).max(2000).optional(), - assessment: z.string().min(10).max(2000).optional(), + task: z.string().trim().min(10).max(2000).optional(), + assessment: z.string().trim().min(10).max(2000).optional(), docClass: z .enum(['born_digital', 'scanned', 'mixed', 'unknown']) .optional() @@ -3670,7 +3670,7 @@ Use only evidence already available. Do not guess missing content, diagnose caus headers['Authorization'] = `Bearer ${credential}`; } else if (isHostedKeylessSession(session)) { if (!session?.keylessClientIp || !process.env.KEYLESS_PROXY_SECRET) { - return asText({ + return structuredText({ success: false, error: 'Feedback requires a trusted client identity.', retryable: false, @@ -3737,7 +3737,11 @@ Use only evidence already available. Do not guess missing content, diagnose caus status: response.status, feedbackErrorCode: parsed?.feedbackErrorCode, error: parsed?.error ?? `HTTP ${response.status}`, - retryable: response.status >= 500, + retryable: response.status >= 500 || (!credential && response.status === 429), + ...(!credential ? { + details: parsed?.details, + retry_after_seconds: parsed?.retry_after_seconds, + } : {}), ...(readAgentHints(parsed) ? { agent_hints: readAgentHints(parsed) } : {}), @@ -4250,7 +4254,7 @@ Parse one supported document into markdown, HTML, links, summary, targeted answe Local MCP requires an explicitly configured \`FIRECRAWL_API_URL\` before reading \`filePath\` from the server filesystem and uploading it. Parsing happens on that API server. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. -Set \`redactPII\` to request redaction of personally identifiable information in the returned content. \`zeroDataRetention\` requires an eligible authenticated account; omit it for anonymous keyless use. Returns upload instructions for hosted phase one or parsed document content for the final call. Authenticated final responses can include a \`data.metadata.scrapeId\` for optional parse feedback. +Set \`redactPII\` to request redaction of personally identifiable information in the returned content. \`zeroDataRetention\` requires an eligible authenticated account; omit it for anonymous keyless use. Returns upload instructions for hosted phase one or parsed document content for the final call. Authenticated final responses can include a \`data.metadata.scrapeId\` for optional parse feedback. Keyless results include a job reference. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. `, outputSchema: parseOutputSchema, parameters: parseParamsSchema, diff --git a/src/tool-output.ts b/src/tool-output.ts index 3222882f..b4bb9fc5 100644 --- a/src/tool-output.ts +++ b/src/tool-output.ts @@ -160,7 +160,8 @@ export const searchOutputSchema = z data: unknown('Ranked results grouped by source, such as `web`, `news`, `images`, and `alexandria`.'), error, warning, - id: str('Search identifier, for optional `firecrawl_search_feedback`.'), + id: str('Search job identifier: keyless callers use `firecrawl_feedback`; authenticated callers can use `firecrawl_search_feedback`.'), + metadata: unknown('Keyless job reference and optional feedback invitation.'), creditsUsed: num('Credits this search consumed.'), tools: unknown('Domain-matched Alexandria tools for the results.'), nextTool: unknown('A follow-up tool call that continues this search.'), @@ -185,6 +186,8 @@ export const feedbackOutputSchema = z error, status: num('HTTP status when the submission was rejected.'), feedbackErrorCode: str('Machine-readable reason the submission was rejected.'), + details: unknown('Keyless request validation errors.'), + retry_after_seconds: num('Seconds to wait before retrying a throttled keyless submission.'), retryable: bool('Whether retrying the submission can succeed.'), message: str('Human-readable result of the submission.'), feedbackId: str('Identifier of the recorded feedback.'), diff --git a/tests/mcp-alexandria-feedback-hints.test.mjs b/tests/mcp-alexandria-feedback-hints.test.mjs index f808e006..914c0cca 100644 --- a/tests/mcp-alexandria-feedback-hints.test.mjs +++ b/tests/mcp-alexandria-feedback-hints.test.mjs @@ -23,7 +23,7 @@ test('Alexandria selection metadata omits feedback workflow', async (t) => { const byName = new Map(tools.map((tool) => [tool.name, tool])); assert(byName.has('firecrawl_feedback')); for (const name of ['firecrawl_scrape', 'firecrawl_find_tools', 'firecrawl_search']) { - assert.doesNotMatch(byName.get(name).description, /firecrawl_feedback|feedbackTool|Alexandria quality feedback/i, name); + assert.doesNotMatch(byName.get(name).description, /feedbackTool|Alexandria quality feedback/i, name); } }); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 5babc932..f2cdbcbf 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1561,7 +1561,7 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar const apiKeyGuidance = init.instructions.slice(apiKeyBoundaryIndex); assert.match( keylessGuidance, - /Keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits/i + /Keyless sessions expose firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits/i ); assert.match( keylessGuidance, @@ -1574,7 +1574,7 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar assert.match(keylessGuidance, /firecrawl_scrape retrieves one supplied page/i); assert.match( keylessGuidance, - /In local MCP, firecrawl_parse requires FIRECRAWL_API_URL.*before reading or uploading files.*Hosted MCP uses a two-phase upload flow/i + /Hosted firecrawl_parse.*two-phase upload flow.*Local firecrawl_parse requires FIRECRAWL_API_URL.*before reading or uploading files/i ); assert.doesNotMatch( keylessGuidance, @@ -4381,7 +4381,8 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv assert.equal(result.success, status === 200); if (status !== 200) { assert.equal(result.feedbackErrorCode, body.feedbackErrorCode); - assert.equal(result.retryable, status >= 500); + assert.equal(result.retryable, status >= 500 || status === 429); + if (status === 429) assert.equal(result.retry_after_seconds, 60); } const submission = backend.requests.find( (req) => req.url === '/v2/feedback' @@ -4493,11 +4494,10 @@ for (const disabled of [false, true]) { arguments: { query: 'retry behavior' }, }, }); - assert.deepEqual( - JSON.parse(parseSseJson(await response.text()).result.content[0].text) - .metadata, - metadata - ); + const result = parseSseJson(await response.text()).result; + assert.deepEqual(JSON.parse(result.content[0].text).metadata, metadata); + assert.deepEqual(result.structuredContent.metadata, metadata); + assert.equal(result.structuredContent.id, metadata.jobId); const call = backend.requests.find((req) => req.url === '/v2/search'); assert.equal( call.headers['x-firecrawl-no-feedback'], @@ -4687,6 +4687,8 @@ test('keyless Search failure preserves the feedback reference', async (t) => { const result = parseSseJson(await response.text()).result; assert.equal(result.isError, true); assert.deepEqual(result.structuredContent.metadata, metadata); +}); + test('every listed tool declares an output schema and returns structured content', async (t) => { const fakeApi = await startFakeFirecrawlApi(); t.after(() => fakeApi.close()); From d06a2a78a9966c4f6f25e0ddc8143284c1ceed6e Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 1 Oct 2026 11:49:19 -0500 Subject: [PATCH 22/25] fix(feedback): validate keyless evidence and preserve failed job text --- README.md | 58 ++++--------------------------- src/index.ts | 37 +++++++++++++++++--- tests/mcp-search-profile.test.mjs | 16 --------- tests/mcp-smoke.test.mjs | 53 ++++++++++++++++------------ 4 files changed, 68 insertions(+), 96 deletions(-) diff --git a/README.md b/README.md index 839a4167..02a6c1f9 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ A Model Context Protocol (MCP) server that brings [Firecrawl](https://github.com - Use `firecrawl_credit_usage` to check credits left or monthly consumption, optionally broken down by API key. - Consider something else when you need to hold a browser session open across many of your own steps with your own retry and termination logic: each `firecrawl_interact` call runs one `prompt` or `code` turn to completion and returns control — the session can persist across calls via `scrapeId` and ends with `firecrawl_interact_stop`, but you cannot drive it interactively step-by-step from the client side within a single call. -This server lists 26 tools when the full profile registers with default settings (feedback tools included, not running in local-keyless mode). Setting `FIRECRAWL_NO_SEARCH_FEEDBACK=1` and/or `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` removes the corresponding authenticated feedback tools and reduces this count, as does local keyless startup. For clients with a tool-slot limit: the hosted keyless endpoint (`https://mcp.firecrawl.dev/v2/mcp`, no API key) exposes 4 tools: `firecrawl_scrape`, `firecrawl_search`, `firecrawl_parse`, and `firecrawl_feedback`. The dedicated [search-only endpoint](#search-only-endpoint) (`https://mcp.firecrawl.dev/v2/mcp-search`) exposes a fixed set of 8 tools (search, developer and research search, plus Alexandria catalogue lookup and execution). +Authenticated sessions expose the full tool set. Setting `FIRECRAWL_NO_SEARCH_FEEDBACK=1` and/or `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` hides the corresponding authenticated feedback tools. For clients with a tool-slot limit: the hosted keyless endpoint (`https://mcp.firecrawl.dev/v2/mcp`, no API key) exposes 4 tools: `firecrawl_scrape`, `firecrawl_search`, `firecrawl_parse`, and `firecrawl_feedback`. The dedicated [search-only endpoint](#search-only-endpoint) (`https://mcp.firecrawl.dev/v2/mcp-search`) exposes a fixed set of 8 tools (search, developer and research search, plus Alexandria catalogue lookup and execution). ## Installation @@ -578,57 +578,11 @@ job's authentication; adding credentials does not convert a keyless job. Keep feedback concise: use issue codes, tags, short notes, URLs, page numbers, and small metadata objects. Do not include raw scrape/parse outputs. -**Keyless observation fields** - -Search: useful and irrelevant require a one-based position within the delivered group. source names the response group the position refers to: web, images, or news. It is required for multi-source jobs. Omission defaults to web, so images-only and news-only jobs must explicitly name their source. The position must exist in that requested group. irrelevant requires reason: aggregator_over_official, off_topic, stale, wrong_content_type, snippet_misleading, or blocked_or_paywalled. vertical is required on missing and optional on useful/irrelevant: web_general, social, business, research, developer, news, government, finance, or other. missing may include topic (up to 200 characters). missing and irrelevant may include knownSources (up to 20 HTTP(S) URLs): where absent content lives or the source that should have ranked instead. Unmentioned results are unassessed; a full ranking is not required. Do not submit engine attribution. - -Scrape: kind correct, wrong_success, incomplete, or incorrect. wrong_success requires reason: blocked_shell, login_required, paywall, empty, wrong_page, stale, or wrong_locale. incomplete requires reason: partial_content, dynamic_content, pagination, main_content_stripped, or format_lost. incorrect requires reason: wrong, hallucinated, or missing_fields. correct has no reason. Optional location is up to 200 characters. No retryOutcome. hallucinated applies only to json, deterministicJson, summary, question, highlights, and changeTracking in json mode; missing_fields applies only to json and deterministicJson. For incomplete and incorrect, prefer source_comparison when the source is already available. - -Parse: docClass is required once per submission: born_digital, scanned, mixed, or unknown. Observation kind: correct, text_ocr, table, formula, chart_figure, reading_order, headers_footers, headings_formatting, completeness, images_dropped, or incorrect. text_ocr requires reason: misread_chars, garbled, or missing_text. table requires reason: structure, cells_glued, or digits. completeness requires reason: pages_missing, truncated_at_max_pages, or sections_dropped. incorrect requires reason: wrong, hallucinated, or missing_fields. Other kinds have no reason subtype. Optional page is a one-based positive integer. incorrect applies to json and summary outputs. For text_ocr and table, include the correct text or cell values in comparison.detail when already known. Parse feedback does not automatically retain the document, extracted output, page images, or layout blocks; submitted observations and corrections are retained. - -Scrape and Parse observations other than failure: format must be a format type the job requested. It is required for output and source_comparison observations when multiple formats were requested; optional for expectation observations and single-format jobs. All observations retain detail and basis; source_comparison requires comparison: {reference, detail}. comparison.detail contains the correct content from the inspected source. - -Failed Search, Scrape, or Parse jobs: use kind failure with reason timeout, transport_error, proxy_error, or other. Accepted only for a failed job. Include detail and basis; do not supply position, source, format, location, or page. Parse still requires docClass (unknown is allowed). - -The stored keyless submission must fit within 8 KiB (8192 UTF-8 bytes), including server defaults and verification flags. Each job accepts one submission, and retrying returns the original feedback ID. Submission attempts are rate limited. Submit within 24 hours from the same caller IP. Contract and example: https://docs.firecrawl.dev/api-reference/endpoint/feedback. - -If the saved Search response is unavailable, otherwise valid observations are accepted and stored with metadata.unverified: true because their positions could not be checked. Job ownership and requested sources are still checked. Available results must contain every referenced position. - -Reason definitions: - -- aggregator_over_official: An intermediary was returned where the task needed an available official or primary source. -- off_topic: The result addresses a different topic from the task. -- stale: The content is outdated for the time or version the task requires. -- wrong_content_type: The destination has the wrong content type for the task, such as a discussion instead of a reference. -- snippet_misleading: The returned description misrepresents source content already inspected. -- blocked_or_paywalled: Access to the destination was observed to be blocked or require a subscription; do not infer this from its URL or snippet. -- blocked_shell: The successful response contains a bot challenge or access-blocking shell instead of the requested content. -- login_required: The successful response contains a login requirement instead of the requested content. -- paywall: The successful response contains a subscription barrier instead of the requested content. -- empty: The successful response contains no meaningful requested content. -- wrong_page: The successful response contains a different page or resource. -- wrong_locale: The response uses the wrong language or region for the task. -- partial_content: Only part of the expected content was returned, without a more specific known cause. -- dynamic_content: Content loaded by client-side rendering or interaction is missing. -- pagination: Expected content on additional pages is missing. -- main_content_stripped: Content filtering removed requested primary content. -- format_lost: Text is present, but meaningful structure such as headings, lists, or code formatting was lost. -- wrong: Returned facts or values conflict with the inspected source. -- hallucinated: The output asserts content unsupported by the inspected source. -- missing_fields: Requested fields are absent from the structured output. -- misread_chars: Characters were recognized incorrectly. -- garbled: Extracted text is corrupted or unreadable. -- missing_text: Visible source text was omitted. -- structure: Table rows, columns, or header relationships were reconstructed incorrectly. -- cells_glued: Distinct table cells were merged. -- digits: Numeric table values were recognized incorrectly. -- pages_missing: Source pages are absent from the output. -- truncated_at_max_pages: Extraction ended at the configured page limit; this does not by itself imply a parser error. -- sections_dropped: Sections within processed pages were omitted. -- timeout: The operation explicitly reported a timeout. -- transport_error: The operation explicitly reported a network, connection, or TLS failure. -- proxy_error: The operation explicitly reported a proxy failure. -- other: Another operation failure was reported; describe the returned error without guessing its cause. +Search observations identify delivered result positions or missing information. Scrape and Parse observations describe the requested output formats. Failed jobs use a `failure` observation based on the returned error. Parse additionally requires `docClass`: `born_digital`, `scanned`, `mixed`, or `unknown`. + +Task, assessment, and observation detail each require 10-2000 characters after trimming whitespace. Stored keyless feedback must fit within 8 KiB, including server defaults and verification flags. Submit from the same caller IP; attempts are rate limited. + +The tool's `observations` parameter lists the endpoint-specific categories, fields, and reason codes. See the [API feedback contract](https://docs.firecrawl.dev/api-reference/endpoint/feedback) for examples, format constraints, and Parse retention behavior. **Authenticated feedback preference:** set `FIRECRAWL_NO_ENDPOINT_FEEDBACK=1` (or `FIRECRAWL_DISABLE_ENDPOINT_FEEDBACK=1`) to hide `firecrawl_feedback` from authenticated sessions. Keyless sessions retain the tool and server-issued invitations regardless of these flags. The API includes a pointer on every eligible keyless job response. Feedback is optional; continued keyless access does not depend on it. diff --git a/src/index.ts b/src/index.ts index 6b3874eb..2bc07655 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1292,7 +1292,7 @@ const openAiAppsChallengeToken = normalizeHeader( const FULL_PROFILE_INSTRUCTIONS = `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. Provide only the required inputs and account for stated network or external side effects.`; -const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions expose firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. Feedback does not consume operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. Hosted firecrawl_parse processes supported local files through its two-phase upload flow. Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; +const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions can use firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. Feedback does not consume operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. Hosted firecrawl_parse processes supported local files through its two-phase upload flow. Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; // The search surface exposes web/developer/research search plus the two Alexandria // tools (catalogue lookup and provider execution). Its instructions @@ -1576,6 +1576,13 @@ async function runWithCredentialRecovery( // different fault and keeps its own reconnect guidance; a keyless session // never sent an account credential at all. if (session?.authType !== 'api-key' || !isCoreCredentialRejection(error)) { + // Job failures already expose the full API envelope in text and extras. + if (!hasCredential(session) && error instanceof UserError) { + const metadata = error.extras?.metadata; + if (metadata && typeof metadata === 'object' && 'jobId' in metadata) { + throw error; + } + } const hints = readErrorAgentHints(error); if (hints) { const message = error instanceof Error ? error.message : String(error); @@ -3172,7 +3179,7 @@ async function keylessPost( body: JSON.stringify(body), }); const json: any = await response.json().catch(() => ({})); - if (!response.ok) { + if (!response.ok || json?.success === false) { if (isKeylessMode(session) && response.status === 429) { // The API normally supplies requests|credits. Preserve a structured, // non-specific recovery payload during a skewed or legacy deployment. @@ -3193,7 +3200,7 @@ async function keylessPost( } if (json?.metadata?.jobId) { throw new UserError( - json.error || `Firecrawl request failed (HTTP ${response.status})`, + JSON.stringify(json, null, 2), json ); } @@ -3703,6 +3710,26 @@ Use evidence already available; no extra investigation. The stored submission mu origin, }); + if (isKeylessMode(session)) { + if (!['search', 'scrape', 'parse'].includes(endpoint)) { + throw new UserError( + 'Keyless feedback supports Search, Scrape, and Parse. Other endpoints require authentication.' + ); + } + const required = { + task, + assessment, + observations, + ...(endpoint === 'parse' ? { docClass } : {}), + }; + const missing = Object.entries(required) + .filter(([, value]) => value === undefined) + .map(([name]) => name); + if (missing.length) { + throw new UserError(`Keyless feedback requires ${missing.join(', ')}.`); + } + } + log.info('Submitting endpoint feedback', { endpoint, jobId, rating }); const response = await fetch(`${apiBase}/v2/feedback`, { method: 'POST', @@ -4334,10 +4361,10 @@ Set \`redactPII\` to request redaction of personally identifiable information in } catch { result = undefined; } - if (!response.ok) { + if (!response.ok || (!hasCredential(session) && result?.success === false)) { if (!hasCredential(session) && result?.metadata?.jobId) { throw new UserError( - result.error || `Parse request failed (HTTP ${response.status})`, + JSON.stringify(result, null, 2), result ); } diff --git a/tests/mcp-search-profile.test.mjs b/tests/mcp-search-profile.test.mjs index 3414c3b3..08556f19 100644 --- a/tests/mcp-search-profile.test.mjs +++ b/tests/mcp-search-profile.test.mjs @@ -130,7 +130,6 @@ async function startFakeBackend(options = {}) { const { apiKeyFromIntrospection = 'fc-from-introspection', introspectionAud, - searchMetadata, } = options; const requests = []; const server = createServer(async (req, res) => { @@ -243,7 +242,6 @@ async function startFakeBackend(options = {}) { : [{ title: 'Example Domain', url: 'https://example.com/' }], }, id: '00000000-0000-4000-8000-000000000000', - ...(searchMetadata ? { metadata: searchMetadata } : {}), success: true, }) ); @@ -1271,20 +1269,6 @@ test('companion telemetry follows credential precedence without resolving API ke assert.doesNotMatch(getStdout(), /fc-primary-credential|fco_secondary-credential/); }); -test('search profile leaves operation requests and metadata unchanged by feedback flags', async (t) => { - const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { message: 'Optional feedback.' } }; - const backend = await startFakeBackend({ searchMetadata: metadata }); - t.after(() => backend.close()); - const { searchPort } = await startHostedServer(t, { FIRECRAWL_API_URL: backend.url, FIRECRAWL_NO_ENDPOINT_FEEDBACK: '1' }); - const response = await jsonRpc(searchPort, SEARCH_ENDPOINT, { id: 1, method: 'tools/call', headers: { 'x-api-key': 'fc-test' }, - params: { name: 'firecrawl_search', arguments: { query: 'retry behavior' } } }); - const message = parseSseJson(await response.text()); - assert.notEqual(message.result.isError, true); - assert.deepEqual(JSON.parse(message.result.content[0].text).metadata, metadata); - const call = backend.requests.find(req => req.url === '/v2/search'); - assert.equal(call.headers['x-firecrawl-no-feedback'], undefined); -}); - test('search-only surface rejects catalogue browsing and preserves semantic plus contextual discovery', async (t) => { const backend = await startFakeBackend(); t.after(() => backend.close()); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index f2cdbcbf..6488285f 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -703,7 +703,7 @@ async function startFakeFirecrawlBackend(options = {}) { status: 200, body: { success: true, - feedbackId: '00000000-0000-4000-8000-000000000001', + feedbackId: '00000000-0000-4000-8000-000000000102', creditsRefunded: 0, }, }; @@ -812,18 +812,6 @@ async function startFakeFirecrawlBackend(options = {}) { return; } - if (req.method === 'POST' && req.url === '/v2/feedback') { - res.writeHead(200, { 'content-type': 'application/json' }); - res.end( - JSON.stringify({ - creditsRefunded: 0, - feedbackId: '00000000-0000-4000-8000-000000000102', - success: true, - }) - ); - return; - } - if (req.method === 'GET' && req.url?.startsWith('/v2/monitor')) { res.writeHead(200, { 'content-type': 'application/json' }); res.end(JSON.stringify({ success: true, data: [] })); @@ -1561,7 +1549,7 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar const apiKeyGuidance = init.instructions.slice(apiKeyBoundaryIndex); assert.match( keylessGuidance, - /Keyless sessions expose firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits/i + /Keyless sessions can use firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits/i ); assert.match( keylessGuidance, @@ -1600,9 +1588,9 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar assert.equal(toolNames.includes('firecrawl_feedback'), true); const search = tools.tools.find((tool) => tool.name === 'firecrawl_search'); assert.ok(search); - // Tool descriptions are shared by every session; keyless feedback is - // pointed to in keyless results and the feedback tool instead. - assert.doesNotMatch(search.description, /Keyless responses include/i); + for (const name of ['firecrawl_search', 'firecrawl_scrape', 'firecrawl_parse']) { + assert.match(tools.tools.find(tool => tool.name === name).description, /Keyless results.*Consider submitting feedback through firecrawl_feedback/); + } }); test('monitor create gives queries precedence over page targets', async (t) => { @@ -4422,6 +4410,19 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv } } + if (status === 200) { + for (const field of ['task', 'assessment', 'observations', 'docClass']) { + const args = { endpoint: 'parse', jobId: '00000000-0000-4000-8000-000000000000', rating: 'partial', task: 'Read the retry reference', assessment: 'The retry interval was omitted from the output.', observations: [{kind: 'correct', basis: 'output', detail: 'The output contains the retry heading.'}], docClass: 'unknown' }; + delete args[field]; + const before = backend.requests.length; + const response = await httpToolCall(port, { id: `missing-${field}`, headers: {'x-forwarded-for': '203.0.113.71'}, params: {name: 'firecrawl_feedback', arguments: args} }); + const result = parseSseJson(await response.text()).result; + assert.equal(result.isError, true); + assert.match(result.content[0].text, new RegExp(`requires.*${field}`)); + assert.equal(backend.requests.length, before); + } + } + assert.equal( backend.requests.some((req) => req.url === '/v2/keyless/eligibility'), false @@ -4511,9 +4512,9 @@ test('local Parse preserves job evidence on success and failure and retains invi t.after(() => rm(directory, { recursive: true, force: true })); const filePath = join(directory, 'fixture.html'); await writeFile(filePath, '

Parsed fixture

'); - for (const status of [200, 500]) { + for (const [status, success] of [[200, true], [500, false], [200, false]]) { for (const disabled of [false, true]) { - await t.test(`HTTP ${status}, disabled ${disabled}`, async (t) => { + await t.test(`HTTP ${status}, success ${success}, disabled ${disabled}`, async (t) => { const metadata = { jobId: '00000000-0000-4000-8000-000000000000', feedback: { @@ -4522,7 +4523,7 @@ test('local Parse preserves job evidence on success and failure and retains invi }, }; const body = - status === 200 + success ? { success: true, data: { markdown: '# Parsed fixture', metadata }, @@ -4557,8 +4558,13 @@ test('local Parse preserves job evidence on success and failure and retains invi const returnedMetadata = payload.data?.metadata ?? payload.metadata; assert.equal(returnedMetadata.jobId, metadata.jobId); assert.equal(Boolean(returnedMetadata.feedback), true); - assert.equal(result.isError === true, status !== 200); - if (status !== 200) assert.equal(result.content[0].text, 'Parsing failed'); + assert.equal(result.isError === true, !success); + if (!success) { + const textPayload = JSON.parse(result.content[0].text); + assert.equal(textPayload.error, 'Parsing failed'); + assert.equal(textPayload.metadata.jobId, metadata.jobId); + assert.equal(payload.error, 'Parsing failed'); + } const call = backend.requests.find((req) => req.url === '/v2/parse'); assert.equal(call.headers.authorization, undefined); assert.equal( @@ -4669,7 +4675,7 @@ test('keyless Search failure preserves the feedback reference', async (t) => { }; const backend = await startFakeFirecrawlBackend({ keylessEligible: true, - searchResponse: { status: 500, body: { success: false, error: 'Search transport failed', metadata } }, + searchResponse: { status: 200, body: { success: false, error: 'Search transport failed', metadata } }, }); t.after(() => backend.close()); const port = await getFreePort(); @@ -4687,6 +4693,7 @@ test('keyless Search failure preserves the feedback reference', async (t) => { const result = parseSseJson(await response.text()).result; assert.equal(result.isError, true); assert.deepEqual(result.structuredContent.metadata, metadata); + assert.equal(JSON.parse(result.content[0].text).metadata.jobId, metadata.jobId); }); test('every listed tool declares an output schema and returns structured content', async (t) => { From 8a0dc231da86dc6690c18e7ef46dcd8d28ad4e82 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 2 Oct 2026 22:10:38 -0500 Subject: [PATCH 23/25] test: verify complete keyless feedback forwarding --- tests/mcp-smoke.test.mjs | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 2c496beb..a2d9805c 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -4439,6 +4439,17 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv (req) => req.url === '/v2/feedback' ); assert.ok(submission); + assert.equal(submission.body.endpoint, 'search'); + assert.equal( + submission.body.jobId, + '00000000-0000-4000-8000-000000000000' + ); + assert.equal(submission.body.rating, 'partial'); + assert.equal(submission.body.task, 'Read the API retry reference'); + assert.equal( + submission.body.assessment, + 'The reference explains supported retry intervals.' + ); assert.equal(submission.headers.authorization, undefined); assert.equal( submission.headers['x-firecrawl-keyless-ip'], @@ -4468,6 +4479,9 @@ test('hosted keyless feedback bypasses exhausted operation allowance and preserv const call = await httpToolCall(port, {id: `feedback-${endpoint}`, headers: {'x-forwarded-for': '203.0.113.71'}, params: {name: 'firecrawl_feedback', arguments: args}}); assert.equal(JSON.parse(parseSseJson(await call.text()).result.content[0].text).success, true); const posted = backend.requests.filter(req => req.url === '/v2/feedback').at(-1).body; + for (const field of ['endpoint', 'jobId', 'rating', 'task', 'assessment']) { + assert.equal(posted[field], args[field]); + } assert.deepEqual(posted.observations, args.observations); assert.equal(posted.docClass, args.docClass); } From a08ef76684c51f10b309717110ddc7f723cc4e4b Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Fri, 2 Oct 2026 22:23:13 -0500 Subject: [PATCH 24/25] docs: tie keyless feedback guidance to the invitation deadline --- README.md | 3 ++- src/index.ts | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index f756340b..cd2083ee 100644 --- a/README.md +++ b/README.md @@ -567,7 +567,8 @@ Parse jobs is optional. Consider submitting feedback through `firecrawl_feedback The tool's `observations` parameter lists the category fields below. Use only available evidence and keep unverified expectations distinct from source comparisons. -Keyless job references are valid for 24 hours. Each job accepts one submission; +Submit before the invitation's `expiresAt` deadline, which provides a 24-hour +feedback window for the job. Each job accepts one submission; retrying it returns the original feedback ID. Feedback remains available after operation allowance is exhausted and does not consume or restore that allowance. diff --git a/src/index.ts b/src/index.ts index 294c5e48..ba46b9f4 100644 --- a/src/index.ts +++ b/src/index.ts @@ -3603,7 +3603,7 @@ For an Alexandria session, set endpoint to \`alexandria\`, omit jobId, and provi Consider submitting keyless Search, Scrape, and Parse feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. It requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. -Use evidence already available; no extra investigation. The stored submission must fit within 8 KiB, including server defaults. Submit within 24 hours from the same caller IP. One submission per job; retries return the original feedback ID. Attempts are rate limited. Feedback remains available after operation quota exhaustion and does not consume or restore quota. Contract: https://docs.firecrawl.dev/api-reference/endpoint/feedback. +Use evidence already available; no extra investigation. The stored submission must fit within 8 KiB, including server defaults. Submit from the same caller IP before the invitation's expiresAt deadline (24-hour feedback window for the job). One submission per job; retries return the original feedback ID. Attempts are rate limited. Feedback remains available after operation quota exhaustion and does not consume or restore quota. Contract: https://docs.firecrawl.dev/api-reference/endpoint/feedback. `, outputSchema: feedbackOutputSchema, parameters: z.object({ From a70df22ccdf93def2248ab3db6ebd13922be87a1 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Sat, 3 Oct 2026 14:33:11 -0500 Subject: [PATCH 25/25] fix(feedback): clarify keyless feedback guidance and keep it out of shared descriptions Keyless instructions and the feedback tool now ask agents to submit concise feedback on observed result quality or missing coverage when the host permits it, especially for wrong, incomplete, blocked, or failed results. They also state that feedback does not determine whether the task is complete and should not be retried after a decline or rejection. The keyless sentence leaves the Search, Scrape, and Parse descriptions, which authenticated sessions also read. Keyless agents still see the invitation in each eligible result, the keyless instructions, and the feedback tool. The selection metadata guard covers firecrawl_feedback again. --- README.md | 2 +- src/index.ts | 10 +++++----- tests/mcp-alexandria-feedback-hints.test.mjs | 2 +- tests/mcp-smoke.test.mjs | 9 ++++++++- 4 files changed, 15 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index cd2083ee..2836f83d 100644 --- a/README.md +++ b/README.md @@ -559,7 +559,7 @@ Sends structured feedback on a previous `firecrawl_search` result. The first fee ### 3c. Generic Feedback Tool (`firecrawl_feedback`) Sends evidence through `/v2/feedback`. Feedback on keyless Search, Scrape, and -Parse jobs is optional. Consider submitting feedback through `firecrawl_feedback`, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. These jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 +Parse jobs is optional. Keyless guidance asks agents to submit concise feedback on observed result quality or missing coverage when the host permits it, especially if a result is wrong, incomplete, blocked, or an error. Feedback does not determine whether a task is complete. These jobs require `endpoint`, `jobId`, `rating`, `task`, `assessment`, and 1-20 `observations`. Each observation has `kind`, `detail`, and `basis`: `output`, `source_comparison`, or `expectation`. Source comparisons also require `comparison: {reference, detail}`. diff --git a/src/index.ts b/src/index.ts index ba46b9f4..bf838175 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1314,7 +1314,7 @@ const openAiAppsChallengeToken = normalizeHeader( const FULL_PROFILE_INSTRUCTIONS = `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. Provide only the required inputs and account for stated network or external side effects.`; -const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions can use firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. Feedback does not consume operation quota. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. Hosted firecrawl_parse processes supported local files through its two-phase upload flow. Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; +const KEYLESS_PROFILE_INSTRUCTIONS = `Keyless sessions can use firecrawl_search, firecrawl_scrape, firecrawl_parse, and firecrawl_feedback with usage limits. Submit concise feedback through firecrawl_feedback on observed result quality or missing coverage when a keyless Search, Scrape, or Parse result includes a feedback invitation and the host permits it, especially if the result is wrong, incomplete, blocked, or an error. Use its job reference and evidence already available, and omit sensitive information. Feedback does not determine whether the task is complete and does not consume operation quota. If it is unavailable, declined, or rejected, continue without retries. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. Hosted firecrawl_parse processes supported local files through its two-phase upload flow. Local firecrawl_parse requires FIRECRAWL_API_URL to be explicitly configured before reading or uploading files. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; // The search surface exposes web/developer/research search plus the two Alexandria // tools (catalogue lookup and provider execution). Its instructions @@ -2737,7 +2737,7 @@ const scrapeTool: RegisteredTool = { description: ` Scrape one URL and return its content: markdown by default, or HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema. Use it when the request identifies a page and needs its content or defined fields. Use \`firecrawl_search\` when additional web sources are needed; on an authenticated session, \`firecrawl_map\` lists a site's URLs and \`firecrawl_crawl\` collects a set of pages. -Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. Keyless results include a job reference. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. +Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. On an authenticated session with Alexandria access, \`firecrawl_search\` with \`sources\` unset and \`firecrawl_find_tools\` can discover providers for the same fields across several pages; a matching provider returns typed records in one call. Keyless sessions have no provider matches. @@ -2868,7 +2868,7 @@ ${ALEXANDRIA_SEARCH_LEAD} On an authenticated session, tool matches describe available capabilities; \`firecrawl_find_tools\` returns their contracts and \`firecrawl_scrape\` with an \`alexandria\` body executes a selected capability. Keyless sessions get no Alexandria matches in data.tools. -For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. Keyless results include a job reference. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. +For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. `, outputSchema: searchOutputSchema, parameters: z @@ -3601,7 +3601,7 @@ Submit job quality feedback through /v2/feedback. Authenticated jobs use existin For an Alexandria session, set endpoint to \`alexandria\`, omit jobId, and provide requestedWebsite (url and requestedFunctionality), objective, rationale, and rating. objective is the underlying goal behind the session: what you or your user were ultimately trying to accomplish (for example, "shortlist federal IT contracts to bid on this quarter"), not only what was needed from this website. Optional providerFeedback and capabilityFeedback describe gaps or errors. Capability issues: new_capability_request (requires requestedFunctionality), missing_capability, insufficient_functionality, incorrect_result, execution_error, other. Alexandria feedback has no job-age deadline and no credit refund. -Consider submitting keyless Search, Scrape, and Parse feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. It requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. +Submit concise keyless Search, Scrape, or Parse feedback on observed result quality or missing coverage when the host permits it, especially if a result is wrong, incomplete, blocked, or an error. Feedback does not determine whether the task is complete. It requires task, assessment, rating, and 1-20 observations. Task, assessment, and each detail contain 10-2000 characters. Each observation includes kind, detail, and basis: output, source_comparison, or expectation. A source_comparison also requires comparison: {reference, detail}. The observations parameter describes each category's fields and reason codes. Use evidence already available; no extra investigation. The stored submission must fit within 8 KiB, including server defaults. Submit from the same caller IP before the invitation's expiresAt deadline (24-hour feedback window for the job). One submission per job; retries return the original feedback ID. Attempts are rate limited. Feedback remains available after operation quota exhaustion and does not consume or restore quota. Contract: https://docs.firecrawl.dev/api-reference/endpoint/feedback. `, @@ -4357,7 +4357,7 @@ Parse one supported document into markdown, HTML, links, summary, targeted answe Local MCP requires an explicitly configured \`FIRECRAWL_API_URL\` before reading \`filePath\` from the server filesystem and uploading it. Parsing happens on that API server. Hosted MCP uses two calls: first provide \`filePath\` to receive upload instructions, upload locally, then call again with the returned \`uploadRef\`; do not send both fields together. Remote web URLs belong in \`firecrawl_scrape\`. -Set \`redactPII\` to request redaction of personally identifiable information in the returned content. \`zeroDataRetention\` requires an eligible authenticated account; omit it for anonymous keyless use. Returns upload instructions for hosted phase one or parsed document content for the final call. Authenticated final responses can include a \`data.metadata.scrapeId\` for optional parse feedback. Keyless results include a job reference. Consider submitting feedback through firecrawl_feedback, especially if this result is wrong, incomplete, blocked, or an error. Include specific evidence to help improve Firecrawl. +Set \`redactPII\` to request redaction of personally identifiable information in the returned content. \`zeroDataRetention\` requires an eligible authenticated account; omit it for anonymous keyless use. Returns upload instructions for hosted phase one or parsed document content for the final call. Authenticated final responses can include a \`data.metadata.scrapeId\` for optional parse feedback. `, outputSchema: parseOutputSchema, parameters: parseParamsSchema, diff --git a/tests/mcp-alexandria-feedback-hints.test.mjs b/tests/mcp-alexandria-feedback-hints.test.mjs index 914c0cca..f808e006 100644 --- a/tests/mcp-alexandria-feedback-hints.test.mjs +++ b/tests/mcp-alexandria-feedback-hints.test.mjs @@ -23,7 +23,7 @@ test('Alexandria selection metadata omits feedback workflow', async (t) => { const byName = new Map(tools.map((tool) => [tool.name, tool])); assert(byName.has('firecrawl_feedback')); for (const name of ['firecrawl_scrape', 'firecrawl_find_tools', 'firecrawl_search']) { - assert.doesNotMatch(byName.get(name).description, /feedbackTool|Alexandria quality feedback/i, name); + assert.doesNotMatch(byName.get(name).description, /firecrawl_feedback|feedbackTool|Alexandria quality feedback/i, name); } }); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index a2d9805c..f68bbf43 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1647,9 +1647,16 @@ test('local keyless stdio keeps profile guidance keyless-scoped and exposes shar assert.equal(toolNames.includes('firecrawl_feedback'), true); const search = tools.tools.find((tool) => tool.name === 'firecrawl_search'); assert.ok(search); + // Tool descriptions are shared by every session; keyless feedback is + // pointed to in keyless results, the keyless instructions and the feedback + // tool instead. for (const name of ['firecrawl_search', 'firecrawl_scrape', 'firecrawl_parse']) { - assert.match(tools.tools.find(tool => tool.name === name).description, /Keyless results.*Consider submitting feedback through firecrawl_feedback/); + assert.doesNotMatch(tools.tools.find(tool => tool.name === name).description, /firecrawl_feedback/); } + assert.match( + keylessGuidance, + /Submit concise feedback through firecrawl_feedback.*Feedback does not determine whether the task is complete/i + ); }); test('monitor create gives queries precedence over page targets', async (t) => {