From 5761746311a4b89c3886d33ee59a14b29cb94720 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Wed, 23 Sep 2026 14:53:41 -0500 Subject: [PATCH 1/9] fix(mcp): clarify optional Alexandria feedback --- src/alexandria-feedback.ts | 4 ++-- src/index.ts | 6 +++--- tests/mcp-alexandria-feedback-hints.test.mjs | 12 ++++++------ 3 files changed, 11 insertions(+), 11 deletions(-) diff --git a/src/alexandria-feedback.ts b/src/alexandria-feedback.ts index 6cf49458..709a763e 100644 --- a/src/alexandria-feedback.ts +++ b/src/alexandria-feedback.ts @@ -62,12 +62,12 @@ export const alexandriaFeedbackFields = { * Wording is checked by scripts/agent-metadata-policy.mjs. */ export const ALEXANDRIA_FEEDBACK_GUIDANCE = - 'After an Alexandria task, whether a capability ran or discovery found nothing for the website, call firecrawl_feedback once per website with endpoint "alexandria": requestedWebsite (the url the user needed data from and the requestedFunctionality they needed), a rating, a rationale from observed results, and any providerFeedback or capabilityFeedback gaps. It is free, needs no job ID, has no deadline, and follows the answer rather than replacing it.'; + 'Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria", whether a capability ran or discovery found nothing for the website. One submission describes one requested website and functionality, a rating, a rationale from observed results, and any provider or capability gaps. It is free, needs no job ID, and has no deadline.'; /** Appended to Alexandria results so the pointer travels with the data the agent is reading. */ export const ALEXANDRIA_FEEDBACK_HINT = { name: 'firecrawl_feedback', - when: 'Once per website after the task is complete, including when no provider covered the site. Free; no job ID or deadline.', + when: 'Optional, once per website after the task is complete, including when no provider covered the site. Free; no job ID or deadline.', arguments: { endpoint: 'alexandria', rating: '', diff --git a/src/index.ts b/src/index.ts index 800a99f2..6865ff26 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2441,7 +2441,7 @@ Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch On an authenticated session with Alexandria access, if you are about to scrape the same fields from several pages, first run \`firecrawl_search\` with \`sources\` unset (or \`firecrawl_find_tools\`): a matching Alexandria provider returns those fields as typed records in one call. Keyless sessions have no provider matches; scrape directly. -Alexandria mode, on an authenticated session with Alexandria access: pass \`alexandria\` instead of \`url\` to execute catalogued capabilities; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms, and its results include a \`feedbackTool\` pointer: after the task, report how the catalogue served the website through \`firecrawl_feedback\` with endpoint \`alexandria\` (free, no job ID). +Alexandria mode, on an authenticated session with Alexandria access: pass \`alexandria\` instead of \`url\` to execute catalogued capabilities; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms, and its results include a \`feedbackTool\` pointer for optional \`firecrawl_feedback\` with endpoint \`alexandria\` (free, no job ID). `, parameters: scrapeToolParamsSchema, execute: async (args: unknown, { session, log, client: mcpClient }): Promise => { @@ -2551,7 +2551,7 @@ Search web, news, or image sources and return ranked results with query-relevant ${ALEXANDRIA_SEARCH_LEAD} -On an authenticated session, tool matches are discovery, not executed data: execute one through \`firecrawl_scrape\` with an \`alexandria\` body, or read its full contract with \`firecrawl_find_tools\`; after an Alexandria task, call firecrawl_feedback once per website with endpoint "alexandria" (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. +On an authenticated session, tool matches are discovery, not executed data: execute one through \`firecrawl_scrape\` with an \`alexandria\` body, or read its full contract with \`firecrawl_find_tools\`. Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria" on authenticated sessions (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. `, @@ -2689,7 +2689,7 @@ const findToolsTool: RegisteredTool = { idempotentHint: true, }, description: - 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Inspect a selected contract before executing through firecrawl_scrape; reuse contracts already returned. Follow nextTool for further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results. Results include a feedbackTool pointer: after the task, report how the catalogue served the website through firecrawl_feedback with endpoint alexandria, including when nothing covered it (free, no job ID).', + 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Inspect a selected contract before executing through firecrawl_scrape; reuse contracts already returned. Follow nextTool for further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results. Results include a feedbackTool pointer for optional firecrawl_feedback with endpoint alexandria, including when nothing covered it (free, no job ID).', parameters: findToolsSchema, execute: async (args, { session, client: mcpClient }) => { const origin = requestOrigin(mcpClient, session); diff --git a/tests/mcp-alexandria-feedback-hints.test.mjs b/tests/mcp-alexandria-feedback-hints.test.mjs index 57dea740..1b2ba1e1 100644 --- a/tests/mcp-alexandria-feedback-hints.test.mjs +++ b/tests/mcp-alexandria-feedback-hints.test.mjs @@ -12,19 +12,19 @@ function assertFeedbackHint(payload) { assert.equal(payload.feedbackTool?.name, 'firecrawl_feedback'); assert.equal(payload.feedbackTool.arguments.endpoint, 'alexandria'); assert.match(payload.feedbackTool.arguments.requestedWebsite.url, /website/i); - assert.match(payload.feedbackTool.when, /once per website/i); + assert.match(payload.feedbackTool.when, /^Optional, once per website/i); } -test('server instructions and Alexandria tool descriptions point at firecrawl_feedback', async (t) => { +test('server instructions and Alexandria tool descriptions present feedback as optional', async (t) => { const { client, init } = await startStdioWithApi(t); - assert.match(init.instructions, /call firecrawl_feedback once per website with endpoint "alexandria"/); + assert.match(init.instructions, /Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria"/); assert.match(init.instructions, /whether a capability ran or discovery found nothing for the website/); const { tools } = await client.request('tools/list', {}); const byName = new Map(tools.map((tool) => [tool.name, tool])); assert(byName.has('firecrawl_feedback')); - assert.match(byName.get('firecrawl_scrape').description, /feedbackTool.*firecrawl_feedback.*endpoint `alexandria`/s); - assert.match(byName.get('firecrawl_find_tools').description, /feedbackTool.*firecrawl_feedback.*including when nothing covered it/s); - assert.match(byName.get('firecrawl_search').description, /call firecrawl_feedback once per website/); + assert.match(byName.get('firecrawl_scrape').description, /feedbackTool.*optional.*firecrawl_feedback.*endpoint `alexandria`/s); + assert.match(byName.get('firecrawl_find_tools').description, /feedbackTool.*optional.*firecrawl_feedback.*including when nothing covered it/s); + assert.match(byName.get('firecrawl_search').description, /Optional Alexandria quality feedback is available through firecrawl_feedback/); }); test('Alexandria executions and discovery results carry the feedback hint; utility calls do not', async (t) => { From 55092344b2e7f89e7a779204e220d97c5f74c40d Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Wed, 23 Sep 2026 15:14:41 -0500 Subject: [PATCH 2/9] fix(mcp): clarify Alexandria execution metadata --- src/alexandria.ts | 4 ++-- src/index.ts | 24 ++++++++++---------- tests/helpers/alexandria-metadata.mjs | 32 +++++++++++++++++++++++++++ tests/mcp-description-budget.test.mjs | 8 ++++--- tests/mcp-search-profile.test.mjs | 4 ++++ tests/mcp-smoke.test.mjs | 4 ++-- 6 files changed, 57 insertions(+), 19 deletions(-) create mode 100644 tests/helpers/alexandria-metadata.mjs diff --git a/src/alexandria.ts b/src/alexandria.ts index 15f1d60d..0a0bdd22 100644 --- a/src/alexandria.ts +++ b/src/alexandria.ts @@ -74,7 +74,7 @@ export const ALEXANDRIA_SOURCES_OPT_OUT = export const ALEXANDRIA_SEARCH_LEAD = 'Authenticated search also returns matching Alexandria data providers in data.tools (' + ALEXANDRIA_CATALOGUE_VERTICALS + '). Prefer a provider over scraping pages when the task needs the same fields across several entities, exact figures or timestamps, provenance, or many records; use web results when they already answer the question. ' + ALEXANDRIA_SOURCES_OPT_OUT; export const ALEXANDRIA_CONTRACT_GUIDANCE = - 'Read the selected contract before executing: required inputs and requiresOneOf groups (at least one member per group), example.request/example.response when present, and response.key (do not assume records is the result key). Follow the declared pagination input and response cursor, preserving filters; catalogue next is separate from provider pagination.'; + 'The selected contract defines required inputs, requiresOneOf groups (at least one member per group), example.request/example.response when present, and response.key (which may differ from records). Provider pagination uses the declared input and response cursor with the same filters; catalogue next is separate from provider pagination.'; export function findToolsOptions(args: z.infer) { const level = args.level ?? (args.query || args.capabilities?.length || args.providers?.length || args.groups?.length || args.urls?.length ? 'tools' : args.categories?.length ? 'providers' : 'categories'); @@ -105,4 +105,4 @@ export function withFindToolsNavigation(envelope: any) { } export const ALEXANDRIA_SEARCH_INSTRUCTIONS = - 'Authenticated search combines web results, semantic tool summaries and domain matches. Use sources: ["alexandria"] for semantic tools only, or sources: ["web"] for web only. domainTools: false disables domain matching. Tool matches describe available structured-data capabilities, not executed data. toolDetail: "compact" (default) returns only provider, capability and description; "summary" adds metadata; "full" includes their input and output contracts. Execute a matched tool through firecrawl_scrape with an alexandria body; use firecrawl_find_tools to browse the catalogue or read a full contract.'; + 'Authenticated search combines web results, semantic tool summaries and domain matches. Use sources: ["alexandria"] for semantic tools only, or sources: ["web"] for web only. domainTools: false disables domain matching. Tool matches describe available structured-data capabilities, not executed data. toolDetail: "compact" (default) returns only provider, capability and description; "summary" adds metadata; "full" includes their input and output contracts. firecrawl_scrape with an alexandria body executes a selected capability; firecrawl_find_tools provides catalogue browsing and full contracts.'; diff --git a/src/index.ts b/src/index.ts index 6865ff26..a424cc99 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1197,7 +1197,7 @@ const openAiAppsChallengeToken = normalizeHeader( ); const FULL_PROFILE_INSTRUCTIONS = - `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} When a provider matches, prefer it over scraping pages if the task needs the same fields across several entities, provenance, exact figures or timestamps, or a large set of records; execute it through firecrawl_scrape with the returned contract. If web results already answer the question, use them. Before scraping more than one page for the same fields, spend one free firecrawl_find_tools call to check for a provider. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question — code behaviour, a library or framework, an API contract, an error message, or a known bug — firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Execution with firecrawl_scrape alexandria uses requestId; reuse the returned ID for retries of the same payload, never a new ID to bypass a pending or uncertain 409. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Follow the returned terms/show and terms/accept calls through firecrawl_scrape; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. Otherwise direct an organization admin to the dashboard URL. Do not repeat successful provider calls just because the client could not display their output. Provide only the required inputs and account for stated network or external side effects. + `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Alexandria requestId identifies one logical execution; repeated attempts of the identical payload require the same ID, including pending or uncertain executions. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Through firecrawl_scrape, terms/show displays the agreement and terms/accept records acceptance; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. A data request does not authorize acceptance. The dashboard URL provides an alternative for an organization admin. Retained results remain accessible without repeating successful provider calls. Provide only the required inputs and account for stated network or external side effects. ${ALEXANDRIA_FEEDBACK_GUIDANCE}`; const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse processes supported local files through its two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; @@ -1207,7 +1207,7 @@ const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_s // uses them. const SEARCH_PROFILE_INSTRUCTIONS = ALEXANDRIA_SEARCH_INSTRUCTIONS + - ` Firecrawl provides web, developer, and research search, and executes catalogued Alexandria data providers. Use firecrawl_search to find relevant results across the web and specialized indexes; authenticated searches also return matching Alexandria providers in data.tools. Use firecrawl_find_tools to browse the Alexandria catalogue or read a provider's full contract, and firecrawl_scrape with an alexandria body to execute a capability discovered that way; firecrawl_scrape with a url retrieves one supplied page. For a programming question, firecrawl_developer_search searches indexed public repositories, GitHub issues, merged pull requests, READMEs, and code documentation and returns the matched passages, and skills: "only" narrows it to agent-skill files; firecrawl_search with categories: ["developer"] reaches the same index beside ordinary web results, returning the hits in the web group rather than as passages and offering no skills filter. For a biomedical, life-science, clinical, or arXiv literature question, the firecrawl_research_* tools search the paper index, while categories: ["research"] on firecrawl_search filters ordinary web results to research-affiliated websites. Use the firecrawl_research_* tools to search academic and research literature, expand from anchor papers via the citation graph, and read full-text passages from a specific paper. Search and discovery tools are read-only and return ranked results. Billing: web, developer and research search are billed per request; Alexandria discovery (firecrawl_search with sources ["alexandria"] alone, and firecrawl_find_tools) is free; firecrawl_scrape is billed, as a page retrieval in url mode or at each executed capability's listed price in alexandria mode.`; + ` Firecrawl provides web, developer, and research search, and executes catalogued Alexandria data providers. Use firecrawl_search to find relevant results across the web and specialized indexes; authenticated searches also return matching Alexandria providers in data.tools. firecrawl_find_tools provides catalogue browsing and provider contracts; firecrawl_scrape with an alexandria body executes a selected capability; firecrawl_scrape with a url retrieves one supplied page. For a programming question, firecrawl_developer_search searches indexed public repositories, GitHub issues, merged pull requests, READMEs, and code documentation and returns the matched passages, and skills: "only" narrows it to agent-skill files; firecrawl_search with categories: ["developer"] reaches the same index beside ordinary web results, returning the hits in the web group rather than as passages and offering no skills filter. For a biomedical, life-science, clinical, or arXiv literature question, the firecrawl_research_* tools search the paper index, while categories: ["research"] on firecrawl_search filters ordinary web results to research-affiliated websites. Use the firecrawl_research_* tools to search academic and research literature, expand from anchor papers via the citation graph, and read full-text passages from a specific paper. Search and discovery tools are read-only and return ranked results. Billing: web, developer and research search are billed per request; Alexandria discovery (firecrawl_search with sources ["alexandria"] alone, and firecrawl_find_tools) is free; firecrawl_scrape is billed, as a page retrieval in url mode or at each executed capability's listed price in alexandria mode.`; // The exact set of tools the search surface exposes. Registration is filtered // against this set, so anything not listed here can never appear on that @@ -2005,14 +2005,14 @@ const scrapeToolParamsSchema = scrapeParamsSchema .regex(/^[A-Za-z0-9._:-]{1,128}$/) .optional() .describe( - 'Alexandria execution ID. Reuse the returned ID for retries of the identical payload, never a new ID to bypass pending or uncertain execution; generated when omitted. For potentially large workflow results, supply and preserve one before execution. Errors relay a code and chargeId: request_in_flight (409) retry the same requestId later; request_unresolved (503) keep the requestId for reconciliation, never mint a new one; duplicate_request (409) the requestId belongs to a different payload; unknown_provider (404), insufficient_credits (402) and billing_unavailable (503) mean nothing executed.' + 'Identifies one logical Alexandria execution; generated when omitted and returned with the result. Repeated attempts of the identical payload require the same ID. A new ID cannot reconcile a pending or uncertain execution. A caller-supplied ID supports recovery if no response is received. Errors relay a code and may include chargeId: request_in_flight (409) means the execution is pending; request_unresolved (503) requires reconciliation under the same ID; duplicate_request (409) means the ID belongs to a different payload; unknown_provider (404), insufficient_credits (402) and billing_unavailable (503) mean nothing executed.' ), alexandria: z.union([exchangeCallSchema, exchangeCallsSchema]) .optional() .describe( - 'Execute catalogued Alexandria capabilities instead of scraping a URL. Exactly one of url or alexandria. One {provider, capability, options} object or an array of 1-10, found through firecrawl_search or firecrawl_find_tools. Each call may include version to pin a published workflow; omitting it uses latest. Only timeout also applies at the top level. ' + + 'Catalogued Alexandria capability invocation, mutually exclusive with url. One {provider, capability, options} object or an array of 1-10, with contracts available through firecrawl_search or firecrawl_find_tools. Each call may include version to pin a published workflow; omitting it uses latest. Only requestId and timeout are supported alongside alexandria. ' + ALEXANDRIA_CONTRACT_GUIDANCE + - ' Returns per-capability results in data.alexandria with data, records, or an error with a code; check each item even when the outer response succeeds. If a response provides nextTool, follow it to read a large result instead of repeating a successful provider call. Needs an API key on a team with Alexandria enabled. A terms-gated provider returns THIRD_PARTY_DATA_TERMS_REQUIRED (403) with requiresAction.url: follow the returned terms/show and terms/accept calls through this tool, accepting only after explicit user authorization for the reviewed version and digest; an organization admin can instead accept at the dashboard URL. Retry only after confirmed acceptance.' + ' Returns per-capability results in data.alexandria with data, records, or an error with a code; individual capabilities can fail even when the outer response succeeds. nextTool identifies access to retained results without repeating a successful provider call. Requires an API key on a team with Alexandria enabled. THIRD_PARTY_DATA_TERMS_REQUIRED (403) means execution is blocked by provider terms, with requiresAction.url for review by an organization admin. Through this tool, terms/show displays the agreement and terms/accept records acceptance; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. A data request does not authorize acceptance. Provider execution remains blocked until acceptance is confirmed.' ), toolDetail: z.enum(['compact', 'summary', 'full']).optional().describe('URL domain discovery detail: summary by default, compact returns provider/capability/description, full includes contracts.'), domainTools: z @@ -2439,9 +2439,9 @@ Scrape one URL and return its content: markdown by default, or HTML, links, scre Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch or a smaller \`maxAge\` to bound staleness. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled. Authenticated responses can include a \`metadata.scrapeId\` for optional scrape feedback. -On an authenticated session with Alexandria access, if you are about to scrape the same fields from several pages, first run \`firecrawl_search\` with \`sources\` unset (or \`firecrawl_find_tools\`): a matching Alexandria provider returns those fields as typed records in one call. Keyless sessions have no provider matches; scrape directly. +On an authenticated session with Alexandria access, \`firecrawl_search\` with \`sources\` unset and \`firecrawl_find_tools\` can discover providers for the same fields across several pages; a matching provider returns typed records in one call. Keyless sessions have no provider matches. -Alexandria mode, on an authenticated session with Alexandria access: pass \`alexandria\` instead of \`url\` to execute catalogued capabilities; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms, and its results include a \`feedbackTool\` pointer for optional \`firecrawl_feedback\` with endpoint \`alexandria\` (free, no job ID). +Alexandria mode, on an authenticated session with Alexandria access: \`alexandria\` selects catalogued capability execution and is mutually exclusive with \`url\`; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms, and its results include a \`feedbackTool\` pointer for optional \`firecrawl_feedback\` with endpoint \`alexandria\` (free, no job ID). `, parameters: scrapeToolParamsSchema, execute: async (args: unknown, { session, log, client: mcpClient }): Promise => { @@ -2551,7 +2551,7 @@ Search web, news, or image sources and return ranked results with query-relevant ${ALEXANDRIA_SEARCH_LEAD} -On an authenticated session, tool matches are discovery, not executed data: execute one through \`firecrawl_scrape\` with an \`alexandria\` body, or read its full contract with \`firecrawl_find_tools\`. Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria" on authenticated sessions (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. +On an authenticated session, tool matches describe available capabilities; \`firecrawl_find_tools\` returns their contracts and \`firecrawl_scrape\` with an \`alexandria\` body executes a selected capability. Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria" on authenticated sessions (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. `, @@ -2689,7 +2689,7 @@ const findToolsTool: RegisteredTool = { idempotentHint: true, }, description: - 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Inspect a selected contract before executing through firecrawl_scrape; reuse contracts already returned. Follow nextTool for further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results. Results include a feedbackTool pointer for optional firecrawl_feedback with endpoint alexandria, including when nothing covered it (free, no job ID).', + 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Contracts describe the inputs and outputs for execution through firecrawl_scrape. nextTool identifies further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results. Results include a feedbackTool pointer for optional firecrawl_feedback with endpoint alexandria, including when nothing covered it (free, no job ID).', parameters: findToolsSchema, execute: async (args, { session, client: mcpClient }) => { const origin = requestOrigin(mcpClient, session); @@ -2728,9 +2728,9 @@ server.addTool(findToolsTool); const SEARCH_SURFACE_SCRAPE_DESCRIPTION = ` Scrape one URL and return its content, or execute catalogued Alexandria capabilities. URL mode returns markdown by default, or HTML, links, screenshots, branding data, a targeted answer, or JSON matching a supplied schema, plus page metadata. Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch. A successful response does not by itself confirm the page is still current. Browser actions can change the live page when interactive actions are enabled, and a named browser profile can load saved session data and overwrite its stored state. -Before scraping the same fields from several pages, first run \`firecrawl_search\` with \`sources\` unset (or \`firecrawl_find_tools\`): a matching Alexandria provider returns those fields as typed records in one call. +\`firecrawl_search\` with \`sources\` unset and \`firecrawl_find_tools\` can discover providers for the same fields across several pages; a matching Alexandria provider returns typed records in one call. -Alexandria mode: pass \`alexandria\` instead of \`url\`; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms. Alexandria execution is billed at the capability's listed price and needs a team with Alexandria enabled. +Alexandria mode: \`alexandria\` selects catalogued capability execution and is mutually exclusive with \`url\`; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms. Alexandria execution is billed at the capability's listed price and needs a team with Alexandria enabled. `; const searchSurfaceScrapeTool: RegisteredTool = { ...scrapeTool, @@ -2738,7 +2738,7 @@ const searchSurfaceScrapeTool: RegisteredTool = { }; const SEARCH_SURFACE_FIND_TOOLS_DESCRIPTION = - 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Inspect a selected contract before executing through firecrawl_scrape; reuse contracts already returned. Follow nextTool for further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results.'; + 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Contracts describe the inputs and outputs for execution through firecrawl_scrape. nextTool identifies further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results.'; const searchSurfaceFindToolsTool: RegisteredTool = { ...findToolsTool, description: SEARCH_SURFACE_FIND_TOOLS_DESCRIPTION, diff --git a/tests/helpers/alexandria-metadata.mjs b/tests/helpers/alexandria-metadata.mjs new file mode 100644 index 00000000..d483e993 --- /dev/null +++ b/tests/helpers/alexandria-metadata.mjs @@ -0,0 +1,32 @@ +import assert from 'node:assert/strict'; + +export function assertAlexandriaMetadata(tools, instructions = '') { + const scrape = tools.find((tool) => tool.name === 'firecrawl_scrape'); + const { requestId, alexandria } = scrape.inputSchema.properties; + assert.equal(scrape.annotations.readOnlyHint, false); + assert.match(requestId.description, /identical payload.*same ID/i); + assert.match(requestId.description, /new ID cannot reconcile a pending or uncertain execution/i); + assert.match(requestId.description, /request_in_flight.*pending/); + assert.match(requestId.description, /request_unresolved.*reconciliation under the same ID/); + assert.match(alexandria.description, /mutually exclusive with url/); + assert.match(alexandria.description, /array of 1-10/); + assert.match(alexandria.description, /terms\/show.*terms\/accept/); + assert.match(alexandria.description, /explicit user authorization for the reviewed version and digest and confirmed:true/); + assert.match(alexandria.description, /data request does not authorize acceptance/i); + assert.match(alexandria.description, /execution remains blocked until acceptance is confirmed/i); + + const descriptions = [instructions]; + function collect(value) { + if (!value || typeof value !== 'object') return; + for (const [key, child] of Object.entries(value)) { + if (key === 'description' && typeof child === 'string') descriptions.push(child); + else collect(child); + } + } + collect(tools); + for (const description of descriptions) { + assert.doesNotMatch(description, /follow the returned terms\/show and terms\/accept|retry the same requestId later|call firecrawl_feedback once per website|after the task, report how the catalogue/i); + const paragraphs = description.split('\n').map((line) => line.trim()).filter(Boolean); + assert.equal(new Set(paragraphs).size, paragraphs.length, 'metadata has no duplicate paragraphs'); + } +} diff --git a/tests/mcp-description-budget.test.mjs b/tests/mcp-description-budget.test.mjs index 799e040a..5eb69cec 100644 --- a/tests/mcp-description-budget.test.mjs +++ b/tests/mcp-description-budget.test.mjs @@ -2,6 +2,7 @@ import assert from 'node:assert/strict'; import test from 'node:test'; import { startStdioWithApi } from './helpers/exchange-mcp.mjs'; import { CLAUDE_CODE_TEXT_CAP as CAP } from './helpers/description-budget.mjs'; +import { assertAlexandriaMetadata } from './helpers/alexandria-metadata.mjs'; // Claude Code truncates each tool description (and server instructions) at 2,048 // characters. The routing copy that changed agent behaviour in the AX runs has to @@ -9,8 +10,9 @@ import { CLAUDE_CODE_TEXT_CAP as CAP } from './helpers/description-budget.mjs'; // scrape-first pointer; the scrape tool has to read first as one URL -> the page. test('every tool description fits the 2,048-character cap and keeps the routing copy inside it', async (t) => { - const { client } = await startStdioWithApi(t); + const { client, init } = await startStdioWithApi(t); const { tools } = await client.request('tools/list', {}); + assertAlexandriaMetadata(tools, init.instructions); for (const tool of tools) { assert.ok((tool.description ?? '').length <= CAP, `${tool.name} description is ${(tool.description ?? '').length} chars`); } @@ -21,9 +23,9 @@ test('every tool description fits the 2,048-character cap and keeps the routing assert.match(search, /Prefer a provider over scraping pages/); const scrape = byName.get('firecrawl_scrape'); assert.match(scrape, /^Scrape one URL and return its content/); - assert.match(scrape, /if you are about to scrape the same fields from several pages, first run `firecrawl_search` with `sources` unset/); + assert.match(scrape, /`firecrawl_search` with `sources` unset.*can discover providers for the same fields across several pages/); assert.match(byName.get('firecrawl_find_tools'), /Prefer normal firecrawl_search/); // The retry rule moved out of the instructions' first window; it lives on the parameter. const scrapeParams = tools.find((tool) => tool.name === 'firecrawl_scrape').inputSchema.properties; - assert.match(scrapeParams.requestId.description, /Reuse the returned ID for retries of the identical payload, never a new ID/); + assert.match(scrapeParams.requestId.description, /Repeated attempts of the identical payload require the same ID/); }); diff --git a/tests/mcp-search-profile.test.mjs b/tests/mcp-search-profile.test.mjs index cea31435..ae8b37b8 100644 --- a/tests/mcp-search-profile.test.mjs +++ b/tests/mcp-search-profile.test.mjs @@ -7,6 +7,7 @@ import test from 'node:test'; import { setTimeout as delay } from 'node:timers/promises'; import { assertAgentMetadataPolicy } from '../scripts/agent-metadata-policy.mjs'; import { CLAUDE_CODE_TEXT_CAP } from './helpers/description-budget.mjs'; +import { assertAlexandriaMetadata } from './helpers/alexandria-metadata.mjs'; const { version: serverVersion } = JSON.parse( readFileSync(new URL('../package.json', import.meta.url), 'utf8') @@ -1057,6 +1058,7 @@ test('primary search profile agent language satisfies metadata policy gates', as const initialize = await initializeProfile(port, SEARCH_ENDPOINT, headers); const tools = await listToolDefinitions(port, SEARCH_ENDPOINT, headers); + assertAlexandriaMetadata(tools, initialize.instructions); assertAgentMetadataPolicy( [initialize.instructions, ...tools.map((tool) => tool.description ?? '')], assert @@ -1100,6 +1102,7 @@ test('account (mcp-oauth) full-surface instructions satisfy the same metadata po const initialize = await initializeProfile(port, '/v2/mcp-oauth', headers); const tools = await listToolDefinitions(port, '/v2/mcp-oauth', headers); + assertAlexandriaMetadata(tools, initialize.instructions); assertAgentMetadataPolicy( [initialize.instructions, ...tools.map((tool) => tool.description ?? '')], assert @@ -1294,6 +1297,7 @@ test('search surface registers the two Alexandria tools with surface-scoped copy const tools = await listToolDefinitions(searchPort, SEARCH_ENDPOINT, { 'x-api-key': 'fc-test', }); + assertAlexandriaMetadata(tools); // Claude Code truncates tool descriptions at CLAUDE_CODE_TEXT_CAP characters. for (const tool of tools) { assert.ok((tool.description ?? '').length <= CLAUDE_CODE_TEXT_CAP, `${tool.name} description is ${(tool.description ?? '').length} chars`); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 28e593a7..65131a05 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1089,7 +1089,7 @@ test('stdio transport initializes and lists Firecrawl tools', async (t) => { // and the requestId parameter description (retry rule, asserted in the budget test). const instructionsHead = init.instructions.slice(0, CLAUDE_CODE_TEXT_CAP); assert.match(instructionsHead, /Alexandria is Firecrawl's catalogue of data providers/); - assert.match(instructionsHead, /Before scraping more than one page for the same fields/); + assert.match(instructionsHead, /For the same fields across multiple pages/); assert.match(instructionsHead, /Passing sources without alexandria in it/); assert.match( init.instructions, @@ -1101,7 +1101,7 @@ test('stdio transport initializes and lists Firecrawl tools', async (t) => { ); assert.match( init.instructions, - /Before scraping more than one page for the same fields, spend one free firecrawl_find_tools call/ + /For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery/ ); assert.match( init.instructions, From 5d81b14cfbe1e684e3dbd780a5d3df5dad7342eb Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Wed, 23 Sep 2026 17:34:25 -0500 Subject: [PATCH 3/9] fix(mcp): clarify error metadata and simplify checks --- src/alexandria-feedback.ts | 6 +----- src/index.ts | 2 +- tests/helpers/alexandria-metadata.mjs | 13 +++---------- tests/mcp-description-budget.test.mjs | 3 --- 4 files changed, 5 insertions(+), 19 deletions(-) diff --git a/src/alexandria-feedback.ts b/src/alexandria-feedback.ts index 709a763e..b1691f52 100644 --- a/src/alexandria-feedback.ts +++ b/src/alexandria-feedback.ts @@ -56,11 +56,7 @@ export const alexandriaFeedbackFields = { .optional(), }; -/** - * Shared by the server instructions and the Alexandria tool descriptions so an - * agent that only reads one of them still learns the feedback loop exists. - * Wording is checked by scripts/agent-metadata-policy.mjs. - */ +/** Optional feedback guidance for the full-profile server instructions. */ export const ALEXANDRIA_FEEDBACK_GUIDANCE = 'Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria", whether a capability ran or discovery found nothing for the website. One submission describes one requested website and functionality, a rating, a rationale from observed results, and any provider or capability gaps. It is free, needs no job ID, and has no deadline.'; diff --git a/src/index.ts b/src/index.ts index a424cc99..bb0fc301 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2005,7 +2005,7 @@ const scrapeToolParamsSchema = scrapeParamsSchema .regex(/^[A-Za-z0-9._:-]{1,128}$/) .optional() .describe( - 'Identifies one logical Alexandria execution; generated when omitted and returned with the result. Repeated attempts of the identical payload require the same ID. A new ID cannot reconcile a pending or uncertain execution. A caller-supplied ID supports recovery if no response is received. Errors relay a code and may include chargeId: request_in_flight (409) means the execution is pending; request_unresolved (503) requires reconciliation under the same ID; duplicate_request (409) means the ID belongs to a different payload; unknown_provider (404), insufficient_credits (402) and billing_unavailable (503) mean nothing executed.' + 'Identifies one logical Alexandria execution; generated when omitted and returned with the result. Repeated attempts of the identical payload require the same ID. A new ID cannot reconcile a pending or uncertain execution. A caller-supplied ID supports recovery if no response is received. Errors relay a code and may include chargeId. request_in_flight (409) means the execution is pending; request_unresolved (503) requires reconciliation under the same ID; duplicate_request (409) means the ID belongs to a different payload; unknown_provider (404), insufficient_credits (402) and billing_unavailable (503) mean nothing executed.' ), alexandria: z.union([exchangeCallSchema, exchangeCallsSchema]) .optional() diff --git a/tests/helpers/alexandria-metadata.mjs b/tests/helpers/alexandria-metadata.mjs index d483e993..137795d3 100644 --- a/tests/helpers/alexandria-metadata.mjs +++ b/tests/helpers/alexandria-metadata.mjs @@ -15,18 +15,11 @@ export function assertAlexandriaMetadata(tools, instructions = '') { assert.match(alexandria.description, /data request does not authorize acceptance/i); assert.match(alexandria.description, /execution remains blocked until acceptance is confirmed/i); - const descriptions = [instructions]; - function collect(value) { - if (!value || typeof value !== 'object') return; - for (const [key, child] of Object.entries(value)) { - if (key === 'description' && typeof child === 'string') descriptions.push(child); - else collect(child); - } + const descriptions = [instructions, requestId.description, alexandria.description]; + for (const name of ['firecrawl_search', 'firecrawl_scrape', 'firecrawl_find_tools']) { + descriptions.push(tools.find((tool) => tool.name === name).description); } - collect(tools); for (const description of descriptions) { assert.doesNotMatch(description, /follow the returned terms\/show and terms\/accept|retry the same requestId later|call firecrawl_feedback once per website|after the task, report how the catalogue/i); - const paragraphs = description.split('\n').map((line) => line.trim()).filter(Boolean); - assert.equal(new Set(paragraphs).size, paragraphs.length, 'metadata has no duplicate paragraphs'); } } diff --git a/tests/mcp-description-budget.test.mjs b/tests/mcp-description-budget.test.mjs index 5eb69cec..0f1f4aa5 100644 --- a/tests/mcp-description-budget.test.mjs +++ b/tests/mcp-description-budget.test.mjs @@ -25,7 +25,4 @@ test('every tool description fits the 2,048-character cap and keeps the routing assert.match(scrape, /^Scrape one URL and return its content/); assert.match(scrape, /`firecrawl_search` with `sources` unset.*can discover providers for the same fields across several pages/); assert.match(byName.get('firecrawl_find_tools'), /Prefer normal firecrawl_search/); - // The retry rule moved out of the instructions' first window; it lives on the parameter. - const scrapeParams = tools.find((tool) => tool.name === 'firecrawl_scrape').inputSchema.properties; - assert.match(scrapeParams.requestId.description, /Repeated attempts of the identical payload require the same ID/); }); From e2c74b9dfa3e552861b88a46e10b603ab362d1a9 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 24 Sep 2026 12:30:15 -0500 Subject: [PATCH 4/9] fix(mcp): keep Alexandria feedback out of selection metadata --- src/alexandria-feedback.ts | 4 ---- src/index.ts | 10 ++++------ tests/mcp-alexandria-feedback-hints.test.mjs | 11 +++++------ 3 files changed, 9 insertions(+), 16 deletions(-) diff --git a/src/alexandria-feedback.ts b/src/alexandria-feedback.ts index b1691f52..c3cb53f7 100644 --- a/src/alexandria-feedback.ts +++ b/src/alexandria-feedback.ts @@ -56,10 +56,6 @@ export const alexandriaFeedbackFields = { .optional(), }; -/** Optional feedback guidance for the full-profile server instructions. */ -export const ALEXANDRIA_FEEDBACK_GUIDANCE = - 'Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria", whether a capability ran or discovery found nothing for the website. One submission describes one requested website and functionality, a rating, a rationale from observed results, and any provider or capability gaps. It is free, needs no job ID, and has no deadline.'; - /** Appended to Alexandria results so the pointer travels with the data the agent is reading. */ export const ALEXANDRIA_FEEDBACK_HINT = { name: 'firecrawl_feedback', diff --git a/src/index.ts b/src/index.ts index 3cd67c79..3c2c397f 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,6 +1,5 @@ #!/usr/bin/env node import { - ALEXANDRIA_FEEDBACK_GUIDANCE, ALEXANDRIA_FEEDBACK_HINT, alexandriaCallsWarrantFeedback, alexandriaFeedbackFields, @@ -1215,8 +1214,7 @@ const openAiAppsChallengeToken = normalizeHeader( ); const FULL_PROFILE_INSTRUCTIONS = - `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Alexandria requestId identifies one logical execution; repeated attempts of the identical payload require the same ID, including pending or uncertain executions. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Through firecrawl_scrape, terms/show displays the agreement and terms/accept records acceptance; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. A data request does not authorize acceptance. The dashboard URL provides an alternative for an organization admin. Retained results remain accessible without repeating successful provider calls. Provide only the required inputs and account for stated network or external side effects. -${ALEXANDRIA_FEEDBACK_GUIDANCE}`; + `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Alexandria requestId identifies one logical execution; repeated attempts of the identical payload require the same ID, including pending or uncertain executions. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Through firecrawl_scrape, terms/show displays the agreement and terms/accept records acceptance; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. A data request does not authorize acceptance. The dashboard URL provides an alternative for an organization admin. Retained results remain accessible without repeating successful provider calls. Provide only the required inputs and account for stated network or external side effects.`; const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse processes supported local files through its two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; // The search surface exposes web/developer/research search plus the two Alexandria @@ -2455,7 +2453,7 @@ Firecrawl may serve recently indexed content; set \`maxAge: 0\` for a live fetch On an authenticated session with Alexandria access, \`firecrawl_search\` with \`sources\` unset and \`firecrawl_find_tools\` can discover providers for the same fields across several pages; a matching provider returns typed records in one call. Keyless sessions have no provider matches. -Alexandria mode, on an authenticated session with Alexandria access: \`alexandria\` selects catalogued capability execution and is mutually exclusive with \`url\`; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms, and its results include a \`feedbackTool\` pointer for optional \`firecrawl_feedback\` with endpoint \`alexandria\` (free, no job ID). +Alexandria mode, on an authenticated session with Alexandria access: \`alexandria\` selects catalogued capability execution and is mutually exclusive with \`url\`; the \`alexandria\` and \`requestId\` parameters describe batching, retries, errors and provider terms. `, outputSchema: scrapeOutputSchema, parameters: scrapeToolParamsSchema, @@ -2573,7 +2571,7 @@ Search web, news, or image sources and return ranked results with query-relevant ${ALEXANDRIA_SEARCH_LEAD} -On an authenticated session, tool matches describe available capabilities; \`firecrawl_find_tools\` returns their contracts and \`firecrawl_scrape\` with an \`alexandria\` body executes a selected capability. Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria" on authenticated sessions (free, no job ID). Keyless sessions get no Alexandria matches in data.tools. +On an authenticated session, tool matches describe available capabilities; \`firecrawl_find_tools\` returns their contracts and \`firecrawl_scrape\` with an \`alexandria\` body executes a selected capability. Keyless sessions get no Alexandria matches in data.tools. For a programming question, add \`categories: ["developer"]\`; its hits return in \`data.web\` with \`category: "developer"\`. \`categories: ["research"]\` restricts web results to research-affiliated websites; the \`firecrawl_research_*\` tools are a separate surface over paper abstracts and full text (PubMed, bioRxiv, medRxiv, arXiv). Query operators, domain filters, \`categories\`, \`toolDetail\` and \`scrapeOptions\` are described on their parameters. Returns source-type result groups and usage metadata. Authenticated responses can include an \`id\` for optional search feedback. `, @@ -2712,7 +2710,7 @@ const findToolsTool: RegisteredTool = { idempotentHint: true, }, description: - 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Contracts describe the inputs and outputs for execution through firecrawl_scrape. nextTool identifies further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results. Results include a feedbackTool pointer for optional firecrawl_feedback with endpoint alexandria, including when nothing covered it (free, no job ID).', + 'Browse Alexandria data providers and workflows or read a selected contract. Alexandria covers ' + ALEXANDRIA_CATALOGUE_VERTICALS + ': typed, sourced records through published contracts. Prefer normal firecrawl_search for a data task; it already returns matching providers. Use this tool when the contract you need was not returned in full, to browse a category when search found nothing, or before scraping the same fields from several pages. Discovery is free. Use query for semantic discovery or urls to find providers for a website. With no arguments, browse categories, then providers and tools. Contracts describe the inputs and outputs for execution through firecrawl_scrape. nextTool identifies further discovery or pagination. Discovery does not execute providers. Use firecrawl_search when you also need web results.', outputSchema: findToolsOutputSchema, parameters: findToolsSchema, execute: async (args, { session, client: mcpClient }) => { diff --git a/tests/mcp-alexandria-feedback-hints.test.mjs b/tests/mcp-alexandria-feedback-hints.test.mjs index 1b2ba1e1..0087f82f 100644 --- a/tests/mcp-alexandria-feedback-hints.test.mjs +++ b/tests/mcp-alexandria-feedback-hints.test.mjs @@ -15,16 +15,15 @@ function assertFeedbackHint(payload) { assert.match(payload.feedbackTool.when, /^Optional, once per website/i); } -test('server instructions and Alexandria tool descriptions present feedback as optional', async (t) => { +test('Alexandria selection metadata omits feedback workflow', async (t) => { const { client, init } = await startStdioWithApi(t); - assert.match(init.instructions, /Optional Alexandria quality feedback is available through firecrawl_feedback with endpoint "alexandria"/); - assert.match(init.instructions, /whether a capability ran or discovery found nothing for the website/); + assert.doesNotMatch(init.instructions, /firecrawl_feedback|feedbackTool|Alexandria quality feedback/i); const { tools } = await client.request('tools/list', {}); const byName = new Map(tools.map((tool) => [tool.name, tool])); assert(byName.has('firecrawl_feedback')); - assert.match(byName.get('firecrawl_scrape').description, /feedbackTool.*optional.*firecrawl_feedback.*endpoint `alexandria`/s); - assert.match(byName.get('firecrawl_find_tools').description, /feedbackTool.*optional.*firecrawl_feedback.*including when nothing covered it/s); - assert.match(byName.get('firecrawl_search').description, /Optional Alexandria quality feedback is available through firecrawl_feedback/); + for (const name of ['firecrawl_scrape', 'firecrawl_find_tools', 'firecrawl_search']) { + assert.doesNotMatch(byName.get(name).description, /firecrawl_feedback|feedbackTool|Alexandria quality feedback/i, name); + } }); test('Alexandria executions and discovery results carry the feedback hint; utility calls do not', async (t) => { From 7ea863c13d9fa0f5355c8c36feda008259d3bfaf Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 24 Sep 2026 12:38:06 -0500 Subject: [PATCH 5/9] docs(mcp): streamline Alexandria feedback hint --- src/alexandria-feedback.ts | 2 +- tests/mcp-alexandria-feedback-hints.test.mjs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/alexandria-feedback.ts b/src/alexandria-feedback.ts index c3cb53f7..6a8941be 100644 --- a/src/alexandria-feedback.ts +++ b/src/alexandria-feedback.ts @@ -59,7 +59,7 @@ export const alexandriaFeedbackFields = { /** Appended to Alexandria results so the pointer travels with the data the agent is reading. */ export const ALEXANDRIA_FEEDBACK_HINT = { name: 'firecrawl_feedback', - when: 'Optional, once per website after the task is complete, including when no provider covered the site. Free; no job ID or deadline.', + when: 'Available after task completion; at most once per website, including uncovered sites. Free; no job ID or deadline.', arguments: { endpoint: 'alexandria', rating: '', diff --git a/tests/mcp-alexandria-feedback-hints.test.mjs b/tests/mcp-alexandria-feedback-hints.test.mjs index 0087f82f..fb7cce59 100644 --- a/tests/mcp-alexandria-feedback-hints.test.mjs +++ b/tests/mcp-alexandria-feedback-hints.test.mjs @@ -12,7 +12,7 @@ function assertFeedbackHint(payload) { assert.equal(payload.feedbackTool?.name, 'firecrawl_feedback'); assert.equal(payload.feedbackTool.arguments.endpoint, 'alexandria'); assert.match(payload.feedbackTool.arguments.requestedWebsite.url, /website/i); - assert.match(payload.feedbackTool.when, /^Optional, once per website/i); + assert.match(payload.feedbackTool.when, /^Available after task completion; at most once per website/i); } test('Alexandria selection metadata omits feedback workflow', async (t) => { From ba816a2624ae44c7ad5cfe6db542eb8ebec17721 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 24 Sep 2026 12:46:13 -0500 Subject: [PATCH 6/9] docs(mcp): clarify optional feedback at result boundary --- README.md | 4 ++-- src/alexandria-feedback.ts | 2 +- tests/mcp-alexandria-feedback-hints.test.mjs | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 77769611..98d268c4 100644 --- a/README.md +++ b/README.md @@ -1196,7 +1196,7 @@ Find Tools and execution are available on both the full surface and the search s ### Alexandria session feedback -Use the existing `firecrawl_feedback` tool with `endpoint: "alexandria"`: +The existing `firecrawl_feedback` tool accepts `endpoint: "alexandria"`: ```json { @@ -1212,4 +1212,4 @@ Use the existing `firecrawl_feedback` tool with `endpoint: "alexandria"`: This uses authenticated `POST /v2/feedback`, without a job ID, job-age deadline, or credit refund. Optional `providerFeedback` and `capabilityFeedback` arrays describe coverage gaps and execution issues; the tool schema lists supported issue values. A `new_capability_request` requires `requestedFunctionality`; `missing_capability` (the provider exists but lacks the capability) does not. Existing feedback opt-out and authentication controls apply. -Agents are pointed at this loop from three places: the server instructions, the `firecrawl_scrape` and `firecrawl_find_tools` descriptions, and a `feedbackTool` object attached to every Alexandria execution and discovery result (with the tool name and a skeleton of the arguments). The hint is omitted for Firecrawl-internal calls such as `bash` and when `firecrawl_feedback` is not registered (`FIRECRAWL_NO_ENDPOINT_FEEDBACK` or keyless startup). +Eligible Alexandria execution and discovery results include a `feedbackTool` pointer with the tool name and a skeleton of the arguments. The pointer is omitted for Firecrawl-internal calls such as `bash` and when `firecrawl_feedback` is not registered (`FIRECRAWL_NO_ENDPOINT_FEEDBACK` or keyless startup). diff --git a/src/alexandria-feedback.ts b/src/alexandria-feedback.ts index 6a8941be..4bfd0d29 100644 --- a/src/alexandria-feedback.ts +++ b/src/alexandria-feedback.ts @@ -59,7 +59,7 @@ export const alexandriaFeedbackFields = { /** Appended to Alexandria results so the pointer travels with the data the agent is reading. */ export const ALEXANDRIA_FEEDBACK_HINT = { name: 'firecrawl_feedback', - when: 'Available after task completion; at most once per website, including uncovered sites. Free; no job ID or deadline.', + when: 'Optional after task completion; at most once per website, including uncovered sites. Free; no job ID or deadline.', arguments: { endpoint: 'alexandria', rating: '', diff --git a/tests/mcp-alexandria-feedback-hints.test.mjs b/tests/mcp-alexandria-feedback-hints.test.mjs index fb7cce59..015ec90c 100644 --- a/tests/mcp-alexandria-feedback-hints.test.mjs +++ b/tests/mcp-alexandria-feedback-hints.test.mjs @@ -12,7 +12,7 @@ function assertFeedbackHint(payload) { assert.equal(payload.feedbackTool?.name, 'firecrawl_feedback'); assert.equal(payload.feedbackTool.arguments.endpoint, 'alexandria'); assert.match(payload.feedbackTool.arguments.requestedWebsite.url, /website/i); - assert.match(payload.feedbackTool.when, /^Available after task completion; at most once per website/i); + assert.match(payload.feedbackTool.when, /^Optional after task completion; at most once per website/i); } test('Alexandria selection metadata omits feedback workflow', async (t) => { From 24dbb29a6a00cd400af127b00d965a45cac06106 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 24 Sep 2026 22:33:36 -0500 Subject: [PATCH 7/9] docs(mcp): clarify Alexandria discovery and contract metadata --- src/alexandria.ts | 6 +++--- src/index.ts | 2 +- tests/mcp-alexandria-search.test.mjs | 1 + tests/mcp-description-budget.test.mjs | 2 +- tests/mcp-smoke.test.mjs | 4 ++-- 5 files changed, 8 insertions(+), 7 deletions(-) diff --git a/src/alexandria.ts b/src/alexandria.ts index 0a0bdd22..f91076d8 100644 --- a/src/alexandria.ts +++ b/src/alexandria.ts @@ -66,7 +66,7 @@ export const ALEXANDRIA_CATALOGUE_VERTICALS = export const ALEXANDRIA_CATALOGUE_SENTENCE = "Alexandria is Firecrawl's catalogue of data providers and workflows across " + ALEXANDRIA_CATALOGUE_VERTICALS + '; providers return typed, sourced records through published contracts.'; export const ALEXANDRIA_SOURCES_OPT_OUT = - 'Passing sources without alexandria in it (for example ["web"] or ["news"]) excludes Alexandria provider matches; omit sources unless you specifically need web-only or news-only results, or include "alexandria" alongside them.'; + 'A search with sources: ["web"] omits semantic provider discovery; domainTools: true can still return website-matched tools. Web-only results use domainTools: false.'; // Claude Code truncates each tool description at 2,048 characters, so the routing // copy that changes behaviour sits in the first lines of each description and the @@ -74,7 +74,7 @@ export const ALEXANDRIA_SOURCES_OPT_OUT = export const ALEXANDRIA_SEARCH_LEAD = 'Authenticated search also returns matching Alexandria data providers in data.tools (' + ALEXANDRIA_CATALOGUE_VERTICALS + '). Prefer a provider over scraping pages when the task needs the same fields across several entities, exact figures or timestamps, provenance, or many records; use web results when they already answer the question. ' + ALEXANDRIA_SOURCES_OPT_OUT; export const ALEXANDRIA_CONTRACT_GUIDANCE = - 'The selected contract defines required inputs, requiresOneOf groups (at least one member per group), example.request/example.response when present, and response.key (which may differ from records). Provider pagination uses the declared input and response cursor with the same filters; catalogue next is separate from provider pagination.'; + 'The selected contract marks required inputs and any requiresOneOf groups (at least one member per group); it may include example.request/example.response and response.key (which may differ from records). Where pagination is declared, its fields govern paging with the same filters; catalogue next is separate from provider pagination.'; export function findToolsOptions(args: z.infer) { const level = args.level ?? (args.query || args.capabilities?.length || args.providers?.length || args.groups?.length || args.urls?.length ? 'tools' : args.categories?.length ? 'providers' : 'categories'); @@ -105,4 +105,4 @@ export function withFindToolsNavigation(envelope: any) { } export const ALEXANDRIA_SEARCH_INSTRUCTIONS = - 'Authenticated search combines web results, semantic tool summaries and domain matches. Use sources: ["alexandria"] for semantic tools only, or sources: ["web"] for web only. domainTools: false disables domain matching. Tool matches describe available structured-data capabilities, not executed data. toolDetail: "compact" (default) returns only provider, capability and description; "summary" adds metadata; "full" includes their input and output contracts. firecrawl_scrape with an alexandria body executes a selected capability; firecrawl_find_tools provides catalogue browsing and full contracts.'; + 'Authenticated search combines web results, semantic tool summaries and domain matches. Use sources: ["alexandria"] for semantic tools only, or sources: ["web"] with domainTools: false for web only. Tool matches describe available structured-data capabilities, not executed data. toolDetail: "compact" (default) returns only provider, capability and description; "summary" adds metadata; "full" includes their input and output contracts. firecrawl_scrape with an alexandria body executes a selected capability; firecrawl_find_tools provides catalogue browsing and full contracts.'; diff --git a/src/index.ts b/src/index.ts index 3c2c397f..66822510 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1214,7 +1214,7 @@ const openAiAppsChallengeToken = normalizeHeader( ); const FULL_PROFILE_INSTRUCTIONS = - `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Alexandria requestId identifies one logical execution; repeated attempts of the identical payload require the same ID, including pending or uncertain executions. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Through firecrawl_scrape, terms/show displays the agreement and terms/accept records acceptance; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. A data request does not authorize acceptance. The dashboard URL provides an alternative for an organization admin. Retained results remain accessible without repeating successful provider calls. Provide only the required inputs and account for stated network or external side effects.`; + `Firecrawl provides web search, page retrieval, site URL discovery, multi-page collection, structured page data, monitoring, and multi-source research that returns structured data. Match the requested operation to the tool boundary: firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema, firecrawl_map enumerates URLs under a site without retrieving their content, and firecrawl_agent runs multi-source research and returns structured data when the URLs are not known or the answer spans several sites (an entity plus its fields, a list, a dataset); its result is read with firecrawl_agent_status. Authenticated firecrawl_search returns web results together with matching Alexandria providers in data.tools. ${ALEXANDRIA_CATALOGUE_SENTENCE} A matching provider can return the same fields across several entities, provenance, exact figures or timestamps, or a large set of records through firecrawl_scrape with its published contract. If web results already answer the question, use them. For the same fields across multiple pages, firecrawl_find_tools offers free provider discovery. Use firecrawl_find_tools to read a contract that was not returned in full or to browse the catalogue by category. ${ALEXANDRIA_SOURCES_OPT_OUT} If no provider fits, continue with web search or firecrawl_agent. For biomedical, life-science, clinical, or arXiv literature, the firecrawl_research_* tools search a paper index of abstracts and full text; firecrawl_search with categories: ["research"] is a website filter over ordinary web results and reaches different sources. For a programming question (code behaviour, a library or framework, an API contract, an error message, or a known bug), firecrawl_developer_search (or firecrawl_search with categories: ["developer"]) searches an index of public repositories, GitHub issues, merged pull requests, READMEs, and code documentation. Alexandria requestId identifies one logical execution; repeated attempts of the identical payload require the same ID, including pending or uncertain executions. firecrawl_search with sources: [{type: "alexandria"}] returns compact tool summaries in data.tools; toolDetail: "full" includes contracts, firecrawl_find_tools starts with categories, lists providers, then compact tools, and expands the selected full contract, and firecrawl_scrape with alexandria: [{provider, capability, options}] executes up to ten capabilities and returns their results. Alexandria access needs an API key on a team with it enabled. A terms-gated Alexandria provider fails with code THIRD_PARTY_DATA_TERMS_REQUIRED and a requiresAction.url. Through firecrawl_scrape, terms/show displays the agreement and terms/accept records acceptance; acceptance requires explicit user authorization for the reviewed version and digest and confirmed:true. A data request does not authorize acceptance. The dashboard URL provides an alternative for an organization admin. A retained result provides nextTool access without repeating a successful provider call. Provide only the required inputs and account for stated network or external side effects.`; const KEYLESS_PROFILE_INSTRUCTIONS = `Hosted keyless sessions expose firecrawl_search, firecrawl_scrape, and firecrawl_parse with usage limits. firecrawl_search searches the web. For programming questions, firecrawl_search with categories: ["developer"] searches indexed public repositories, GitHub issues, merged pull requests, repository READMEs, and code documentation. For biomedical, life-science, clinical, or arXiv literature, firecrawl_search with categories: ["research"] filters ordinary web results to research-affiliated websites. firecrawl_scrape retrieves one supplied page and can return JSON matching a supplied schema. firecrawl_parse processes supported local files through its two-phase upload flow. An Authorization bearer API key can provide higher usage limits and expose additional tools, subject to plan, deployment, and team policy, including firecrawl_map for site URL discovery, firecrawl_agent and firecrawl_agent_status for multi-source research that returns structured data when the URLs are not known, firecrawl_research_* for paper-index and repository research, and firecrawl_find_tools as the progressive Alexandria catalogue lookup alongside the Alexandria options of firecrawl_search and firecrawl_scrape for catalogued data providers.`; // The search surface exposes web/developer/research search plus the two Alexandria diff --git a/tests/mcp-alexandria-search.test.mjs b/tests/mcp-alexandria-search.test.mjs index db45aee5..759144d6 100644 --- a/tests/mcp-alexandria-search.test.mjs +++ b/tests/mcp-alexandria-search.test.mjs @@ -8,6 +8,7 @@ test('ordinary search defaults to web and both tool matches, with explicit opt-o [{}, ['web', 'alexandria'], true], [{ domainTools: false }, ['web', 'alexandria'], false], [{ sources: ['web'] }, ['web'], false], + [{ sources: ['web'], domainTools: true }, ['web'], true], [{ sources: ['alexandria'] }, ['alexandria'], false], [{ sources: [{ type: 'alexandria' }] }, [{ type: 'alexandria' }], false], [{ sources: ['alexandria'], domainTools: true }, ['alexandria'], true], diff --git a/tests/mcp-description-budget.test.mjs b/tests/mcp-description-budget.test.mjs index 0f1f4aa5..fe8da66e 100644 --- a/tests/mcp-description-budget.test.mjs +++ b/tests/mcp-description-budget.test.mjs @@ -19,7 +19,7 @@ test('every tool description fits the 2,048-character cap and keeps the routing const byName = new Map(tools.map((tool) => [tool.name, tool.description.trim()])); const search = byName.get('firecrawl_search'); assert.match(search, /Alexandria data providers in data\.tools/); - assert.match(search, /Passing sources without alexandria in it .* excludes Alexandria provider matches/); + assert.match(search, /sources: \["web"\] omits semantic provider discovery; domainTools: true can still return website-matched tools/); assert.match(search, /Prefer a provider over scraping pages/); const scrape = byName.get('firecrawl_scrape'); assert.match(scrape, /^Scrape one URL and return its content/); diff --git a/tests/mcp-smoke.test.mjs b/tests/mcp-smoke.test.mjs index 34b596f9..9f85a619 100644 --- a/tests/mcp-smoke.test.mjs +++ b/tests/mcp-smoke.test.mjs @@ -1134,14 +1134,14 @@ test('stdio transport initializes and lists Firecrawl tools', async (t) => { const instructionsHead = init.instructions.slice(0, CLAUDE_CODE_TEXT_CAP); assert.match(instructionsHead, /Alexandria is Firecrawl's catalogue of data providers/); assert.match(instructionsHead, /For the same fields across multiple pages/); - assert.match(instructionsHead, /Passing sources without alexandria in it/); + assert.match(instructionsHead, /sources: \["web"\] omits semantic provider discovery/); assert.match( init.instructions, /Alexandria is Firecrawl's catalogue of data providers and workflows.*firecrawl_scrape with alexandria.*executes up to ten capabilities/is ); assert.match( init.instructions, - /Passing sources without alexandria in it \(for example \["web"\] or \["news"\]\) excludes Alexandria provider matches; omit sources unless you specifically need web-only or news-only results, or include "alexandria" alongside them/ + /sources: \["web"\] omits semantic provider discovery; domainTools: true can still return website-matched tools. Web-only results use domainTools: false/ ); assert.match( init.instructions, From 2074ecf1a6e93f82db3c26020d0a171c64f94190 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 24 Sep 2026 22:38:03 -0500 Subject: [PATCH 8/9] docs(mcp): reuse shared Alexandria source guidance --- src/alexandria.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/alexandria.ts b/src/alexandria.ts index f91076d8..76f64325 100644 --- a/src/alexandria.ts +++ b/src/alexandria.ts @@ -105,4 +105,4 @@ export function withFindToolsNavigation(envelope: any) { } export const ALEXANDRIA_SEARCH_INSTRUCTIONS = - 'Authenticated search combines web results, semantic tool summaries and domain matches. Use sources: ["alexandria"] for semantic tools only, or sources: ["web"] with domainTools: false for web only. Tool matches describe available structured-data capabilities, not executed data. toolDetail: "compact" (default) returns only provider, capability and description; "summary" adds metadata; "full" includes their input and output contracts. firecrawl_scrape with an alexandria body executes a selected capability; firecrawl_find_tools provides catalogue browsing and full contracts.'; + 'Authenticated search combines web results, semantic tool summaries and domain matches. Use sources: ["alexandria"] for semantic tools only. ' + ALEXANDRIA_SOURCES_OPT_OUT + ' Tool matches describe available structured-data capabilities, not executed data. toolDetail: "compact" (default) returns only provider, capability and description; "summary" adds metadata; "full" includes their input and output contracts. firecrawl_scrape with an alexandria body executes a selected capability; firecrawl_find_tools provides catalogue browsing and full contracts.'; From 9c86a4817ef1d655069890d10df3f786e88b7758 Mon Sep 17 00:00:00 2001 From: Max Loffgren Date: Thu, 24 Sep 2026 22:52:15 -0500 Subject: [PATCH 9/9] test(mcp): cover Alexandria parameter metadata --- tests/helpers/alexandria-metadata.mjs | 17 +++++++++++++---- tests/mcp-search-profile.test.mjs | 8 ++++---- 2 files changed, 17 insertions(+), 8 deletions(-) diff --git a/tests/helpers/alexandria-metadata.mjs b/tests/helpers/alexandria-metadata.mjs index 137795d3..699f4223 100644 --- a/tests/helpers/alexandria-metadata.mjs +++ b/tests/helpers/alexandria-metadata.mjs @@ -1,8 +1,12 @@ import assert from 'node:assert/strict'; -export function assertAlexandriaMetadata(tools, instructions = '') { +export function assertAlexandriaMetadata(tools, instructions) { + assert.ok(instructions, 'Alexandria server instructions are required'); const scrape = tools.find((tool) => tool.name === 'firecrawl_scrape'); - const { requestId, alexandria } = scrape.inputSchema.properties; + assert.ok(scrape, 'firecrawl_scrape must be registered'); + const { requestId, alexandria } = scrape.inputSchema?.properties ?? {}; + assert.ok(requestId?.description, 'firecrawl_scrape.requestId needs a description'); + assert.ok(alexandria?.description, 'firecrawl_scrape.alexandria needs a description'); assert.equal(scrape.annotations.readOnlyHint, false); assert.match(requestId.description, /identical payload.*same ID/i); assert.match(requestId.description, /new ID cannot reconcile a pending or uncertain execution/i); @@ -15,9 +19,14 @@ export function assertAlexandriaMetadata(tools, instructions = '') { assert.match(alexandria.description, /data request does not authorize acceptance/i); assert.match(alexandria.description, /execution remains blocked until acceptance is confirmed/i); - const descriptions = [instructions, requestId.description, alexandria.description]; + const descriptions = [instructions]; for (const name of ['firecrawl_search', 'firecrawl_scrape', 'firecrawl_find_tools']) { - descriptions.push(tools.find((tool) => tool.name === name).description); + const tool = tools.find((item) => item.name === name); + assert.ok(tool, `${name} must be registered`); + descriptions.push(tool.description ?? ''); + for (const field of Object.values(tool.inputSchema?.properties ?? {})) { + if (field.description) descriptions.push(field.description); + } } for (const description of descriptions) { assert.doesNotMatch(description, /follow the returned terms\/show and terms\/accept|retry the same requestId later|call firecrawl_feedback once per website|after the task, report how the catalogue/i); diff --git a/tests/mcp-search-profile.test.mjs b/tests/mcp-search-profile.test.mjs index 825bed85..96ae1ab0 100644 --- a/tests/mcp-search-profile.test.mjs +++ b/tests/mcp-search-profile.test.mjs @@ -1303,10 +1303,10 @@ test('ordinary search profile enables semantic and domain tools by default', asy test('search surface registers the two Alexandria tools with surface-scoped copy', async (t) => { const { searchPort } = await startHostedServer(t); - const tools = await listToolDefinitions(searchPort, SEARCH_ENDPOINT, { - 'x-api-key': 'fc-test', - }); - assertAlexandriaMetadata(tools); + const headers = { 'x-api-key': 'fc-test' }; + const initialize = await initializeProfile(searchPort, SEARCH_ENDPOINT, headers); + const tools = await listToolDefinitions(searchPort, SEARCH_ENDPOINT, headers); + assertAlexandriaMetadata(tools, initialize.instructions); // Claude Code truncates tool descriptions at CLAUDE_CODE_TEXT_CAP characters. for (const tool of tools) { assert.ok((tool.description ?? '').length <= CLAUDE_CODE_TEXT_CAP, `${tool.name} description is ${(tool.description ?? '').length} chars`);