From 77b0b74dfef703a16dd3b72c7046fef4608def83 Mon Sep 17 00:00:00 2001 From: Developers Digest <124798203+developersdigest@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:28:41 -0400 Subject: [PATCH 1/4] Support Alexandria tool paths as scrape shorthand --- README.md | 13 +++ src/__tests__/alexandria-beta.test.ts | 1 + src/__tests__/scrape-shorthand.test.ts | 100 +++++++++++++++++++ src/__tests__/utils/scrape-target.test.ts | 113 ++++++++++++++++++++++ src/commands/alexandria.ts | 2 +- src/index.ts | 64 +++++------- src/utils/scrape-target.ts | 99 +++++++++++++++++++ 7 files changed, 349 insertions(+), 43 deletions(-) create mode 100644 src/__tests__/scrape-shorthand.test.ts create mode 100644 src/__tests__/utils/scrape-target.test.ts create mode 100644 src/utils/scrape-target.ts diff --git a/README.md b/README.md index 3caf3bb47b..efcc3c8e7e 100644 --- a/README.md +++ b/README.md @@ -1013,6 +1013,19 @@ firecrawl setup workflows For more details, visit the [Firecrawl Documentation](https://docs.firecrawl.dev). +### Alexandria tool shorthand + +Use a provider/capability in place of a URL. The explicit `--alexandria` form remains supported: + +```bash +firecrawl scrape benzinga/news/search --options '{"pageSize":10}' +firecrawl scrape firecrawl-research-index/read --options '{"paperId":"123","query":"methodology","k":4}' +firecrawl scrape --alexandria benzinga/news/search --options '{"pageSize":10}' +firecrawl scrape https://example.com +``` + +URLs (including domains, IP addresses and localhost) continue to scrape websites. Tool addresses go directly to Alexandria, which validates the provider and capability; they never fall back to URL scraping. There is no extra catalog lookup. Bare names such as `firecrawl scrape amazon` fail locally with a suggested website URL and directions to `firecrawl list`. Suggestions are not verified or executed. Mixing URLs and tools in one command is rejected. + ### Alexandria provider terms (beta) When a provider returns `THIRD_PARTY_DATA_TERMS_REQUIRED`, review its linked terms. diff --git a/src/__tests__/alexandria-beta.test.ts b/src/__tests__/alexandria-beta.test.ts index e84e17d4de..890ab2cc36 100644 --- a/src/__tests__/alexandria-beta.test.ts +++ b/src/__tests__/alexandria-beta.test.ts @@ -528,6 +528,7 @@ it('retains successful results and billing when one provider fails', async () => 'scrape', '--alexandria', 'provider/lookup', + '--alexandria', 'other/lookup', ]); expect(result.code).toBe(1); diff --git a/src/__tests__/scrape-shorthand.test.ts b/src/__tests__/scrape-shorthand.test.ts new file mode 100644 index 0000000000..4217619777 --- /dev/null +++ b/src/__tests__/scrape-shorthand.test.ts @@ -0,0 +1,100 @@ +import { spawn } from 'node:child_process'; +import { createServer } from 'node:http'; +import { resolve } from 'node:path'; +import { expect, it } from 'vitest'; + +it('sends shorthand and explicit calls identically, rejects bare names locally, and preserves tool errors', async () => { + const requests: { path: string; body: any }[] = []; + const server = createServer(async (request, response) => { + const chunks: Buffer[] = []; + for await (const chunk of request) chunks.push(Buffer.from(chunk)); + const body = JSON.parse(Buffer.concat(chunks).toString()); + requests.push({ path: request.url!, body }); + const call = body.alexandria[0]; + const result = + call.provider === 'benzing' + ? { + ...call, + error: { + code: 'unknown_provider', + message: 'Unknown provider "benzing".', + status: 404, + }, + } + : { ...call, data: { results: [] } }; + response.setHeader('content-type', 'application/json'); + response.end( + JSON.stringify({ + success: true, + data: { alexandria: [result], creditsCost: 0 }, + }) + ); + }); + await new Promise((done) => server.listen(0, '127.0.0.1', done)); + const address = server.address() as { port: number }; + const run = (args: string[]) => + new Promise<{ code: number | null; stdout: string; stderr: string }>( + (done, reject) => { + const child = spawn( + process.execPath, + [resolve('dist/index.js'), 'scrape', ...args], + { + env: { + ...process.env, + FIRECRAWL_API_URL: `http://127.0.0.1:${address.port}`, + FIRECRAWL_API_KEY: 'fc-local-test', + FIRECRAWL_NO_UPDATE_CHECK: '1', + FIRECRAWL_NO_TELEMETRY: '1', + }, + stdio: ['ignore', 'pipe', 'pipe'], + } + ); + let stdout = '', + stderr = ''; + child.stdout.on('data', (chunk) => { + stdout += chunk; + }); + child.stderr.on('data', (chunk) => { + stderr += chunk; + }); + child.on('error', reject); + child.on('close', (code) => done({ code, stdout, stderr })); + } + ); + try { + const common = [ + '--options', + '{"pageSize":1}', + '--request-id', + 'local-test', + ]; + expect((await run(['benzinga/news/search', ...common])).code).toBe(0); + expect( + (await run(['--alexandria', 'benzinga/news/search', ...common])).code + ).toBe(0); + expect(requests).toHaveLength(2); + expect(requests[0]).toEqual(requests[1]); + expect(requests[0].path).toBe('/v2/scrape'); + expect(requests[0].body.alexandria).toEqual([ + { + provider: 'benzinga', + capability: 'news/search', + options: { pageSize: 1 }, + }, + ]); + const bare = await run(['amazon']); + expect(bare.code).toBe(1); + expect(bare.stderr).toContain('https://amazon.com (suggestion only)'); + expect(bare.stderr).toContain('firecrawl list'); + expect(requests).toHaveLength(2); + const typo = await run(['benzing/news/search', ...common]); + expect(typo.code).toBe(1); + expect(typo.stdout).toContain('unknown_provider'); + expect(requests).toHaveLength(3); + expect(requests[2].body.url).toBeUndefined(); + } finally { + await new Promise((done, reject) => + server.close((error) => (error ? reject(error) : done())) + ); + } +}, 15_000); diff --git a/src/__tests__/utils/scrape-target.test.ts b/src/__tests__/utils/scrape-target.test.ts new file mode 100644 index 0000000000..fc56fffd88 --- /dev/null +++ b/src/__tests__/utils/scrape-target.test.ts @@ -0,0 +1,113 @@ +import { describe, expect, it } from 'vitest'; +import { resolveScrapeTarget } from '../../utils/scrape-target'; + +describe('scrape target routing', () => { + it('routes shorthand and explicit tools through identical Alexandria calls', () => { + const options = ['{"query":"test","k":4}']; + const shorthand = resolveScrapeTarget(['firecrawl-research-index/read'], { + options, + }); + expect(shorthand).toEqual( + resolveScrapeTarget([], { + alexandria: ['firecrawl-research-index/read'], + options, + }) + ); + expect(shorthand).toMatchObject({ + kind: 'alexandria', + calls: [ + { + provider: 'firecrawl-research-index', + capability: 'read', + options: { query: 'test', k: 4 }, + }, + ], + }); + expect(resolveScrapeTarget(['benzing/news/serch'], {}).kind).toBe( + 'alexandria' + ); + }); + + it.each([ + 'https://example.com/a', + 'http://internal/a', + 'example.com/a', + 'example.com:8080/a?x=1', + 'localhost:3000/a', + 'localhost', + '127.0.0.1:3000/a', + '[::1]:3000/a', + ])('keeps %s on URL scraping', (target) => { + expect(resolveScrapeTarget([target], {}).kind).toBe('url'); + }); + + it('preserves multiple URLs and positional output formats', () => { + expect( + resolveScrapeTarget( + ['example.com/a', 'example.org', 'Markdown, links'], + {} + ) + ).toMatchObject({ + kind: 'url', + urls: ['https://example.com/a', 'https://example.org'], + positionalFormats: ['Markdown, links'], + }); + }); + + it('pairs multiple tool addresses with their options in order', () => { + expect( + resolveScrapeTarget( + ['benzinga/news/search', 'firecrawl-research-index/search'], + { options: ['{"pageSize":1}', '{"query":"attention"}'] } + ) + ).toMatchObject({ + calls: [ + { provider: 'benzinga', options: { pageSize: 1 } }, + { + provider: 'firecrawl-research-index', + options: { query: 'attention' }, + }, + ], + }); + }); + + it.each([ + 'amazon', + 'amazon/', + '/news/search', + 'https://', + 'provider//search', + ])('rejects ambiguous or malformed %s locally', (value) => { + expect(() => resolveScrapeTarget([value], {})).toThrow('firecrawl list'); + }); + + it('suggests both URL and Alexandria paths for a bare name', () => { + expect(() => resolveScrapeTarget(['amazon'], {})).toThrow( + 'https://amazon.com (suggestion only)' + ); + expect(() => resolveScrapeTarget(['amazon'], {})).toThrow('--alexandria'); + }); + + it('refuses mixed modes and malformed options', () => { + expect(() => + resolveScrapeTarget(['example.com', 'benzinga/news/search'], {}) + ).toThrow('cannot be combined'); + expect(() => + resolveScrapeTarget(['benzinga/news/search'], { + alexandria: ['benzinga/news/search'], + }) + ).toThrow('not both'); + expect(() => + resolveScrapeTarget(['amazon'], { alexandria: ['benzinga/news/search'] }) + ).toThrow('firecrawl list'); + expect(() => + resolveScrapeTarget(['benzinga/news/search'], { domainTools: true }) + ).toThrow('cannot be combined'); + expect(() => + resolveScrapeTarget(['benzinga/news/search'], { options: ['[]'] }) + ).toThrow('JSON object'); + expect(() => + resolveScrapeTarget(['example.com'], { options: ['{}'] }) + ).toThrow('require a provider/capability'); + }); +}); diff --git a/src/commands/alexandria.ts b/src/commands/alexandria.ts index 0ca1a11a71..812ee84764 100644 --- a/src/commands/alexandria.ts +++ b/src/commands/alexandria.ts @@ -245,7 +245,7 @@ export function addAlexandriaScrapeOptions(command: Command): void { .addOption( new Option( '--options ', - 'Input object for each --alexandria call, in matching order' + 'Input object for each provider/capability call, in matching order' ).argParser((value: string, previous: string[] = []) => [ ...previous, value, diff --git a/src/index.ts b/src/index.ts index ecb1ba8244..2ff2bdb722 100644 --- a/src/index.ts +++ b/src/index.ts @@ -9,7 +9,6 @@ import { Command, Option } from 'commander'; import { addFormatsAlias } from './utils/format-option'; import { addAlexandriaScrapeOptions, - buildCalls, createFindToolsCommand, handleAlexandria, } from './commands/alexandria'; @@ -71,6 +70,7 @@ import { handleEnvPullCommand } from './commands/env'; import { handleStatusCommand } from './commands/status'; import { handleDoctorCommand } from './commands/doctor'; import { isUrl, normalizeUrl } from './utils/url'; +import { resolveScrapeTarget } from './utils/scrape-target'; import { parseMaxPages, parseScrapeOptions } from './utils/options'; import { isJobId } from './utils/job'; import { ensureAuthenticated, printBanner } from './utils/auth'; @@ -338,6 +338,8 @@ program // Check if this command requires authentication const commandName = actionCommand.name(); + if (commandName === 'scrape') + resolveScrapeTarget(actionCommand.args, commandOptions); if (AUTH_REQUIRED_COMMANDS.includes(commandName)) { // Skip auth for custom API URLs (e.g., local development) // Check both global and command-level options @@ -356,9 +358,9 @@ program function createScrapeCommand(): Command { const scrapeCmd = new Command('scrape') .description( - 'Scrape one or more URLs. Multiple URLs are scraped concurrently and saved to .firecrawl/' + 'Scrape URLs or execute Alexandria provider/capability tools. Multiple URLs are saved to .firecrawl/' ) - .argument('[urls...]', 'URL(s) to scrape') + .argument('[urls...]', 'URL(s) or provider/capability tool address(es)') .option( '-u, --url ', 'URL to scrape (alternative to positional argument)' @@ -435,44 +437,12 @@ function createScrapeCommand(): Command { .option('--proxy ', 'Proxy mode for scraping (e.g., auto, basic)') .action(async (positionalArgs, options) => { - // Collect URLs from positional args and --url option - let urls: string[] = []; - - if (positionalArgs && positionalArgs.length > 0) { - for (const arg of positionalArgs) { - if (isUrl(arg)) { - urls.push(normalizeUrl(arg)); - } - } - } - - if (options.url) { - urls.push(normalizeUrl(options.url)); - } - - // Remove duplicates - urls = [...new Set(urls)]; - - if (options.alexandria) { - if (urls.length || options.domainTools) - throw new Error( - 'Provider execution cannot be combined with URL scraping.' - ); - await handleAlexandria( - buildCalls(options.alexandria, options.options), - options - ); + const target = resolveScrapeTarget(positionalArgs ?? [], options); + if (target.kind === 'alexandria') { + await handleAlexandria(target.calls, options); return; } - if (options.options || options.requestId) - throw new Error('--options and --request-id require --alexandria.'); - - if (urls.length === 0) { - console.error( - 'Error: URL is required. Provide it as argument or use --url option.' - ); - process.exit(1); - } + const { urls, positionalFormats } = target; let schema: Record | undefined; let actions: Record[] | undefined; @@ -504,9 +474,6 @@ function createScrapeCommand(): Command { // Determine format let format: string; - const positionalFormats = (positionalArgs || []).filter( - (arg: string) => !isUrl(arg) - ); if (positionalFormats.length > 0) { format = positionalFormats.join(','); } else if (options.html) { @@ -538,6 +505,19 @@ function createScrapeCommand(): Command { } }); + scrapeCmd.addHelpText( + 'after', + ` +Examples: + firecrawl scrape https://example.com + firecrawl scrape example.com markdown + firecrawl scrape benzinga/news/search --options '{"pageSize":10}' + firecrawl scrape --alexandria benzinga/news/search --options '{"pageSize":10}' + +Bare names such as "amazon" show guidance without a lookup or execution. +Tool addresses are validated by Alexandria; unknown tools never fall back to URL scraping. +` + ); addAlexandriaScrapeOptions(scrapeCmd); return addFormatsAlias(scrapeCmd); } diff --git a/src/utils/scrape-target.ts b/src/utils/scrape-target.ts new file mode 100644 index 0000000000..7bcef2c534 --- /dev/null +++ b/src/utils/scrape-target.ts @@ -0,0 +1,99 @@ +import { buildCalls } from '../commands/alexandria'; +import { normalizeUrl } from './url'; +import { parseFormats } from './options'; + +type ScrapeTargetOptions = { + url?: string; + alexandria?: string[]; + options?: string[]; + requestId?: string; + domainTools?: boolean; +}; + +function isPositionalFormat(value: string): boolean { + try { + return parseFormats(value).length > 0; + } catch { + return false; + } +} + +function isScrapeUrl(value: string): boolean { + if (/^https?:\/\//i.test(value)) { + try { + return Boolean(new URL(value).hostname); + } catch { + return false; + } + } + const host = value.split(/[/?#]/, 1)[0]; + if ( + !host.includes('.') && + !/^localhost(?::\d+)?$/i.test(host) && + !host.startsWith('[') + ) + return false; + try { + return Boolean(new URL(normalizeUrl(value)).hostname); + } catch { + return false; + } +} + +function ambiguous(value: string): never { + const suggestion = /^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?$/i.test(value) + ? `\nIf you meant a website, try https://${value}.com (suggestion only).` + : '\nFor a website, use an http(s) URL, for example https://example.com.'; + throw new Error( + `${JSON.stringify(value)} needs a URL or provider/capability.${suggestion}\n` + + 'Browse Alexandria providers and tools: firecrawl list\n' + + 'Run a tool: firecrawl scrape / --options \'{"query":"..."}\'\n' + + 'You can also use --alexandria /.' + ); +} + +/** Resolve intent locally; Exchange validates provider and capability existence. */ +export function resolveScrapeTarget( + args: string[], + options: ScrapeTargetOptions +) { + const urls: string[] = []; + const tools: string[] = []; + const positionalFormats: string[] = []; + const unknown: string[] = []; + for (const arg of args) { + if (isScrapeUrl(arg)) urls.push(normalizeUrl(arg)); + else if ( + /^[a-z0-9][a-z0-9-]*\/[a-z0-9_~.-]+(?:\/[a-z0-9_~.-]+)*$/i.test(arg) + ) + tools.push(arg); + else if (isPositionalFormat(arg)) positionalFormats.push(arg); + else unknown.push(arg); + } + if (options.url) urls.push(normalizeUrl(options.url)); + if (unknown.length) ambiguous(unknown[0]); + if (tools.length && options.alexandria?.length) + throw new Error('Use positional tool addresses or --alexandria, not both.'); + const addresses = options.alexandria ?? tools; + if (addresses.length) { + if (urls.length || options.domainTools || positionalFormats.length) + throw new Error( + 'Provider execution cannot be combined with URL scraping or positional output formats.' + ); + return { + kind: 'alexandria' as const, + calls: buildCalls(addresses, options.options), + }; + } + if (!urls.length) { + if (positionalFormats.length) ambiguous(positionalFormats[0]); + throw new Error( + 'Provide a URL or provider/capability. Browse Alexandria tools with firecrawl list.' + ); + } + if (options.options || options.requestId) + throw new Error( + '--options and --request-id require a provider/capability or --alexandria.' + ); + return { kind: 'url' as const, urls: [...new Set(urls)], positionalFormats }; +} From 90c0c818a67bac1bd883a03fa9e321d8ce48b3fb Mon Sep 17 00:00:00 2001 From: Developers Digest <124798203+developersdigest@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:31:04 -0400 Subject: [PATCH 2/4] chore: bump Alexandria CLI prerelease to beta.18 --- package.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/package.json b/package.json index 1d1c3e6df4..7f4a073786 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "firecrawl-cli", - "version": "1.23.4-alexandria-beta.17", + "version": "1.23.4-alexandria-beta.18", "publishConfig": { "tag": "alexandria" }, From 715bdac3231ea58c10a65c6d8cccda93f3d16d8e Mon Sep 17 00:00:00 2001 From: Developers Digest <124798203+developersdigest@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:38:31 -0400 Subject: [PATCH 3/4] fix: validate scrape targets consistently --- src/__tests__/scrape-shorthand.test.ts | 6 ++++++ src/__tests__/utils/scrape-target.test.ts | 17 +++++++++++++++++ src/utils/scrape-target.ts | 9 +++++---- 3 files changed, 28 insertions(+), 4 deletions(-) diff --git a/src/__tests__/scrape-shorthand.test.ts b/src/__tests__/scrape-shorthand.test.ts index 4217619777..d2673ad3dc 100644 --- a/src/__tests__/scrape-shorthand.test.ts +++ b/src/__tests__/scrape-shorthand.test.ts @@ -74,6 +74,8 @@ it('sends shorthand and explicit calls identically, rejects bare names locally, ).toBe(0); expect(requests).toHaveLength(2); expect(requests[0]).toEqual(requests[1]); + expect(requests[0].body.url).toBeUndefined(); + expect(requests[1].body.url).toBeUndefined(); expect(requests[0].path).toBe('/v2/scrape'); expect(requests[0].body.alexandria).toEqual([ { @@ -87,6 +89,10 @@ it('sends shorthand and explicit calls identically, rejects bare names locally, expect(bare.stderr).toContain('https://amazon.com (suggestion only)'); expect(bare.stderr).toContain('firecrawl list'); expect(requests).toHaveLength(2); + const bareUrl = await run(['--url', 'amazon']); + expect(bareUrl.code).toBe(1); + expect(bareUrl.stderr).toContain('https://amazon.com (suggestion only)'); + expect(requests).toHaveLength(2); const typo = await run(['benzing/news/search', ...common]); expect(typo.code).toBe(1); expect(typo.stdout).toContain('unknown_provider'); diff --git a/src/__tests__/utils/scrape-target.test.ts b/src/__tests__/utils/scrape-target.test.ts index fc56fffd88..c3c9373d57 100644 --- a/src/__tests__/utils/scrape-target.test.ts +++ b/src/__tests__/utils/scrape-target.test.ts @@ -39,8 +39,25 @@ describe('scrape target routing', () => { '[::1]:3000/a', ])('keeps %s on URL scraping', (target) => { expect(resolveScrapeTarget([target], {}).kind).toBe('url'); + expect(resolveScrapeTarget([], { url: target }).kind).toBe('url'); }); + it.each(['provider_name/search', 'provider~name/search', '_provider/search'])( + 'accepts supported address characters in %s', + (address) => { + expect(resolveScrapeTarget([address], {})).toEqual( + resolveScrapeTarget([], { alexandria: [address] }) + ); + } + ); + + it.each(['amazon', 'provider/search', 'https://', ''])( + 'rejects invalid --url %s locally', + (url) => { + expect(() => resolveScrapeTarget([], { url })).toThrow('firecrawl list'); + } + ); + it('preserves multiple URLs and positional output formats', () => { expect( resolveScrapeTarget( diff --git a/src/utils/scrape-target.ts b/src/utils/scrape-target.ts index 7bcef2c534..3f97627138 100644 --- a/src/utils/scrape-target.ts +++ b/src/utils/scrape-target.ts @@ -63,14 +63,15 @@ export function resolveScrapeTarget( const unknown: string[] = []; for (const arg of args) { if (isScrapeUrl(arg)) urls.push(normalizeUrl(arg)); - else if ( - /^[a-z0-9][a-z0-9-]*\/[a-z0-9_~.-]+(?:\/[a-z0-9_~.-]+)*$/i.test(arg) - ) + else if (/^[a-z0-9_~.-]+\/[a-z0-9_~.-]+(?:\/[a-z0-9_~.-]+)*$/i.test(arg)) tools.push(arg); else if (isPositionalFormat(arg)) positionalFormats.push(arg); else unknown.push(arg); } - if (options.url) urls.push(normalizeUrl(options.url)); + if (options.url !== undefined) { + if (!isScrapeUrl(options.url)) ambiguous(options.url); + urls.push(normalizeUrl(options.url)); + } if (unknown.length) ambiguous(unknown[0]); if (tools.length && options.alexandria?.length) throw new Error('Use positional tool addresses or --alexandria, not both.'); From 0f514060b9c444732cfc3a14fbcc81f46db96cc1 Mon Sep 17 00:00:00 2001 From: Developers Digest <124798203+developersdigest@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:34:31 -0400 Subject: [PATCH 4/4] style: format terms command tests --- src/__tests__/commands/terms.test.ts | 20 +++++++++----------- 1 file changed, 9 insertions(+), 11 deletions(-) diff --git a/src/__tests__/commands/terms.test.ts b/src/__tests__/commands/terms.test.ts index a833bcd294..fa3c62ff28 100644 --- a/src/__tests__/commands/terms.test.ts +++ b/src/__tests__/commands/terms.test.ts @@ -29,17 +29,15 @@ it('presents the selected agreement and requests human approval without acceptin }); it('preserves refusal details and gives actionable guidance without retrying', async () => { - const fetch = vi - .fn() - .mockResolvedValue( - Response.json( - { - code: 'forbidden', - error: 'This endpoint is not enabled for this team.', - }, - { status: 403 } - ) - ); + const fetch = vi.fn().mockResolvedValue( + Response.json( + { + code: 'forbidden', + error: 'This endpoint is not enabled for this team.', + }, + { status: 403 } + ) + ); vi.stubGlobal('fetch', fetch); const result = await requestTerms('particle', {}); expect(result).toMatchObject({