Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -1013,6 +1013,19 @@ firecrawl setup workflows

For more details, visit the [Firecrawl Documentation](https://docs.firecrawl.dev).

### Alexandria tool shorthand

Use a provider/capability in place of a URL. The explicit `--alexandria` form remains supported:

```bash
firecrawl scrape benzinga/news/search --options '{"pageSize":10}'
firecrawl scrape firecrawl-research-index/read --options '{"paperId":"123","query":"methodology","k":4}'
firecrawl scrape --alexandria benzinga/news/search --options '{"pageSize":10}'
firecrawl scrape https://example.com
```

URLs (including domains, IP addresses and localhost) continue to scrape websites. Tool addresses go directly to Alexandria, which validates the provider and capability; they never fall back to URL scraping. There is no extra catalog lookup. Bare names such as `firecrawl scrape amazon` fail locally with a suggested website URL and directions to `firecrawl list`. Suggestions are not verified or executed. Mixing URLs and tools in one command is rejected.

### Alexandria provider terms (beta)

When a provider returns `THIRD_PARTY_DATA_TERMS_REQUIRED`, review its linked terms.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "firecrawl-cli",
"version": "1.23.4-alexandria-beta.17",
"version": "1.23.4-alexandria-beta.18",
"publishConfig": {
"tag": "alexandria"
},
Expand Down
1 change: 1 addition & 0 deletions src/__tests__/alexandria-beta.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -528,6 +528,7 @@ it('retains successful results and billing when one provider fails', async () =>
'scrape',
'--alexandria',
'provider/lookup',
'--alexandria',
'other/lookup',
]);
expect(result.code).toBe(1);
Expand Down
20 changes: 9 additions & 11 deletions src/__tests__/commands/terms.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -29,17 +29,15 @@ it('presents the selected agreement and requests human approval without acceptin
});

it('preserves refusal details and gives actionable guidance without retrying', async () => {
const fetch = vi
.fn()
.mockResolvedValue(
Response.json(
{
code: 'forbidden',
error: 'This endpoint is not enabled for this team.',
},
{ status: 403 }
)
);
const fetch = vi.fn().mockResolvedValue(
Response.json(
{
code: 'forbidden',
error: 'This endpoint is not enabled for this team.',
},
{ status: 403 }
)
);
vi.stubGlobal('fetch', fetch);
const result = await requestTerms('particle', {});
expect(result).toMatchObject({
Expand Down
106 changes: 106 additions & 0 deletions src/__tests__/scrape-shorthand.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
import { spawn } from 'node:child_process';
import { createServer } from 'node:http';
import { resolve } from 'node:path';
import { expect, it } from 'vitest';

it('sends shorthand and explicit calls identically, rejects bare names locally, and preserves tool errors', async () => {
const requests: { path: string; body: any }[] = [];
const server = createServer(async (request, response) => {
const chunks: Buffer[] = [];
for await (const chunk of request) chunks.push(Buffer.from(chunk));
const body = JSON.parse(Buffer.concat(chunks).toString());
requests.push({ path: request.url!, body });
const call = body.alexandria[0];
const result =
call.provider === 'benzing'
? {
...call,
error: {
code: 'unknown_provider',
message: 'Unknown provider "benzing".',
status: 404,
},
}
: { ...call, data: { results: [] } };
response.setHeader('content-type', 'application/json');
response.end(
JSON.stringify({
success: true,
data: { alexandria: [result], creditsCost: 0 },
})
);
});
await new Promise<void>((done) => server.listen(0, '127.0.0.1', done));
const address = server.address() as { port: number };
const run = (args: string[]) =>
new Promise<{ code: number | null; stdout: string; stderr: string }>(
(done, reject) => {
const child = spawn(
process.execPath,
[resolve('dist/index.js'), 'scrape', ...args],
{
env: {
...process.env,
FIRECRAWL_API_URL: `http://127.0.0.1:${address.port}`,
FIRECRAWL_API_KEY: 'fc-local-test',
FIRECRAWL_NO_UPDATE_CHECK: '1',
FIRECRAWL_NO_TELEMETRY: '1',
},
stdio: ['ignore', 'pipe', 'pipe'],
}
);
let stdout = '',
stderr = '';
child.stdout.on('data', (chunk) => {
stdout += chunk;
});
child.stderr.on('data', (chunk) => {
stderr += chunk;
});
child.on('error', reject);
child.on('close', (code) => done({ code, stdout, stderr }));
}
);
try {
const common = [
'--options',
'{"pageSize":1}',
'--request-id',
'local-test',
];
expect((await run(['benzinga/news/search', ...common])).code).toBe(0);
expect(
(await run(['--alexandria', 'benzinga/news/search', ...common])).code
).toBe(0);
expect(requests).toHaveLength(2);
expect(requests[0]).toEqual(requests[1]);
expect(requests[0].body.url).toBeUndefined();
expect(requests[1].body.url).toBeUndefined();
expect(requests[0].path).toBe('/v2/scrape');
expect(requests[0].body.alexandria).toEqual([
Comment thread
cubic-dev-ai[bot] marked this conversation as resolved.
{
provider: 'benzinga',
capability: 'news/search',
options: { pageSize: 1 },
},
]);
const bare = await run(['amazon']);
expect(bare.code).toBe(1);
expect(bare.stderr).toContain('https://amazon.com (suggestion only)');
expect(bare.stderr).toContain('firecrawl list');
expect(requests).toHaveLength(2);
const bareUrl = await run(['--url', 'amazon']);
expect(bareUrl.code).toBe(1);
expect(bareUrl.stderr).toContain('https://amazon.com (suggestion only)');
expect(requests).toHaveLength(2);
const typo = await run(['benzing/news/search', ...common]);
expect(typo.code).toBe(1);
expect(typo.stdout).toContain('unknown_provider');
expect(requests).toHaveLength(3);
expect(requests[2].body.url).toBeUndefined();
} finally {
await new Promise<void>((done, reject) =>
server.close((error) => (error ? reject(error) : done()))
);
}
}, 15_000);
130 changes: 130 additions & 0 deletions src/__tests__/utils/scrape-target.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,130 @@
import { describe, expect, it } from 'vitest';
import { resolveScrapeTarget } from '../../utils/scrape-target';

describe('scrape target routing', () => {
it('routes shorthand and explicit tools through identical Alexandria calls', () => {
const options = ['{"query":"test","k":4}'];
const shorthand = resolveScrapeTarget(['firecrawl-research-index/read'], {
options,
});
expect(shorthand).toEqual(
resolveScrapeTarget([], {
alexandria: ['firecrawl-research-index/read'],
options,
})
);
expect(shorthand).toMatchObject({
kind: 'alexandria',
calls: [
{
provider: 'firecrawl-research-index',
capability: 'read',
options: { query: 'test', k: 4 },
},
],
});
expect(resolveScrapeTarget(['benzing/news/serch'], {}).kind).toBe(
'alexandria'
);
});

it.each([
'https://example.com/a',
'http://internal/a',
'example.com/a',
'example.com:8080/a?x=1',
'localhost:3000/a',
'localhost',
'127.0.0.1:3000/a',
'[::1]:3000/a',
])('keeps %s on URL scraping', (target) => {
expect(resolveScrapeTarget([target], {}).kind).toBe('url');
expect(resolveScrapeTarget([], { url: target }).kind).toBe('url');
});

it.each(['provider_name/search', 'provider~name/search', '_provider/search'])(
'accepts supported address characters in %s',
(address) => {
expect(resolveScrapeTarget([address], {})).toEqual(
resolveScrapeTarget([], { alexandria: [address] })
);
}
);

it.each(['amazon', 'provider/search', 'https://', ''])(
'rejects invalid --url %s locally',
(url) => {
expect(() => resolveScrapeTarget([], { url })).toThrow('firecrawl list');
}
);

it('preserves multiple URLs and positional output formats', () => {
expect(
resolveScrapeTarget(
['example.com/a', 'example.org', 'Markdown, links'],
{}
)
).toMatchObject({
kind: 'url',
urls: ['https://example.com/a', 'https://example.org'],
positionalFormats: ['Markdown, links'],
});
});

it('pairs multiple tool addresses with their options in order', () => {
expect(
resolveScrapeTarget(
['benzinga/news/search', 'firecrawl-research-index/search'],
{ options: ['{"pageSize":1}', '{"query":"attention"}'] }
)
).toMatchObject({
calls: [
{ provider: 'benzinga', options: { pageSize: 1 } },
{
provider: 'firecrawl-research-index',
options: { query: 'attention' },
},
],
});
});

it.each([
'amazon',
'amazon/',
'/news/search',
'https://',
'provider//search',
])('rejects ambiguous or malformed %s locally', (value) => {
expect(() => resolveScrapeTarget([value], {})).toThrow('firecrawl list');
});

it('suggests both URL and Alexandria paths for a bare name', () => {
expect(() => resolveScrapeTarget(['amazon'], {})).toThrow(
'https://amazon.com (suggestion only)'
);
expect(() => resolveScrapeTarget(['amazon'], {})).toThrow('--alexandria');
});

it('refuses mixed modes and malformed options', () => {
expect(() =>
resolveScrapeTarget(['example.com', 'benzinga/news/search'], {})
).toThrow('cannot be combined');
expect(() =>
resolveScrapeTarget(['benzinga/news/search'], {
alexandria: ['benzinga/news/search'],
})
).toThrow('not both');
expect(() =>
resolveScrapeTarget(['amazon'], { alexandria: ['benzinga/news/search'] })
).toThrow('firecrawl list');
expect(() =>
resolveScrapeTarget(['benzinga/news/search'], { domainTools: true })
).toThrow('cannot be combined');
expect(() =>
resolveScrapeTarget(['benzinga/news/search'], { options: ['[]'] })
).toThrow('JSON object');
expect(() =>
resolveScrapeTarget(['example.com'], { options: ['{}'] })
).toThrow('require a provider/capability');
});
});
2 changes: 1 addition & 1 deletion src/commands/alexandria.ts
Original file line number Diff line number Diff line change
Expand Up @@ -245,7 +245,7 @@ export function addAlexandriaScrapeOptions(command: Command): void {
.addOption(
new Option(
'--options <json>',
'Input object for each --alexandria call, in matching order'
'Input object for each provider/capability call, in matching order'
).argParser((value: string, previous: string[] = []) => [
...previous,
value,
Expand Down
Loading
Loading