diff --git a/README.md b/README.md index 33d1433080..f68dc29107 100644 --- a/README.md +++ b/README.md @@ -281,7 +281,7 @@ firecrawl https://example.com --exclude-tags nav,aside,.ad ### `search` - Search the web -Search the web and optionally scrape content from search results. +Search the web with query-relevant highlights and optionally scrape content from search results. ```bash # Basic search @@ -299,15 +299,14 @@ firecrawl search "landscape photography" --sources images # Multiple sources firecrawl search "machine learning" --sources web,news,images -# Filter by category (GitHub, research-affiliated websites, PDFs) -firecrawl search "web data python" --categories github +# Filter by category (research-affiliated websites, PDFs, developer index) firecrawl search "transformer architecture" --categories research -firecrawl search "machine learning" --categories github,research +firecrawl search "machine learning" --categories pdf,research # Note: --categories research narrows *web* results to research-affiliated # websites. To search papers themselves, use `firecrawl research search-papers`. -# Developer search: GitHub issues, merged PRs, READMEs, and docs +# Developer search: public repositories, GitHub issues, merged PRs, READMEs, and docs firecrawl search "axum middleware ordering" --categories developer # Time-based search @@ -328,23 +327,23 @@ firecrawl search "AI data tools" #### Search Options -| Option | Description | -| ---------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `--limit ` | Maximum results (default: 5, max: 100) | -| `--sources ` | Comma-separated: `web`, `images`, `news` (default: web) | -| `--categories ` | Comma-separated: `github`, `research` (research-affiliated websites -- for papers use [`research search-papers`](#research---search-research-papers)), `pdf`, `developer` | -| `--tbs ` | Time filter: `qdr:h` (hour), `qdr:d` (day), `qdr:w` (week), `qdr:m` (month), `qdr:y` (year) | -| `--location ` | Geo-targeting (e.g., "Germany", "San Francisco,California,United States") | -| `--country ` | ISO country code (default: US) | -| `--timeout ` | Timeout in milliseconds (default: 60000) | -| `--highlights` | Return query-relevant highlights for each result | -| `--no-highlights` | Keep the original search snippets | -| `--ignore-invalid-urls` | Exclude URLs invalid for other Firecrawl endpoints | -| `--scrape` | Enable scraping of search results | -| `--scrape-formats ` | Scrape formats when `--scrape` enabled (default: markdown) | -| `--only-main-content` | Include only main content when scraping (default: true) | -| `-o, --output ` | Save to file | -| `--json` | Output as compact JSON | +| Option | Description | +| ---------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `--limit ` | Maximum results (default: 5, max: 100) | +| `--sources ` | Comma-separated: `web`, `images`, `news` (default: web) | +| `--categories ` | Comma-separated: `research` (research-affiliated websites -- for papers use [`research search-papers`](#research---search-research-papers)), `pdf`, `developer` | +| `--tbs ` | Time filter: `qdr:h` (hour), `qdr:d` (day), `qdr:w` (week), `qdr:m` (month), `qdr:y` (year) | +| `--location ` | Geo-targeting (e.g., "Germany", "San Francisco,California,United States") | +| `--country ` | ISO country code (default: US) | +| `--timeout ` | Timeout in milliseconds (default: 60000) | +| `--highlights` | Query-relevant highlights for web and news when available (default) | +| `--no-highlights` | Keep the original search snippets | +| `--ignore-invalid-urls` | Exclude URLs invalid for other Firecrawl endpoints | +| `--scrape` | Enable scraping of search results | +| `--scrape-formats ` | Scrape formats when `--scrape` enabled (default: markdown) | +| `--only-main-content` | Include only main content when scraping (default: true) | +| `-o, --output ` | Save to file | +| `--json` | Output as compact JSON | #### Examples @@ -352,8 +351,8 @@ firecrawl search "AI data tools" # Research a topic with recent results firecrawl search "React Server Components" --tbs qdr:m --limit 10 -# Find GitHub repositories -firecrawl search "web data library" --categories github --limit 20 +# Search public developer sources +firecrawl search "web data library" --categories developer --limit 20 # Search and get full content firecrawl search "firecrawl documentation" --scrape --scrape-formats markdown --json -o results.json @@ -364,7 +363,7 @@ firecrawl research search-papers "large language models" --json # Narrow web results to research-affiliated websites (not the paper index) firecrawl search "large language models" --categories research --json -# Answer a programming question from issues, merged PRs, READMEs, and docs +# Answer a programming question from public repositories, GitHub issues, merged PRs, READMEs, and docs firecrawl search "tokio select cancellation safety" --categories developer --json # Search with location targeting @@ -378,7 +377,7 @@ firecrawl search "AI startups funding" --sources news --tbs qdr:w --limit 15 ### `developer` - Search developer sources -Search an index built for coding agents: GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. Use it for a programming question: code behaviour, a library or framework, an API contract, an error message, or a known bug. +Search an index built for coding agents: public repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. Use it for a programming question: code behaviour, a library or framework, an API contract, an error message, or a known bug. The CLI intentionally keeps this agent-facing surface lean: it accepts only the query and result count. Express repository, source, result-kind, language, topic, license, and other scoping intent in the query text; semantic retrieval handles the scoping. For advanced filters, use the [Developer Index REST API](https://docs.firecrawl.dev/features/developer). diff --git a/skills/firecrawl-developer-index/SKILL.md b/skills/firecrawl-developer-index/SKILL.md index affd8619c2..eb6444d783 100644 --- a/skills/firecrawl-developer-index/SKILL.md +++ b/skills/firecrawl-developer-index/SKILL.md @@ -1,6 +1,6 @@ --- name: firecrawl-developer-index -description: Search issues, merged pull requests, and READMEs from public code repositories, plus curated documentation. Use when the question is how a library or API behaves, what an error means, or whether a bug was fixed; prefer this over a general web page. +description: Search an index of public repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. Use when a programming question needs external documentation or upstream evidence. --- # Firecrawl Developer Index diff --git a/skills/firecrawl-search/SKILL.md b/skills/firecrawl-search/SKILL.md index 5f2f9bb75b..3cae0bf9b1 100644 --- a/skills/firecrawl-search/SKILL.md +++ b/skills/firecrawl-search/SKILL.md @@ -1,6 +1,6 @@ --- name: firecrawl-search -description: Find web sources and discover workflows, data APIs, and indexes. Use for web research or finding structured records, listings, transcripts, and datasets. Supports semantic tool discovery, domain matching, and progressive catalogue browsing. +description: Find web sources with query-relevant page excerpts and optional full-page content, and discover workflows, data APIs, and indexes. Use for web research or finding structured records, listings, transcripts, and datasets. Supports semantic tool discovery, domain matching, and progressive catalogue browsing. allowed-tools: - Bash(firecrawl *) - Bash(npx firecrawl-cli *) @@ -27,7 +27,7 @@ firecrawl search "your query" --sources news --tbs qdr:d -o .firecrawl/news.json Use `firecrawl search --help` for search options, `firecrawl list --help` for contract browsing, and `firecrawl scrape --help` for execution options. -`--categories developer` weighs the developer index beside ordinary web results in this same call (no passage control, no index filters). `--categories research` is a website filter, not the paper index. Dedicated skills: [firecrawl-developer-index](../firecrawl-developer-index/SKILL.md) and [firecrawl-research-index](../firecrawl-research-index/SKILL.md). +`--categories developer` searches an index of public repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. `--categories research` is a website filter, not the paper index. Dedicated skills: [firecrawl-developer-index](../firecrawl-developer-index/SKILL.md) and [firecrawl-research-index](../firecrawl-research-index/SKILL.md). **Done when:** relevant results have been inspected, per-call errors and empty results have been checked, the request has been answered with source links, and feedback is sent within the time window unless opted out. @@ -80,7 +80,7 @@ Keep large search responses in `--json -o` output and select the relevant result ## Tips -- **`--highlights` on by default:** results are query-relevant excerpts, not full-page snippets. Use `--no-highlights` for the original snippets. +- **`--highlights` on by default:** results are query-relevant excerpts from the page. Use `--no-highlights` for the original snippets. - **`--scrape` fetches full content** — reuse that content instead of re-scraping result URLs. This saves credits and avoids redundant fetches. - For large results, use `-o` and bounded local reads when a filesystem is available. Do not dump the full response into context. - Use `jq` to extract URLs or titles: `jq -r '.data.web[].url' .firecrawl/search.json` diff --git a/src/__tests__/cli-argv.test.ts b/src/__tests__/cli-argv.test.ts index b3c72d85ad..139c926bd1 100644 --- a/src/__tests__/cli-argv.test.ts +++ b/src/__tests__/cli-argv.test.ts @@ -82,11 +82,53 @@ describe('CLI argv parsing', () => { expect(result.stdout).not.toContain(removedFilter); } expect(result.stdout).toContain('scoping intent in'); + expect(result.stdout.replace(/\s+/g, ' ')).toContain('public repositories'); // Lean surface: the CLI does not point at the REST API for filters. expect(result.stdout).not.toContain('docs.firecrawl.dev'); expect(result.stderr).not.toContain('unknown command'); }); + testWithBuiltCli( + 'describes default search highlights and public developer coverage', + () => { + const result = spawnSync( + process.execPath, + [cliPath, 'search', '--help'], + { + cwd: process.cwd(), + encoding: 'utf8', + } + ); + + expect(result.status).toBe(0); + const flattened = result.stdout.replace(/\s+/g, ' '); + expect(flattened).toContain( + 'Search the web with query-relevant highlights and discover relevant Alexandria tools' + ); + expect(flattened).toContain( + 'Return query-relevant page excerpts for web and news results when available (default).' + ); + expect(flattened).toContain('public repositories'); + expect(flattened).toContain('research, pdf, developer'); + expect(flattened).not.toContain('github, research'); + } + ); + + testWithBuiltCli('rejects the github search category', () => { + const result = spawnSync( + process.execPath, + [cliPath, 'search', 'tokio', '--categories', 'github'], + { + cwd: process.cwd(), + encoding: 'utf8', + } + ); + + expect(result.status).toBe(1); + expect(result.stderr).toContain('Invalid category "github"'); + expect(result.stderr).toContain('research, pdf, developer'); + }); + testWithBuiltCli('lists the research command in root help output', () => { const result = spawnSync(process.execPath, [cliPath, '--help'], { cwd: process.cwd(), diff --git a/src/__tests__/commands/search.test.ts b/src/__tests__/commands/search.test.ts index d922fbc287..632e9d5384 100644 --- a/src/__tests__/commands/search.test.ts +++ b/src/__tests__/commands/search.test.ts @@ -207,14 +207,14 @@ describe('executeSearch', () => { await executeSearch({ query: 'web scraping python', - categories: ['github'], + categories: ['pdf'], }); expect(mockHttpPost).toHaveBeenCalledWith( '/v2/search', expect.objectContaining({ query: 'web scraping python', - categories: [{ type: 'github' }], + categories: [{ type: 'pdf' }], }) ); }); @@ -422,7 +422,7 @@ describe('executeSearch', () => { query: 'comprehensive test', limit: 20, sources: ['web', 'news'], - categories: ['github'], + categories: ['developer'], tbs: 'qdr:w', location: 'Germany', country: 'DE', @@ -438,7 +438,7 @@ describe('executeSearch', () => { integration: 'cli', toolDetail: 'compact', sources: [{ type: 'web' }, { type: 'news' }], - categories: [{ type: 'github' }], + categories: [{ type: 'developer' }], tbs: 'qdr:w', location: 'Germany', country: 'DE', @@ -762,8 +762,7 @@ describe('executeSearch', () => { }); it('should accept valid category types', async () => { - const categoryList: Array<'github' | 'research' | 'pdf' | 'developer'> = [ - 'github', + const categoryList: Array<'research' | 'pdf' | 'developer'> = [ 'research', 'pdf', 'developer', diff --git a/src/index.ts b/src/index.ts index 89360d0892..d9416f6284 100644 --- a/src/index.ts +++ b/src/index.ts @@ -919,7 +919,9 @@ Max upload size: 50 MB */ function createSearchCommand(): Command { const searchCmd = new Command('search') - .description('Search the web and discover relevant Alexandria tools') + .description( + 'Search the web with query-relevant highlights and discover relevant Alexandria tools' + ) .argument('', 'Search query, or alexandria for semantic tool search') .argument('[tool-query]', 'Query for search alexandria') .option( @@ -933,7 +935,7 @@ function createSearchCommand(): Command { ) .option( '--categories ', - 'Comma-separated categories to filter: github, research, pdf, developer (research filters web results to research-affiliated websites -- it is NOT the paper index; for papers use `firecrawl research search-papers`. developer searches indexed GitHub issues, merged PRs, READMEs, and docs)' + 'Comma-separated categories to filter: research, pdf, developer (research filters web results to research-affiliated websites -- it is NOT the paper index; for papers use `firecrawl research search-papers`. developer searches an index of public repositories, GitHub issues, merged PRs, READMEs, and docs)' ) .option( '--tbs ', @@ -959,7 +961,7 @@ function createSearchCommand(): Command { ) .option( '--highlights', - 'Return query-relevant highlights for each search result' + 'Return query-relevant page excerpts for web and news results when available (default).' ) .option( '--no-highlights', @@ -1036,7 +1038,7 @@ function createSearchCommand(): Command { .map((c: string) => c.trim().toLowerCase()) as SearchCategory[]; // Validate categories - const validCategories = ['github', 'research', 'pdf', 'developer']; + const validCategories = ['research', 'pdf', 'developer']; for (const category of categories) { if (!validCategories.includes(category)) { console.error( @@ -1110,7 +1112,7 @@ function createSearchCommand(): Command { function createDeveloperCommand(): Command { const developerCmd = new Command('developer') .description( - 'Search an index built for coding agents: GitHub issues, merged PRs, repository READMEs, and curated documentation sites. Express repository, source, language, topic, license, and other scoping intent in the query text; semantic retrieval handles the scoping.' + 'Search an index built for coding agents: public repositories, GitHub issues, merged PRs, repository READMEs, and curated documentation sites. Express repository, source, language, topic, license, and other scoping intent in the query text; semantic retrieval handles the scoping.' ) .argument('', 'Natural-language developer question or search phrase') .option( diff --git a/src/types/search.ts b/src/types/search.ts index a2e020cd43..c8a094d65c 100644 --- a/src/types/search.ts +++ b/src/types/search.ts @@ -5,7 +5,7 @@ import type { ScrapeFormat } from './scrape'; export type SearchSource = 'web' | 'images' | 'news' | 'alexandria'; -export type SearchCategory = 'github' | 'research' | 'pdf' | 'developer'; +export type SearchCategory = 'research' | 'pdf' | 'developer'; export interface SearchOptions { domainTools?: boolean; @@ -20,7 +20,7 @@ export interface SearchOptions { limit?: number; /** Sources to search: web, images, news, alexandria (CLI default: web,alexandria) */ sources?: SearchSource[]; - /** Categories to filter results: github, research, pdf, developer */ + /** Categories to filter results: research, pdf, developer */ categories?: SearchCategory[]; /** Time-based search parameter (e.g., qdr:h, qdr:d, qdr:w, qdr:m, qdr:y) */ tbs?: string; @@ -101,8 +101,9 @@ export interface NewsSearchResult { } /** - * One hit from the `developer` category. The index covers GitHub issues, - * merged pull requests, repository READMEs, and curated documentation sites. + * One hit from the `developer` category. The index covers public repositories, + * GitHub issues, merged pull requests, repository READMEs, and curated + * documentation sites. * `description` holds the matched passage, which runs to several KB. */ export interface DeveloperSearchResult {