diff --git a/README.md b/README.md index 7536ef7c69..dcfae383e7 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # 🔥 Firecrawl CLI -Command-line interface for Firecrawl. Search, scrape, interact, crawl, map, search research papers and developer sources, and run agent jobs directly from your terminal. +Command-line interface for Firecrawl. Search, scrape, interact, crawl, map, search research papers, developer sources, and government sources, and run agent jobs directly from your terminal. ## Installation @@ -299,7 +299,7 @@ firecrawl search "landscape photography" --sources images # Multiple sources firecrawl search "machine learning" --sources web,news,images -# Filter by category (research-affiliated websites, PDFs, developer index) +# Filter by category (research-affiliated websites, PDFs, developer index, gov index) firecrawl search "transformer architecture" --categories research firecrawl search "machine learning" --categories pdf,research @@ -309,6 +309,9 @@ firecrawl search "machine learning" --categories pdf,research # Developer search: public repositories, GitHub issues, merged PRs, READMEs, and docs firecrawl search "axum middleware ordering" --categories developer +# Government search: US government sources (cannot be combined with other categories) +firecrawl search "California data breach notification statute" --categories gov + # Time-based search firecrawl search "AI announcements" --tbs qdr:d # Past day firecrawl search "tech news" --tbs qdr:w # Past week @@ -327,23 +330,23 @@ firecrawl search "AI data tools" #### Search Options -| Option | Description | -| ---------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `--limit ` | Maximum results (default: 5, max: 100) | -| `--sources ` | Comma-separated: `web`, `images`, `news` (default: web) | -| `--categories ` | Comma-separated: `research` (research-affiliated websites -- for papers use [`research search-papers`](#research---search-research-papers)), `pdf`, `developer` | -| `--tbs ` | Time filter: `qdr:h` (hour), `qdr:d` (day), `qdr:w` (week), `qdr:m` (month), `qdr:y` (year) | -| `--location ` | Geo-targeting (e.g., "Germany", "San Francisco,California,United States") | -| `--country ` | ISO country code (default: US) | -| `--timeout ` | Timeout in milliseconds (default: 60000) | -| `--highlights` | Query-relevant highlights for web and news when available (default) | -| `--no-highlights` | Keep the original search snippets | -| `--ignore-invalid-urls` | Exclude URLs invalid for other Firecrawl endpoints | -| `--scrape` | Enable scraping of search results | -| `--scrape-formats ` | Scrape formats when `--scrape` enabled (default: markdown) | -| `--only-main-content` | Include only main content when scraping (default: true) | -| `-o, --output ` | Save to file | -| `--json` | Output as compact JSON | +| Option | Description | +| ---------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `--limit ` | Maximum results (default: 5, max: 100) | +| `--sources ` | Comma-separated: `web`, `images`, `news` (default: web) | +| `--categories ` | Comma-separated: `research` (research-affiliated websites -- for papers use [`research search-papers`](#research---search-research-papers)), `pdf`, `developer`, `gov` (cannot be combined with other categories) | +| `--tbs ` | Time filter: `qdr:h` (hour), `qdr:d` (day), `qdr:w` (week), `qdr:m` (month), `qdr:y` (year) | +| `--location ` | Geo-targeting (e.g., "Germany", "San Francisco,California,United States") | +| `--country ` | ISO country code (default: US) | +| `--timeout ` | Timeout in milliseconds (default: 60000) | +| `--highlights` | Query-relevant highlights for web and news when available (default) | +| `--no-highlights` | Keep the original search snippets | +| `--ignore-invalid-urls` | Exclude URLs invalid for other Firecrawl endpoints | +| `--scrape` | Enable scraping of search results | +| `--scrape-formats ` | Scrape formats when `--scrape` enabled (default: markdown) | +| `--only-main-content` | Include only main content when scraping (default: true) | +| `-o, --output ` | Save to file | +| `--json` | Output as compact JSON | #### Examples @@ -409,6 +412,37 @@ firecrawl developer "tokio select cancellation safety" --json -o results.json --- +### `gov` - Search the Firecrawl Government Index + +Search the Government Index: primary law and regulatory material from US federal, state, and local government sources, including statutes, regulations, codes, court opinions, and other government publications. + +For the request and response schema, see the [Government Index REST API](https://docs.firecrawl.dev/features/gov). + +```bash +firecrawl gov "food labeling requirements for allergens" +``` + +#### Options + +| Option | Description | +| --------------------- | ----------------------------------------- | +| `--limit ` | Number of results (default: 10, max: 100) | +| `-o, --output ` | Save to file | +| `--json` | Output the raw response as JSON | +| `--pretty` | Pretty print JSON output | + +#### Examples + +```bash +# Find state statutes on a topic +firecrawl gov "California data breach notification statute" --limit 10 + +# Keep the raw response +firecrawl gov "FDA food labeling regulations" --json -o results.json +``` + +--- + ### `research` - Search research papers Search Firecrawl's research paper index: roughly 43M abstracts, around 90% biomedical (PubMed, bioRxiv, medRxiv) plus arXiv. Use this for biomedical, clinical, and scientific literature rather than scraping PubMed, bioRxiv, or Google Scholar by hand. diff --git a/skills/firecrawl-search/SKILL.md b/skills/firecrawl-search/SKILL.md index 578badaa97..f0b85dfb5e 100644 --- a/skills/firecrawl-search/SKILL.md +++ b/skills/firecrawl-search/SKILL.md @@ -27,7 +27,7 @@ firecrawl search "your query" --sources news --tbs qdr:d -o .firecrawl/news.json Use `firecrawl search --help` for search options, `firecrawl list --help` for contract browsing, and `firecrawl scrape --help` for execution options. -`--categories developer` searches an index of public repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. `--categories research` is a website filter, not the paper index. Dedicated skills: [firecrawl-developer-index](../firecrawl-developer-index/SKILL.md) and [firecrawl-research-index](../firecrawl-research-index/SKILL.md). +`--categories developer` searches an index of public repositories, GitHub issues, merged pull requests, repository READMEs, and curated documentation sites. `--categories gov` searches US federal, state, and local government legal and regulatory sources and cannot be combined with other categories. `--categories research` is a website filter, not the paper index. Dedicated skills: [firecrawl-developer-index](../firecrawl-developer-index/SKILL.md) and [firecrawl-research-index](../firecrawl-research-index/SKILL.md). **Done when:** relevant results have been inspected, per-call errors and empty results have been checked, the request has been answered with source links, and feedback is sent within the time window unless opted out. diff --git a/src/__tests__/cli-argv.test.ts b/src/__tests__/cli-argv.test.ts index 139c926bd1..ae82cd5e17 100644 --- a/src/__tests__/cli-argv.test.ts +++ b/src/__tests__/cli-argv.test.ts @@ -88,6 +88,16 @@ describe('CLI argv parsing', () => { expect(result.stderr).not.toContain('unknown command'); }); + testWithBuiltCli('parses the gov command and shows its help', () => { + const result = spawnSync(process.execPath, [cliPath, 'gov', '--help'], { + cwd: process.cwd(), + encoding: 'utf8', + }); + + expect(result.status).toBe(0); + expect(result.stdout).toContain('Usage: firecrawl gov'); + }); + testWithBuiltCli( 'describes default search highlights and public developer coverage', () => { diff --git a/src/__tests__/commands/gov.test.ts b/src/__tests__/commands/gov.test.ts new file mode 100644 index 0000000000..a371483911 --- /dev/null +++ b/src/__tests__/commands/gov.test.ts @@ -0,0 +1,186 @@ +/** + * Tests for gov command + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { handleGovSearchCommand } from '../../commands/gov'; +import { getClient, isKeylessMode } from '../../utils/client'; +import { initializeConfig } from '../../utils/config'; +import { writeOutput } from '../../utils/output'; +import { setupTest, teardownTest } from '../utils/mock-client'; + +vi.mock('../../utils/output', () => ({ writeOutput: vi.fn() })); + +vi.mock('../../utils/client', async () => { + const actual = await vi.importActual('../../utils/client'); + return { + ...actual, + getClient: vi.fn(), + isKeylessMode: vi.fn(() => false), + }; +}); + +describe('handleGovSearchCommand', () => { + let mockHttpGet: ReturnType; + + // Wrap a payload in the axios envelope returned by `client.http.get`. + const mockGovResponse = (web: any[]) => ({ + data: { success: true, data: { web } }, + }); + + const sampleResult = { + url: 'https://www.ecfr.gov/current/title-21/chapter-I/subchapter-B/part-101', + title: '21 CFR Part 101 -- Food Labeling', + description: 'Food labeling requirements for packaged foods.', + position: 1, + }; + + beforeEach(() => { + setupTest(); + initializeConfig({ + apiKey: 'test-api-key', + apiUrl: 'https://api.firecrawl.dev', + }); + + mockHttpGet = vi.fn(); + vi.mocked(getClient).mockReturnValue({ + http: { get: mockHttpGet }, + } as any); + }); + + afterEach(() => { + teardownTest(); + vi.clearAllMocks(); + vi.unstubAllGlobals(); + }); + + describe('API call generation', () => { + it.each([ + [{}, '/v2/search/gov?query=food+labeling&integration=cli'], + [{ k: 5 }, '/v2/search/gov?query=food+labeling&k=5&integration=cli'], + ])('calls /v2/search/gov with %o', async (extra, expectedUrl) => { + mockHttpGet.mockResolvedValue(mockGovResponse([sampleResult])); + + await handleGovSearchCommand({ + query: 'food labeling', + ...extra, + }); + + expect(mockHttpGet).toHaveBeenCalledTimes(1); + expect(mockHttpGet).toHaveBeenCalledWith(expectedUrl); + }); + }); + + describe('output', () => { + it('renders numbered title, url, and description blocks', async () => { + mockHttpGet.mockResolvedValue( + mockGovResponse([ + sampleResult, + { + url: 'https://www.ecfr.gov/current/title-21/part-102', + title: '21 CFR Part 102', + position: 2, + }, + ]) + ); + + await handleGovSearchCommand({ query: 'food labeling' }); + + const [content] = vi.mocked(writeOutput).mock.calls[0]; + expect(content).toBe( + [ + '## 1. 21 CFR Part 101 -- Food Labeling', + sampleResult.url, + 'Food labeling requirements for packaged foods.', + '', + '## 2. 21 CFR Part 102', + 'https://www.ecfr.gov/current/title-21/part-102', + ].join('\n') + ); + }); + + it('prints a placeholder when the response has no data', async () => { + mockHttpGet.mockResolvedValue({ data: { success: true } }); + + await handleGovSearchCommand({ query: 'no hits' }); + + const [content] = vi.mocked(writeOutput).mock.calls[0]; + expect(content).toBe('(no results)'); + }); + + it('outputs the raw response as JSON with --json', async () => { + mockHttpGet.mockResolvedValue(mockGovResponse([sampleResult])); + + await handleGovSearchCommand({ + query: 'food labeling', + json: true, + }); + + const [content] = vi.mocked(writeOutput).mock.calls[0] as [string]; + expect(JSON.parse(content)).toEqual({ + success: true, + data: { web: [sampleResult] }, + }); + }); + }); + + describe('keyless mode', () => { + it('calls the endpoint directly and renders the results', async () => { + vi.mocked(isKeylessMode).mockReturnValueOnce(true); + const fetchMock = vi.fn( + async (_url: string, _init?: RequestInit) => + new Response( + JSON.stringify({ success: true, data: { web: [sampleResult] } }), + { status: 200 } + ) + ); + vi.stubGlobal('fetch', fetchMock); + + await handleGovSearchCommand({ query: 'food labeling' }); + + expect(mockHttpGet).not.toHaveBeenCalled(); + expect(fetchMock).toHaveBeenCalledWith( + 'https://api.firecrawl.dev/v2/search/gov?query=food+labeling&integration=cli', + expect.objectContaining({ method: 'GET' }) + ); + expect(vi.mocked(writeOutput).mock.calls[0][0]).toContain( + sampleResult.title + ); + }); + }); + + describe('error handling', () => { + it.each([ + [ + 'the response reports a failure', + () => + mockHttpGet.mockResolvedValue({ + data: { success: false, error: 'Search failed' }, + }), + 'Search failed', + ], + [ + 'the request fails', + () => mockHttpGet.mockRejectedValue(new Error('boom')), + 'boom', + ], + ])('exits with code 1 when %s', async (_label, arrange, message) => { + arrange(); + const exitSpy = vi + .spyOn(process, 'exit') + .mockImplementation((() => undefined) as any); + const errorSpy = vi + .spyOn(console, 'error') + .mockImplementation(() => undefined); + + await handleGovSearchCommand({ query: 'test' }); + + expect(errorSpy).toHaveBeenCalledWith('Error:', message); + expect(exitSpy).toHaveBeenCalledWith(1); + expect(writeOutput).not.toHaveBeenCalled(); + + exitSpy.mockRestore(); + errorSpy.mockRestore(); + }); + }); +}); diff --git a/src/commands/gov.ts b/src/commands/gov.ts new file mode 100644 index 0000000000..f92f679c2e --- /dev/null +++ b/src/commands/gov.ts @@ -0,0 +1,78 @@ +import { getClient, isKeylessMode, keylessGet } from '../utils/client'; +import { writeOutput } from '../utils/output'; +import type { + GovResult, + GovSearchOptions, + GovSearchResponse, +} from '../types/gov'; + +const BASE = '/v2/search/gov'; + +async function getGov(path: string, options: GovSearchOptions): Promise { + const url = `${path}${path.includes('?') ? '&' : '?'}integration=cli`; + + if (isKeylessMode(options.apiKey, options.apiUrl)) { + return (await keylessGet(url)) as T; + } + + const app = getClient({ apiKey: options.apiKey, apiUrl: options.apiUrl }); + const response = await (app as any).http.get(url); + return (response?.data ?? {}) as T; +} + +function fmtResult(item: GovResult, index: number): string { + const lines = [ + `## ${item.position ?? index + 1}. ${item.title ?? '(untitled)'}`, + item.url, + ]; + if (item.description) lines.push(item.description); + return lines.join('\n'); +} + +function fmtGov(data: GovSearchResponse): string { + const results = data.data?.web ?? []; + if (results.length === 0) return '(no results)'; + return results.map(fmtResult).join('\n\n'); +} + +function writeGovOutput( + data: GovSearchResponse, + readable: string, + options: GovSearchOptions +): void { + const content = + options.json || options.pretty + ? options.pretty + ? JSON.stringify(data, null, 2) + : JSON.stringify(data) + : readable; + writeOutput(content, options.output, !!options.output); +} + +function handleError(error: unknown): never { + console.error( + 'Error:', + error instanceof Error ? error.message : 'Unknown error occurred' + ); + process.exit(1); +} + +export async function handleGovSearchCommand( + options: GovSearchOptions +): Promise { + try { + const params = new URLSearchParams(); + params.append('query', options.query); + if (options.k != null) params.append('k', String(options.k)); + const data = await getGov( + `${BASE}?${params.toString()}`, + options + ); + if (data.success === false) { + throw new Error(data.error ?? 'Government search failed'); + } + writeGovOutput(data, fmtGov(data), options); + } catch (error) { + handleError(error); + } +} diff --git a/src/commands/list.ts b/src/commands/list.ts index c6ef65d59b..c5c0bb239c 100644 --- a/src/commands/list.ts +++ b/src/commands/list.ts @@ -175,8 +175,9 @@ function renderCategories(items: Category[]): string { ...items.map((item) => ` ${item.name} (${item.id}): ${item.description}`), ...(!items.length ? [' No categories are currently visible.'] : []), '', - 'Developer and Research indexes have native commands:', + 'Developer, Government, and Research indexes have native commands:', ' firecrawl developer --help', + ' firecrawl gov --help', ' firecrawl research --help', '', 'All providers: firecrawl alexandria list --providers', diff --git a/src/index.ts b/src/index.ts index aef4eb4bd0..343bdcd552 100644 --- a/src/index.ts +++ b/src/index.ts @@ -28,6 +28,7 @@ import { handleParseCommand } from './commands/parse'; import { createMonitorCommand } from './commands/monitor'; import { handleSearchCommand } from './commands/search'; import { handleDeveloperSearchCommand } from './commands/developer'; +import { handleGovSearchCommand } from './commands/gov'; import { handleInspectPaperCommand, handleReadPaperCommand, @@ -941,7 +942,7 @@ function createSearchCommand(): Command { ) .option( '--categories ', - 'Comma-separated categories to filter: research, pdf, developer (research filters web results to research-affiliated websites -- it is NOT the paper index; for papers use `firecrawl research search-papers`. developer searches an index of public repositories, GitHub issues, merged PRs, READMEs, and docs)' + 'Comma-separated categories to filter: research, pdf, developer, gov (research filters web results to research-affiliated websites -- it is NOT the paper index; for papers use `firecrawl research search-papers`. developer searches an index of public repositories, GitHub issues, merged PRs, READMEs, and docs. gov searches US federal, state, and local government legal and regulatory sources and cannot be combined with other categories)' ) .option( '--tbs ', @@ -1044,7 +1045,7 @@ function createSearchCommand(): Command { .map((c: string) => c.trim().toLowerCase()) as SearchCategory[]; // Validate categories - const validCategories = ['research', 'pdf', 'developer']; + const validCategories = ['research', 'pdf', 'developer', 'gov']; for (const category of categories) { if (!validCategories.includes(category)) { console.error( @@ -1163,6 +1164,52 @@ Examples: return developerCmd; } +/** + * Create and configure the gov command + */ +function createGovCommand(): Command { + const govCmd = new Command('gov') + .description( + 'Search the Firecrawl Government Index: primary law and regulatory material from US federal, state, and local government sources, including statutes, regulations, codes, court opinions, and other government publications.' + ) + .argument('', 'Natural-language legal question or search phrase') + .option( + '--limit ', + 'Number of results to return (default: 10, max: 100)', + parseInt + ) + .addOption(new Option('--k ').argParser(parseInt).hideHelp()) + .option( + '-k, --api-key ', + 'Firecrawl API key (overrides global --api-key)' + ) + .option('--api-url ', 'API URL (overrides global --api-url)') + .option('-o, --output ', 'Output file path (default: stdout)') + .option('--json', 'Output as compact JSON', false) + .option('--pretty', 'Pretty print JSON output', false) + .addHelpText( + 'after', + ` +Examples: + $ firecrawl gov "food labeling requirements for allergens" --limit 10 + $ firecrawl gov "California data breach notification statute" --json +` + ) + .action(async (query, options) => { + await handleGovSearchCommand({ + query, + k: researchLimit(options), + apiKey: options.apiKey, + apiUrl: options.apiUrl, + output: options.output, + json: options.json, + pretty: options.pretty, + }); + }); + + return govCmd; +} + /** * Create and configure the research command group */ @@ -2235,6 +2282,7 @@ program.addCommand(createFindToolsCommand()); program.addCommand(createListCommand()); program.addCommand(createAlexandriaCommand()); program.addCommand(createDeveloperCommand()); +program.addCommand(createGovCommand()); program.addCommand(createResearchCommand()); program.addCommand(createFeedbackCommand()); program.addCommand(createSearchFeedbackCommand()); diff --git a/src/types/gov.ts b/src/types/gov.ts new file mode 100644 index 0000000000..7ccd45f3a2 --- /dev/null +++ b/src/types/gov.ts @@ -0,0 +1,22 @@ +export interface GovSearchOptions { + query: string; + k?: number; + apiKey?: string; + apiUrl?: string; + output?: string; + json?: boolean; + pretty?: boolean; +} + +export interface GovResult { + url: string; + title?: string; + description?: string; + position?: number; +} + +export interface GovSearchResponse { + success: boolean; + error?: string; + data?: { web?: GovResult[] }; +} diff --git a/src/types/search.ts b/src/types/search.ts index ebf52aa5d8..fe59460979 100644 --- a/src/types/search.ts +++ b/src/types/search.ts @@ -5,7 +5,7 @@ import type { ScrapeFormat } from './scrape'; export type SearchSource = 'web' | 'images' | 'news' | 'alexandria'; -export type SearchCategory = 'research' | 'pdf' | 'developer'; +export type SearchCategory = 'research' | 'pdf' | 'developer' | 'gov'; export interface SearchOptions { domainTools?: boolean; @@ -24,7 +24,7 @@ export interface SearchOptions { limit?: number; /** Sources to search: web, images, news, alexandria (CLI default: web,alexandria) */ sources?: SearchSource[]; - /** Categories to filter results: research, pdf, developer */ + /** Categories to filter results: research, pdf, developer, gov */ categories?: SearchCategory[]; /** Time-based search parameter (e.g., qdr:h, qdr:d, qdr:w, qdr:m, qdr:y) */ tbs?: string;