diff --git a/.pi/settings.json b/.pi/settings.json new file mode 100644 index 0000000..eadb0bc --- /dev/null +++ b/.pi/settings.json @@ -0,0 +1,3 @@ +{ + "skills": ["/app/skills"] +} diff --git a/docker-compose.yml b/docker-compose.yml index 6341257..f391771 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -12,5 +12,7 @@ services: volumes: # Persistent memory + schedules. Bind-mounted so you can inspect/edit on host. - ./memory:/app/memory - # Skills are bind-mounted read-only so you can iterate without rebuild. - - ./skills:/app/skills:ro + # Skills are bind-mounted read-write so the agent can create/update skills. + - ./skills:/app/skills + # Pi settings + project-level skills discovery + - ./.pi:/app/.pi diff --git a/skills/firecrawl/SKILL.md b/skills/firecrawl/SKILL.md new file mode 100644 index 0000000..0854497 --- /dev/null +++ b/skills/firecrawl/SKILL.md @@ -0,0 +1,302 @@ +--- +name: firecrawl +description: Web scraping, crawling, searching, and content extraction via Firecrawl API. Use for scraping any URL to clean markdown/HTML/JSON, searching the web with full page content, crawling entire websites, discovering all URLs on a site (map), batch scraping multiple URLs, or AI-powered autonomous data gathering (agent). Requires a Firecrawl API key. +--- + +# Firecrawl + +Turn websites into LLM-ready data. Scrape, crawl, search, and interact with the web at scale. + +- **Base URL:** `https://api.firecrawl.dev/v2` +- **API Key Format:** `fc-...` (get one at https://firecrawl.dev) +- **Docs:** https://docs.firecrawl.dev + +## Setup + +Set your API key as an environment variable before using any endpoint: + +```bash +export FIRECRAWL_API_KEY="fc-YOUR_API_KEY" +``` + +If the key is not set, prompt the user to get one at https://firecrawl.dev. + +Verify your key works with the helper script: + +```bash +node /app/skills/firecrawl/scripts/firecrawl.js scrape "https://example.com" +``` + +## Quick Start — Node.js Helper + +A Node.js helper script wraps all endpoints with auto-polling for async jobs: + +```bash +# Scrape a URL to markdown +node /app/skills/firecrawl/scripts/firecrawl.js scrape "https://example.com" + +# Search the web +node /app/skills/firecrawl/scripts/firecrawl.js search "latest AI news" + +# Crawl a website (polls until done) +node /app/skills/firecrawl/scripts/firecrawl.js crawl "https://docs.example.com" + +# Map URLs on a site +node /app/skills/firecrawl/scripts/firecrawl.js map "https://example.com" + +# AI-powered data gathering +node /app/skills/firecrawl/scripts/firecrawl.js agent "Find the founders of Stripe" + +# Batch scrape multiple URLs +node /app/skills/firecrawl/scripts/firecrawl.js batch "https://a.com" "https://b.com" +``` + +Pass options as a JSON string for the second argument: + +```bash +node /app/skills/firecrawl/scripts/firecrawl.js scrape "https://example.com" '{"waitFor":2000}' +node /app/skills/firecrawl/scripts/firecrawl.js search "AI tools" '{"limit":10}' +node /app/skills/firecrawl/scripts/firecrawl.js crawl "https://docs.example.com" '{"limit":10}' +``` + +## Full API Reference (cURL) + +### Scrape — extract content from a single URL + +Get LLM-ready markdown, HTML, screenshots, or structured JSON from any page. + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/scrape' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "url": "", + "formats": ["markdown"], + "onlyMainContent": true, + "waitFor": 0 + }' +``` + +**Key parameters:** +| Param | Type | Description | +|-------|------|-------------| +| `url` | string | **Required.** The URL to scrape. | +| `formats` | array | Output formats: `markdown`, `html`, `rawHtml`, `screenshot`, `screenshot@fullPage`, `json`, `links`, `images`. Use `["markdown"]` for LLM consumption. | +| `onlyMainContent` | bool | Strip nav, footer, sidebars. Default: `true`. | +| `waitFor` | int | Time in ms to wait for JS rendering. Use `1000`-`5000` for JS-heavy pages. | +| `actions` | array | Browser actions before scraping: click, scroll, type, wait. See [Interact](#interact--actions). | +| `extract` | object | Extract structured data with `{schema: {...}, prompt: "..."}`. | +| `includeTags` | array | HTML tags to include. | +| `excludeTags` | array | HTML tags to exclude. | +| `headers` | object | Custom HTTP headers for the request. | + +**Realistic example** — scrape a JS-heavy docs page: + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/scrape' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "url": "https://docs.firecrawl.dev/api-reference/v2/scrape", + "formats": ["markdown"], + "onlyMainContent": true, + "waitFor": 2000 + }' | jq -r '.data.markdown' | head -100 +``` + +Always use `jq` to extract the relevant fields. The response shape: + +```json +{ + "success": true, + "data": { + "markdown": "# Page content...", + "html": "...", + "metadata": { + "title": "Page Title", + "sourceURL": "https://...", + "statusCode": 200 + } + } +} +``` + +### Search — search the web and get full content from results + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/search' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "query": "", + "limit": 5, + "scrapeOptions": { + "formats": ["markdown"], + "onlyMainContent": true + } + }' +``` + +**Key parameters:** +| Param | Type | Description | +|-------|------|-------------| +| `query` | string | **Required.** The search query. | +| `limit` | int | Max results. Default: 5. | +| `scrapeOptions` | object | Scrape options applied to each result page (same as scrape params). Omit to get metadata only (no page content). | +| `sources` | array | Limit search to `["web"]` or `["news"]`. Default: web. | + +### Crawl — scrape every page on a website + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/crawl' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "url": "", + "limit": 50, + "maxDepth": 3, + "scrapeOptions": { + "formats": ["markdown"], + "onlyMainContent": true + } + }' +``` + +Returns a job with an `id`. Poll for status: + +```bash +curl -s -X GET "https://api.firecrawl.dev/v2/crawl/" \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" +``` + +When `status` is `"completed"`, `data` contains the scraped pages array. Crawl jobs are async — poll every 5-10 seconds until done. + +**Key parameters:** +| Param | Type | Description | +|-------|------|-------------| +| `url` | string | **Required.** Starting URL. | +| `limit` | int | Max pages to crawl. Default: 500. | +| `maxDepth` | int | Max link depth from the starting URL. | +| `includePaths` | array | Glob patterns to match URLs (e.g., `["/docs/*"]`). | +| `excludePaths` | array | Glob patterns to exclude URLs. | +| `scrapeOptions` | object | Scrape options applied to each page. | + +### Map — discover all URLs on a site + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/map' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{"url": ""}' +``` + +**Optional parameters:** +| Param | Type | Description | +|-------|------|-------------| +| `url` | string | **Required.** The URL to map. | +| `search` | string | Filter links by relevance to a search term. | +| `ignoreSitemap` | bool | Skip the sitemap and discover links by crawling. | +| `includeSubdomains` | bool | Include subdomains. | + +### Batch Scrape — scrape many URLs at once + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/batch/scrape' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "urls": ["", "", ""], + "formats": ["markdown"], + "onlyMainContent": true + }' +``` + +Returns a job with an `id`. Poll `GET /v2/batch/scrape/` until `status` is `"completed"`. + +### Agent — AI-powered autonomous data gathering + +Describe what you need — no URLs required. Firecrawl's AI agent searches, navigates, and retrieves data for you. + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/agent' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "prompt": "", + "urls": [""], + "model": "spark-1-mini", + "schema": {} + }' +``` + +Returns a job with an `id`. Poll `GET /v2/agent/` until `status` is `"completed"`. + +**Key parameters:** +| Param | Type | Description | +|-------|------|-------------| +| `prompt` | string | **Required.** Natural language description of what data to gather. | +| `urls` | array | Optional URLs to constrain the search. | +| `model` | string | `"spark-1-mini"` (default, cheaper) or `"spark-1-pro"` (complex tasks). | +| `schema` | object | JSON schema for structured output. | + +Use `spark-1-pro` when: comparing data across multiple sites, navigating complex/auth'd sites, or when accuracy is critical. + +### Interact / Actions — interact with a page before scraping + +Use `actions` in a scrape request to click, scroll, type, or wait before extracting content: + +```bash +curl -s -X POST 'https://api.firecrawl.dev/v2/scrape' \ + -H "Authorization: Bearer $FIRECRAWL_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{ + "url": "https://example.com", + "formats": ["markdown"], + "actions": [ + {"type": "wait", "milliseconds": 2000}, + {"type": "click", "selector": "button.accept-cookies"}, + {"type": "write", "selector": "input[name=\"search\"]", "text": "my query"}, + {"type": "press", "key": "Enter"}, + {"type": "scroll", "direction": "down", "amount": 500} + ] + }' +``` + +Action types: `wait`, `click`, `write`, `press`, `scroll`, `screenshot`, `executeJavascript`. + +## jq Quick Reference + +Most responses are JSON. Use `jq` to extract what you need: + +```bash +# Extract markdown from scrape response +| jq -r '.data.markdown' + +# Get search result titles and URLs +| jq '.data[] | {title: .title, url: .url}' + +# Check crawl/agent status +| jq '{status: .status, completed: .completed, total: .total}' + +# Pretty-print anything +| jq . +``` + +## When to Use Each Endpoint + +| Task | Endpoint | +|------|----------| +| Get content from a known URL | **Scrape** | +| Find information across the web | **Search** | +| Scrape an entire website | **Crawl** | +| List all URLs on a site | **Map** | +| Scrape a known list of URLs | **Batch Scrape** | +| "Find me X about Y" (no specific URL) | **Agent** | +| Page requires login/click/scroll first | **Scrape** with `actions` | + +## Cost and Rate Limits + +- Each scrape, search result, or crawled page consumes credits based on your plan. +- Crawl and batch jobs return async — always poll and wait for completion. +- Check your usage: `GET /v2/team` returns credit and usage info. +- Always use `onlyMainContent: true` to reduce token waste. diff --git a/skills/firecrawl/scripts/firecrawl.js b/skills/firecrawl/scripts/firecrawl.js new file mode 100755 index 0000000..d6ddddb --- /dev/null +++ b/skills/firecrawl/scripts/firecrawl.js @@ -0,0 +1,205 @@ +#!/usr/bin/env node +/** + * Firecrawl CLI helper — wraps all v2 endpoints via Node.js. + * Usage: node scripts/firecrawl.js [args...] + * + * Commands: + * scrape Scrape a URL to markdown + * search Search the web + * crawl Start a crawl, poll until done + * map Map all URLs on a site + * agent AI agent for autonomous data gathering + * batch ... Batch scrape multiple URLs + */ + +const API_KEY = process.env.FIRECRAWL_API_KEY; +const BASE = 'https://api.firecrawl.dev/v2'; + +if (!API_KEY) { + console.error('Error: FIRECRAWL_API_KEY is not set.'); + console.error('Get one at https://firecrawl.dev and export it:'); + console.error(' export FIRECRAWL_API_KEY="fc-YOUR_API_KEY"'); + process.exit(1); +} + +async function request(method, path, body) { + const url = BASE + path; + const res = await fetch(url, { + method, + headers: { + 'Authorization': `Bearer ${API_KEY}`, + 'Content-Type': 'application/json', + 'Accept': 'application/json' + }, + body: body ? JSON.stringify(body) : undefined + }); + const text = await res.text(); + try { return JSON.parse(text); } + catch { return text; } +} + +function sleep(ms) { return new Promise(r => setTimeout(r, ms)); } + +async function scrape(url, opts = {}) { + const body = { + url, + formats: opts.formats || ['markdown'], + onlyMainContent: opts.onlyMainContent !== false, + waitFor: opts.waitFor || 0, + ...opts + }; + const result = await request('POST', '/scrape', body); + if (result.success && result.data?.markdown) { + console.log(result.data.markdown); + } else { + console.log(JSON.stringify(result, null, 2)); + } +} + +async function search(query, opts = {}) { + const body = { + query, + limit: opts.limit || 5, + ...opts + }; + const result = await request('POST', '/search', body); + console.log(JSON.stringify(result, null, 2)); +} + +async function crawl(url, opts = {}) { + const scrapeOpts = { formats: ['markdown'], onlyMainContent: true, ...(opts.scrapeOptions || {}) }; + const { scrapeOptions, ...restOpts } = opts; + const body = { url, limit: 50, scrapeOptions: scrapeOpts, ...restOpts }; + const job = await request('POST', '/crawl', body); + if (!job.id) { + console.log(JSON.stringify(job, null, 2)); + return; + } + console.error(`Crawl started: ${job.id} — polling...`); + for (let i = 0; i < 60; i++) { + await sleep(5000); + const status = await request('GET', `/crawl/${job.id}`); + if (status.status === 'completed') { + for (const page of status.data) { + console.log(`\n## ${page.metadata?.sourceURL || page.metadata?.url}\n`); + console.log(page.markdown || ''); + } + return; + } + if (status.status === 'failed') { + console.error('Crawl failed:', JSON.stringify(status, null, 2)); + process.exit(1); + } + console.error(` ${status.completed || 0}/${status.total || '?'} pages...`); + } + console.error('Crawl timed out.'); +} + +async function map(url, opts = {}) { + const result = await request('POST', '/map', { url, ...opts }); + console.log(JSON.stringify(result, null, 2)); +} + +async function agent(prompt, opts = {}) { + const body = { prompt, ...opts }; + const job = await request('POST', '/agent', body); + if (!job.id) { + console.log(JSON.stringify(job, null, 2)); + return; + } + console.error(`Agent started: ${job.id} — polling...`); + for (let i = 0; i < 60; i++) { + await sleep(5000); + const status = await request('GET', `/agent/${job.id}`); + if (status.status === 'completed') { + console.log(JSON.stringify(status, null, 2)); + return; + } + if (status.status === 'failed') { + console.error('Agent failed:', JSON.stringify(status, null, 2)); + process.exit(1); + } + console.error(' still working...'); + } + console.error('Agent timed out.'); +} + +async function batch(urls, opts = {}) { + const body = { + urls, + formats: opts.formats || ['markdown'], + onlyMainContent: opts.onlyMainContent !== false, + ...opts + }; + const job = await request('POST', '/batch/scrape', body); + if (!job.id) { + console.log(JSON.stringify(job, null, 2)); + return; + } + console.error(`Batch started: ${job.id} — polling...`); + for (let i = 0; i < 60; i++) { + await sleep(5000); + const status = await request('GET', `/batch/scrape/${job.id}`); + if (status.status === 'completed') { + for (const page of status.data) { + console.log(`\n## ${page.metadata?.sourceURL || page.metadata?.url}\n`); + console.log(page.markdown || ''); + } + return; + } + if (status.status === 'failed') { + console.error('Batch failed:', JSON.stringify(status, null, 2)); + process.exit(1); + } + console.error(` ${status.completed || 0}/${status.total || '?'} pages...`); + } + console.error('Batch timed out.'); +} + +const [,, cmd, ...args] = process.argv; + +async function main() { + switch (cmd) { + case 'scrape': + await scrape(args[0], args[1] ? JSON.parse(args[1]) : {}); + break; + case 'search': + await search(args[0], args[1] ? JSON.parse(args[1]) : {}); + break; + case 'crawl': + await crawl(args[0], args[1] ? JSON.parse(args[1]) : {}); + break; + case 'map': + await map(args[0], args[1] ? JSON.parse(args[1]) : {}); + break; + case 'agent': + await agent(args[0], args[1] ? JSON.parse(args[1]) : {}); + break; + case 'batch': + await batch(args, args.length > 1 ? JSON.parse(args[args.length - 1]) : {}); + break; + default: + console.error(`Usage: firecrawl.js [...args] + +Commands: + scrape [options_json] Scrape a URL to markdown + search [options_json] Search the web + crawl [options_json] Crawl a website + map [options_json] Map all URLs on a site + agent [options_json] AI-powered data gathering + batch [url2...] Batch scrape multiple URLs + +Examples: + node scripts/firecrawl.js scrape "https://example.com" "{\"waitFor\":2000}" + node scripts/firecrawl.js search "latest AI news" + node scripts/firecrawl.js crawl "https://docs.example.com" "{\"limit\":10}" + node scripts/firecrawl.js agent "Find the founders of Stripe" + node scripts/firecrawl.js batch "https://a.com" "https://b.com"`); + process.exit(1); + } +} + +main().catch(e => { + console.error('Error:', e.message); + process.exit(1); +});