{"_id":"@ariangibson/firecrawl-lite-mcp-server","_rev":"4-1984169b699a7eba7816fe1a4356f57a","name":"@ariangibson/firecrawl-lite-mcp-server","dist-tags":{"latest":"1.4.0"},"versions":{"1.1.1":{"name":"@ariangibson/firecrawl-lite-mcp-server","version":"1.1.1","license":"MIT","_id":"@ariangibson/firecrawl-lite-mcp-server@1.1.1","maintainers":[{"name":"ariangibson","email":"ariangibson@gmail.com"}],"homepage":"https://github.com/ariangibson/firecrawl-lite-mcp-server#readme","bugs":{"url":"https://github.com/ariangibson/firecrawl-lite-mcp-server/issues"},"bin":{"firecrawl-lite-mcp-server":"dist/index.js"},"dist":{"shasum":"bf3afe384cfee30ef1f358c9c0d66b8ae94a0afa","tarball":"https://registry.npmjs.org/@ariangibson/firecrawl-lite-mcp-server/-/firecrawl-lite-mcp-server-1.1.1.tgz","fileCount":4,"integrity":"sha512-1AnEJDInVN7dpy1ZQyYzIVXGFBWxzg4xUwNqlnjSskd6lIRv8T8F4FhK24LGbD90PCPFagio2ZTdDXWjA8r9bw==","signatures":[{"sig":"MEUCIQCx6YU5AZhfMpTsoQWazewdY5g3YJ/Oz2ose6SBm6ZEbgIgS2adlMDO4TnApbFSu8dQMkNuhpZfSV4saedFmMxP+D0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":46132},"type":"module","engines":{"node":">=18.0.0"},"gitHead":"b27ef3ebbd4477ddd31141c64817bb65cffa4023","scripts":{"dev":"tsc && node dist/index.js","lint":"tsc --noEmit","build":"tsc","start":"node dist/index.js"},"_npmUser":{"name":"ariangibson","email":"ariangibson@gmail.com"},"repository":{"url":"git+https://github.com/ariangibson/firecrawl-lite-mcp-server.git","type":"git"},"_npmVersion":"11.5.2","description":"Privacy-first, standalone MCP server for web scraping and data extraction using local browser automation and your own LLM API key","directories":{},"_nodeVersion":"24.1.0","dependencies":{"ws":"^8.18.1","axios":"^1.11.0","dotenv":"^16.4.7","cheerio":"^1.1.2","express":"^5.1.0","puppeteer":"^24.17.1","@types/cheerio":"^0.22.35","@modelcontextprotocol/sdk":"^1.17.3"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.2","@types/node":"^20.10.5","@types/express":"^5.0.1"},"_npmOperationalInternal":{"tmp":"tmp/firecrawl-lite-mcp-server_1.1.1_1757192574000_0.7706468586504438","host":"s3://npm-registry-packages-npm-production"},"deprecated":"Renamed: use firecrawl-lite-mcp-server instead (npx -y firecrawl-lite-mcp-server)"},"1.1.2":{"name":"@ariangibson/firecrawl-lite-mcp-server","version":"1.1.2","license":"MIT","_id":"@ariangibson/firecrawl-lite-mcp-server@1.1.2","maintainers":[{"name":"ariangibson","email":"ariangibson@gmail.com"}],"homepage":"https://github.com/ariangibson/firecrawl-lite-mcp-server#readme","bugs":{"url":"https://github.com/ariangibson/firecrawl-lite-mcp-server/issues"},"bin":{"firecrawl-lite-mcp-server":"dist/index.js"},"dist":{"shasum":"b434c4cd582aa37dbfa53811d8c9ce6be119009c","tarball":"https://registry.npmjs.org/@ariangibson/firecrawl-lite-mcp-server/-/firecrawl-lite-mcp-server-1.1.2.tgz","fileCount":4,"integrity":"sha512-anc1k6tMcZIbdeXwnLHOFMnBXYvYowSe2Utlnd9UBWHWjoCDycEJPphLW5Btby0MYnQsjbmd/ZwRujxFFLOgog==","signatures":[{"sig":"MEYCIQCgDo89vheCfZaP9f+YXfSojimFK3NvlR/OGU5pUH5vBQIhAMmW0fdgVhjSj0R68HZ0HqyvXWQTkQWnVyqnMGccHdFU","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":46132},"type":"module","engines":{"node":">=18.0.0"},"gitHead":"c16be529a16c5e53e70ef5770ef57409bd11f409","scripts":{"dev":"tsc && node dist/index.js","lint":"tsc --noEmit","build":"tsc","start":"node dist/index.js"},"_npmUser":{"name":"ariangibson","email":"ariangibson@gmail.com"},"repository":{"url":"git+https://github.com/ariangibson/firecrawl-lite-mcp-server.git","type":"git"},"_npmVersion":"11.5.2","description":"Privacy-first, standalone MCP server for web scraping and data extraction using local browser automation and your own LLM API key","directories":{},"_nodeVersion":"24.1.0","dependencies":{"ws":"^8.18.1","axios":"^1.11.0","dotenv":"^16.4.7","cheerio":"^1.1.2","express":"^5.1.0","puppeteer":"^24.17.1","@types/cheerio":"^0.22.35","@modelcontextprotocol/sdk":"^1.17.3"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.2","@types/node":"^20.10.5","@types/express":"^5.0.1"},"_npmOperationalInternal":{"tmp":"tmp/firecrawl-lite-mcp-server_1.1.2_1757192829378_0.7794623446170328","host":"s3://npm-registry-packages-npm-production"},"deprecated":"Renamed: use firecrawl-lite-mcp-server instead (npx -y firecrawl-lite-mcp-server)"},"1.4.0":{"name":"@ariangibson/firecrawl-lite-mcp-server","version":"1.4.0","license":"MIT","_id":"@ariangibson/firecrawl-lite-mcp-server@1.4.0","maintainers":[{"name":"ariangibson","email":"ariangibson@gmail.com"}],"homepage":"https://github.com/ariangibson/firecrawl-lite-mcp-server#readme","bugs":{"url":"https://github.com/ariangibson/firecrawl-lite-mcp-server/issues"},"bin":{"firecrawl-lite-mcp-server":"dist/index.js"},"dist":{"shasum":"4c5d0d6e7cf3e37b5386cdaccd825a152c3d3255","tarball":"https://registry.npmjs.org/@ariangibson/firecrawl-lite-mcp-server/-/firecrawl-lite-mcp-server-1.4.0.tgz","fileCount":10,"integrity":"sha512-pN8I/3amwi9nRGEtKLjYhabIJLQ/YE8btjPcMYKBAvChWaFMwnIa1ZKZv5+V+EZXyb5u4/Fa4QT9dRxOtFq+OA==","signatures":[{"sig":"MEQCICD2schwkT7bwGMZec17IQqNTHWYg5l8ES3sc4RAs4pDAiAXy+/EYsFxWrRNWxVd+3R//v5T3OnkSnjFYqy3kgWiIQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":94795},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"4b742ba533fb270e757465bcc9b895d6eb533a66","scripts":{"dev":"tsc && node dist/index.js","lint":"tsc --noEmit","test":"tsx --test tests/*.test.ts","build":"tsc","start":"node dist/index.js","postinstall":"node -e \"try { require('puppeteer').executablePath(); } catch (e) { console.log('Installing Chrome for Puppeteer...'); require('child_process').execSync('npx puppeteer browsers install chrome', {stdio: 'inherit'}); }\""},"_npmUser":{"name":"ariangibson","email":"ariangibson@gmail.com"},"repository":{"url":"git+https://github.com/ariangibson/firecrawl-lite-mcp-server.git","type":"git"},"_npmVersion":"10.9.8","description":"Privacy-first, single-process web scraping for AI agents: MCP server plus a Firecrawl-compatible REST API, powered by local browser automation and your own LLM key","directories":{},"_nodeVersion":"22.22.3","dependencies":{"axios":"^1.11.0","dotenv":"^16.4.7","cheerio":"^1.1.2","express":"^5.1.0","turndown":"^7.2.4","puppeteer":"^24.17.1","puppeteer-extra":"^3.3.6","turndown-plugin-gfm":"^1.0.2","@modelcontextprotocol/sdk":"^1.17.3","puppeteer-extra-plugin-stealth":"^2.11.2"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.2","typescript":"^5.9.2","@types/node":"^22.0.0","@types/express":"^5.0.1","@types/turndown":"^5.0.6"},"_npmOperationalInternal":{"tmp":"tmp/firecrawl-lite-mcp-server_1.4.0_1787507446769_0.43013785912508684","host":"s3://npm-registry-packages-npm-production"},"deprecated":"Renamed: use firecrawl-lite-mcp-server instead (npx -y firecrawl-lite-mcp-server)"}},"time":{"created":"2025-09-06T21:02:53.897Z","modified":"2026-08-23T18:03:36.170Z","1.1.1":"2025-09-06T21:02:54.177Z","1.1.2":"2025-09-06T21:07:09.567Z","1.4.0":"2026-08-23T17:50:46.911Z"},"bugs":{"url":"https://github.com/ariangibson/firecrawl-lite-mcp-server/issues"},"license":"MIT","homepage":"https://github.com/ariangibson/firecrawl-lite-mcp-server#readme","repository":{"url":"git+https://github.com/ariangibson/firecrawl-lite-mcp-server.git","type":"git"},"description":"Privacy-first, single-process web scraping for AI agents: MCP server plus a Firecrawl-compatible REST API, powered by local browser automation and your own LLM key","maintainers":[{"name":"ariangibson","email":"ariangibson@gmail.com"}],"readme":"<div align=\"center\">\n\n<img src=\"docs/banner.jpg\" alt=\"Firecrawl Lite MCP Server\" width=\"100%\" />\n\n# Firecrawl Lite MCP Server\n\n**Privacy-first web scraping for AI agents — an MCP server _and_ a Firecrawl-compatible API in one small process, powered by local browser automation and your own LLM key.**\n\n[![npm version](https://img.shields.io/npm/v/@ariangibson/firecrawl-lite-mcp-server?logo=npm&color=cb3837)](https://www.npmjs.com/package/@ariangibson/firecrawl-lite-mcp-server)\n[![Docker Pulls](https://img.shields.io/docker/pulls/ariangibson/firecrawl-lite-mcp-server?logo=docker&logoColor=white)](https://hub.docker.com/r/ariangibson/firecrawl-lite-mcp-server)\n[![Image Size](https://img.shields.io/docker/image-size/ariangibson/firecrawl-lite-mcp-server/latest?logo=docker&logoColor=white&label=image%20size)](https://hub.docker.com/r/ariangibson/firecrawl-lite-mcp-server)\n[![Build](https://img.shields.io/github/actions/workflow/status/ariangibson/firecrawl-lite-mcp-server/docker-build.yml?branch=main&logo=github&label=build)](https://github.com/ariangibson/firecrawl-lite-mcp-server/actions/workflows/docker-build.yml)\n[![Tests](https://img.shields.io/github/actions/workflow/status/ariangibson/firecrawl-lite-mcp-server/test.yml?branch=main&logo=github&label=tests)](https://github.com/ariangibson/firecrawl-lite-mcp-server/actions/workflows/test.yml)\n[![Node](https://img.shields.io/node/v/@ariangibson/firecrawl-lite-mcp-server?logo=node.js&logoColor=white)](https://nodejs.org)\n[![MCP](https://img.shields.io/badge/MCP-compatible-6E56CF)](https://modelcontextprotocol.io)\n[![License: MIT](https://img.shields.io/badge/license-MIT-yellow.svg)](LICENSE)\n\n</div>\n\n---\n\nFirecrawl Lite gives AI agents and MCP clients the ability to fetch, render, and extract web pages **without a Firecrawl account and without Firecrawl's multi-service self-hosted stack**. Pages are rendered by a local, stealth-enabled headless browser and converted to clean Markdown. It speaks two protocols from a single Node.js process:\n\n- **[Model Context Protocol](https://modelcontextprotocol.io)** — for Claude Desktop, Claude Code, Cursor, and any other MCP client.\n- **Firecrawl-compatible REST API** — for anything that speaks the Firecrawl SDK, most notably [Hermes Agent](https://hermes-agent.nousresearch.com). Point `FIRECRAWL_API_URL` at it and it behaves like a self-hosted Firecrawl instance.\n\nStructured extraction (`extract_data`, `extract_with_schema`) uses **your own LLM provider** via any OpenAI-compatible endpoint. The LLM is optional — scraping and the Firecrawl-compatible API work with no API keys at all.\n\n## Contents\n\n- [Why Firecrawl Lite](#why-firecrawl-lite)\n- [Use with Hermes Agent](#use-with-hermes-agent)\n- [Use as an MCP Server](#use-as-an-mcp-server)\n- [Available Tools](#available-tools)\n- [Configuration](#configuration)\n- [Remote Deployment](#remote-deployment)\n- [Firecrawl-compatible API](#firecrawl-compatible-api)\n- [Advanced Configuration](#advanced-configuration)\n- [Usage Examples](#usage-examples)\n- [Troubleshooting](#troubleshooting)\n- [Container Images](#container-images)\n- [Development](#development)\n- [Credits](#credits)\n- [License](#license)\n\n## Why Firecrawl Lite\n\n**One process, no infrastructure.** Self-hosting Firecrawl means Redis, a Playwright service, API and worker containers, and more. Firecrawl Lite is a single Node.js process with a bundled headless browser. Run it with `npx`, or as one container.\n\n**Privacy-first.** Scraping and rendering happen on your own machine or server. Page content is only ever sent to the LLM provider you explicitly configure — nothing is routed through a third-party scraping cloud.\n\n**Bring your own model (or none).** Extraction tools work with any OpenAI-compatible `chat/completions` endpoint: OpenAI, xAI (Grok), Anthropic, OpenRouter, or a local model via Ollama. If your agent already has a model in the loop, skip the LLM entirely and just use the scraping tools.\n\n**Built for real scraping.** Stealth browser automation, rotating user agents, configurable delays, DOM-settle detection for JS-heavy pages, optional upstream proxies (including port-range rotation), and tunable retry/backoff.\n\n## Use with Hermes Agent\n\n[Hermes Agent](https://hermes-agent.nousresearch.com) uses Firecrawl by default for its `web_extract` tool, and supports self-hosted Firecrawl via `FIRECRAWL_API_URL`. Firecrawl Lite implements the scrape endpoint Hermes needs (and returns a clear `501` for search), so you get local, private page extraction without installing Firecrawl.\n\n**1. Run Firecrawl Lite with the Firecrawl-compatible API enabled:**\n\n```bash\ndocker run -d -p 3000:3000 --name firecrawl-lite \\\n  ariangibson/firecrawl-lite-mcp-server:latest\n```\n\nThe container image enables the API by default. To run it without Docker (this stays in the foreground; set `PORT` to change the listen port):\n\n```bash\nENABLE_FIRECRAWL_API=true npx -y @ariangibson/firecrawl-lite-mcp-server\n```\n\n**2. Point Hermes at it.** In your Hermes environment (e.g. `~/.hermes/.env`):\n\n```bash\nFIRECRAWL_API_URL=http://localhost:3000\n```\n\n**3. Pick a search backend.** Firecrawl Lite renders pages; it is not a search engine. Tell Hermes to use a free search provider for `web_search` and Firecrawl Lite for `web_extract`, in `~/.hermes/config.yaml`:\n\n```yaml\nweb:\n  search_backend: ddgs        # DuckDuckGo — no API key. Or: searxng, brave-free\n  extract_backend: firecrawl  # → your Firecrawl Lite instance\n```\n\nThat's it. The native Hermes `web_extract` tool now renders through your local stealth browser. If you set `FIRECRAWL_API_KEY` on the server, set the same `FIRECRAWL_API_KEY` in Hermes and the SDK will send it as a bearer token.\n\n> **Tip:** Hermes also supports MCP servers, so you can additionally add Firecrawl Lite over MCP to get `screenshot` and the LLM-backed `extract_with_schema` tool. For most agent use, the Firecrawl-compatible API alone is the cleaner setup.\n\n## Use as an MCP Server\n\nThe fastest way to use Firecrawl Lite locally is over stdio via `npx` — no install or container required. LLM credentials are only needed for the `extract_*` tools (see [LLM provider examples](#llm-provider-examples)); omit them if you only need scraping and screenshots.\n\n### Claude Desktop\n\nAdd to `~/Library/Application Support/Claude/claude_desktop_config.json` (macOS) or `%APPDATA%\\Claude\\claude_desktop_config.json` (Windows):\n\n```json\n{\n  \"mcpServers\": {\n    \"firecrawl-lite\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@ariangibson/firecrawl-lite-mcp-server\"],\n      \"env\": {\n        \"LLM_API_KEY\": \"your_llm_api_key_here\",\n        \"LLM_PROVIDER_BASE_URL\": \"https://api.openai.com/v1\",\n        \"LLM_MODEL\": \"gpt-5.5\"\n      }\n    }\n  }\n}\n```\n\n### Claude Code (CLI)\n\n```bash\nclaude mcp add firecrawl-lite npx -- -y @ariangibson/firecrawl-lite-mcp-server \\\n  --env LLM_API_KEY=your_key \\\n  --env LLM_PROVIDER_BASE_URL=https://api.openai.com/v1 \\\n  --env LLM_MODEL=gpt-5.5\n```\n\n### Cursor\n\nAdd to your Cursor MCP configuration (`~/.cursor/mcp.json`):\n\n```json\n{\n  \"mcpServers\": {\n    \"firecrawl-lite\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@ariangibson/firecrawl-lite-mcp-server\"],\n      \"env\": {\n        \"LLM_API_KEY\": \"your_llm_api_key_here\",\n        \"LLM_PROVIDER_BASE_URL\": \"https://api.openai.com/v1\",\n        \"LLM_MODEL\": \"gpt-5.5\"\n      }\n    }\n  }\n}\n```\n\n## Available Tools\n\n| Tool | Description | Required params | Optional params | Needs LLM |\n| --- | --- | --- | --- | --- |\n| `scrape_page` | Fetch and render a single page, returning clean Markdown. | `url` | `onlyMainContent` | No |\n| `batch_scrape` | Scrape multiple URLs in one request (up to 10). | `urls[]` | `onlyMainContent` | No |\n| `screenshot` | Capture a screenshot of a page via the stealth browser. | `url` | `width` (1920), `height` (1080), `fullPage` (false) | No |\n| `extract_data` | Extract structured data from pages using a natural-language prompt. | `urls[]`, `prompt` | — | Yes |\n| `extract_with_schema` | Extract data conforming to a supplied JSON Schema. | `urls[]`, `schema` | `prompt` | Yes |\n\n## Configuration\n\nAll configuration is via environment variables. Everything has a sensible default; the LLM variables are only required if you use the `extract_*` tools.\n\n### LLM (required for `extract_data` / `extract_with_schema`)\n\n| Variable | Description |\n| --- | --- |\n| `LLM_API_KEY` | API key for your LLM provider. |\n| `LLM_PROVIDER_BASE_URL` | Base URL of an OpenAI-compatible API (the server calls `{base_url}/chat/completions`). |\n| `LLM_MODEL` | Model name to use for extraction. |\n\n### Optional LLM tuning\n\nThese are passed straight through to the provider's `chat/completions` request. Leave any of them unset to use the default; the optional sampling parameters are omitted from the request entirely when unset.\n\n| Variable | Default | Notes |\n| --- | --- | --- |\n| `LLM_TEMPERATURE` | `0.1` | Sampling temperature. |\n| `LLM_MAX_TOKENS` | `2000` | Maximum tokens in the response. Raise this if extractions are being truncated. |\n| `LLM_TOP_P` | _unset_ | Nucleus sampling; omitted from the request unless set. |\n| `LLM_REASONING_EFFORT` | _unset_ | `reasoning_effort` for reasoning-capable models; omitted unless set. |\n\n### LLM provider examples\n\n```bash\n# OpenAI\nLLM_PROVIDER_BASE_URL=https://api.openai.com/v1\nLLM_MODEL=gpt-5.5\n\n# xAI (Grok)\nLLM_PROVIDER_BASE_URL=https://api.x.ai/v1\nLLM_MODEL=grok-4\n\n# Anthropic\nLLM_PROVIDER_BASE_URL=https://api.anthropic.com/v1\nLLM_MODEL=claude-haiku-4-5\n\n# OpenRouter\nLLM_PROVIDER_BASE_URL=https://openrouter.ai/api/v1\nLLM_MODEL=openai/gpt-5.5\n\n# Local (Ollama)\nLLM_PROVIDER_BASE_URL=http://localhost:11434/v1\nLLM_MODEL=llama3.3\n```\n\n### HTTP endpoints\n\nAll HTTP endpoints are **disabled by default** when running from `npx`, and the server speaks MCP over stdio. Enabling any of them switches the process to an HTTP server listening on `PORT` (stdio is not served); `/health` is always available in that mode. The Docker image enables `/mcp` and the Firecrawl-compatible API by default.\n\n| Variable | Default | Enables |\n| --- | --- | --- |\n| `ENABLE_FIRECRAWL_API` | `false` (`true` in Docker) | `POST /v2/scrape` (and `/v1/scrape`) — Firecrawl-compatible API for Hermes Agent and Firecrawl SDK clients. |\n| `FIRECRAWL_API_KEY` | _unset_ | If set, the Firecrawl-compatible API requires `Authorization: Bearer <key>`. |\n| `ENABLE_HTTP_STREAMABLE_ENDPOINT` | `false` (`true` in Docker) | `/mcp` — Streamable HTTP transport for Claude Code and other remote MCP clients. |\n| `ENABLE_SSE_ENDPOINT` | `false` | `/sse` — deprecated SSE transport (Claude Desktop via `mcp-proxy`). |\n| `PORT` | `3000` | HTTP listen port. |\n\nSee [`.env.example`](.env.example) for the full, annotated list of variables.\n\n## Remote Deployment\n\n### Docker\n\n```bash\ndocker run -d \\\n  -p 3000:3000 \\\n  -e LLM_API_KEY=your_key_here \\\n  -e LLM_PROVIDER_BASE_URL=https://api.openai.com/v1 \\\n  -e LLM_MODEL=gpt-5.5 \\\n  ariangibson/firecrawl-lite-mcp-server:latest\n```\n\nThe LLM variables are optional — drop them if you only need the Firecrawl-compatible API or the scraping tools.\n\n### Docker Compose / Portainer / Swarm\n\nA ready-to-use [`docker-compose.yml`](docker-compose.yml) is included. Set your variables in a `.env` file and deploy:\n\n```bash\ndocker compose up -d\n```\n\n> **Note for Docker Swarm / Portainer:** the published image is Alpine-based and does **not** include `curl`. The bundled compose file uses a `wget`-based health check for this reason — see [Troubleshooting](#container-keeps-restarting-or-is-killed-with-sigterm) if you have replaced it with a `curl`-based check.\n\n### Remote MCP client configuration\n\n**Claude Code (Streamable HTTP):**\n\n```bash\nclaude mcp add firecrawl-lite-remote http://your-server:3000/mcp -t http\n```\n\n**Claude Desktop — Connectors (recommended, HTTPS only):**\n\nSettings → Connectors → add `https://your-server.com:3000/mcp`. Requires a valid TLS certificate.\n\n**Claude Desktop — `mcp-proxy` (HTTP fallback, no certificate):**\n\n```bash\npip install mcp-proxy\n```\n\n```json\n{\n  \"mcpServers\": {\n    \"firecrawl-lite\": {\n      \"command\": \"mcp-proxy\",\n      \"args\": [\"http://your-server:3000/sse\"]\n    }\n  }\n}\n```\n\nRequires `ENABLE_SSE_ENDPOINT=true` on the server.\n\n## Firecrawl-compatible API\n\nWhen `ENABLE_FIRECRAWL_API=true`, the server implements the subset of the [Firecrawl v2 API](https://docs.firecrawl.dev/api-reference/endpoint/scrape) that SDK clients use for page extraction. It works with the official `firecrawl-py` / `@mendable/firecrawl-js` SDKs by setting `api_url` / `apiUrl`.\n\n| Endpoint | Behaviour |\n| --- | --- |\n| `POST /v2/scrape` (and `/v1/scrape`) | Renders the page and returns `{ success, data }` where `data` is a Firecrawl Document: `markdown`, `html`, `rawHtml`, `links` (per requested `formats`) plus `metadata` (`title`, `description`, `language`, `sourceURL`, `url`, `statusCode` — always `200` for a successful render). The `screenshot` format is accepted but ignored. |\n| `POST /v2/search` (and `/v1/search`) | Returns `501` with a clear error. Firecrawl Lite is a renderer, not a search engine — pair it with a dedicated search backend. |\n\nRequest body: `{ \"url\": \"https://…\", \"formats\": [\"markdown\", \"html\"], \"onlyMainContent\": true }`. `formats` defaults to `[\"markdown\"]`; both string and `{ \"type\": \"markdown\" }` entries are accepted.\n\n```bash\ncurl -X POST http://localhost:3000/v2/scrape \\\n  -H 'Content-Type: application/json' \\\n  -d '{\"url\":\"https://example.com\",\"formats\":[\"markdown\"]}'\n```\n\n```python\nfrom firecrawl import Firecrawl\n\nfc = Firecrawl(api_key=\"unused\", api_url=\"http://localhost:3000\")\ndoc = fc.scrape(\"https://example.com\", formats=[\"markdown\"])\nprint(doc.markdown)\n```\n\nNot implemented: crawl, map, batch jobs, extract, and other async job-based endpoints.\n\n## Advanced Configuration\n\n### Proxy\n\nRoute the scraping browser through an upstream proxy. A port range (e.g. `:10001-10010`) enables automatic rotation across ports.\n\n```bash\nPROXY_SERVER_URL=http://proxy.example.com:10001-10010\nPROXY_SERVER_USERNAME=your-username\nPROXY_SERVER_PASSWORD=your-password\n```\n\nBy default the proxy is used **only for scraping target sites** — LLM provider API calls go out directly. Routing your own LLM calls through a rotating (often residential) proxy is slower, can trip provider abuse detection, and may fail TLS. If you specifically need the LLM call proxied as well, opt in with `PROXY_LLM_API=true` (default: `false`).\n\n### Anti-detection and rate limiting\n\n`SCRAPE_USER_AGENT` accepts either a single string or a JSON array of strings to rotate through. When using a JSON array, keep it on a single line.\n\n```bash\nSCRAPE_USER_AGENT=[\"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) ... Safari/537.36\",\"Mozilla/5.0 (Windows NT 10.0; Win64; x64) ... Safari/537.36\"]\nSCRAPE_VIEWPORT_WIDTH=1920\nSCRAPE_VIEWPORT_HEIGHT=1080\nSCRAPE_DELAY_MIN=1000          # min delay before navigation (ms)\nSCRAPE_DELAY_MAX=3000          # max delay before navigation (ms)\nSCRAPE_BATCH_DELAY_MIN=2000    # min delay between batch requests (ms)\nSCRAPE_BATCH_DELAY_MAX=5000    # max delay between batch requests (ms)\nSCRAPE_SETTLE_MAX_MS=3000      # max wait for the DOM to stop changing after load (ms); raise for slow JS sites\n```\n\nAfter load the scraper scrolls to trigger lazy/AJAX content and waits for the DOM to settle (exiting early once stable). Pages that inject content via a long `setTimeout` may need a higher `SCRAPE_SETTLE_MAX_MS`.\n\n### Retries\n\nEach scrape or screenshot is attempted up to `FIRECRAWL_RETRY_MAX_ATTEMPTS` times (default `3`), advancing through the proxy and user-agent rotation on each attempt.\n\n```bash\nFIRECRAWL_RETRY_MAX_ATTEMPTS=3\n```\n\n## Usage Examples\n\n**Scrape a page**\n\n```json\n{ \"name\": \"scrape_page\", \"arguments\": { \"url\": \"https://example.com\" } }\n```\n\n**Batch scrape**\n\n```json\n{\n  \"name\": \"batch_scrape\",\n  \"arguments\": {\n    \"urls\": [\"https://example.com\", \"https://example.org\"],\n    \"onlyMainContent\": true\n  }\n}\n```\n\n**Extract with a prompt**\n\n```json\n{\n  \"name\": \"extract_data\",\n  \"arguments\": {\n    \"urls\": [\"https://example.com\"],\n    \"prompt\": \"Extract the main article title and a one-sentence summary.\"\n  }\n}\n```\n\n**Extract with a JSON Schema**\n\n```json\n{\n  \"name\": \"extract_with_schema\",\n  \"arguments\": {\n    \"urls\": [\"https://example.com\"],\n    \"schema\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"title\": { \"type\": \"string\" },\n        \"description\": { \"type\": \"string\" }\n      }\n    }\n  }\n}\n```\n\n## Troubleshooting\n\n### Chrome / Chromium issues\n\nThe container image bundles Chromium. When run via `npx`, Chrome is downloaded automatically on first install. If it's missing (scrapes fail with `Could not find Chrome`):\n\n```bash\nnpx puppeteer browsers install chrome\n# or reset a corrupted install\nrm -rf ~/.cache/puppeteer && npx puppeteer browsers install chrome\n```\n\n### Extraction returns an error\n\nIf `scrape_page` works but `extract_data` fails, the problem is the LLM call, not scraping. The server logs the upstream status, error code, and response body to stderr (`LLM extract_data request failed: ...`) and surfaces the HTTP status in the tool result. Common causes:\n\n- **HTTP 401** — invalid `LLM_API_KEY`.\n- **HTTP 400** — wrong `LLM_MODEL`, or a tuning parameter the model rejects (e.g. `LLM_MAX_TOKENS` above the model's limit, or `LLM_REASONING_EFFORT` on a non-reasoning model).\n- **HTTP 429** — provider rate limit.\n\n### Hermes says \"Firecrawl backend selected but not configured\" or extract fails\n\n- Confirm `FIRECRAWL_API_URL` is set in the Hermes environment and reachable: `curl http://your-server:3000/health` should report `\"endpoints\": { ..., \"firecrawlApi\": \"enabled\" }`.\n- If you set `FIRECRAWL_API_KEY` on the server, Hermes needs the same value in `FIRECRAWL_API_KEY`.\n- `web_search` failing is expected unless `web.search_backend` points at a real search provider (see [Use with Hermes Agent](#use-with-hermes-agent)).\n\n### Container keeps restarting or is killed with `SIGTERM`\n\nIf the logs show the server start (`listening on port 3000`) and then exit with `npm error signal SIGTERM`, the container is being killed by a **failing health check**, not by the app. The Alpine image does not include `curl`, so a `curl`-based health check always fails and Swarm restarts the task in a loop. Use a `wget`-based check (busybox provides `wget`):\n\n```yaml\nhealthcheck:\n  test: [\"CMD-SHELL\", \"wget --no-verbose --tries=1 --spider http://localhost:3000/health || exit 1\"]\n  interval: 30s\n  timeout: 10s\n  retries: 3\n  start_period: 40s\n```\n\nThe bundled `docker-compose.yml` already uses this form, and the image's built-in `HEALTHCHECK` uses Node, so neither needs `curl`.\n\n## Container Images\n\nPre-built, multi-architecture (`amd64`, `arm64`) images are published automatically on every push to `main` and on release:\n\n- **Docker Hub:** `ariangibson/firecrawl-lite-mcp-server:latest`\n- **GitHub Container Registry:** `ghcr.io/ariangibson/firecrawl-lite-mcp-server:latest`\n\n## Development\n\nRequires Node.js 20 or newer (the container image uses Node 22).\n\n```bash\nnpm install        # install dependencies (downloads Chrome for Puppeteer)\nnpm run build      # compile TypeScript to dist/\nnpm run lint       # type-check without emitting\nnpm test           # run the unit test suite\nnpm start          # run the built server\n```\n\nThe code is organised so that the interesting behaviour is testable without a browser or network:\n\n| Module | Responsibility |\n| --- | --- |\n| `src/config.ts` | `loadConfig(env)` — every environment variable, parsed once into a typed object. |\n| `src/browser.ts` | One stealth browser session: launch flags, proxy/user-agent, navigation, cleanup, retries. The browser is injected. |\n| `src/scraper.ts` | `createScraper(config, deps)` → `{ scrape, screenshot }`: rotation, retries, scroll/settle heuristics, HTML → Markdown. |\n| `src/htmlToMarkdown.ts` | HTML cleaning, Markdown/text conversion, page metadata. |\n| `src/firecrawlApi.ts` | Firecrawl-compatible request parsing, Document shaping, bearer auth. |\n| `src/index.ts` | MCP tool definitions, LLM extraction, and the stdio/HTTP transports. |\n\nUnit tests run in well under a second: the scraper is exercised end to end through a stub browser and a fake clock (`tests/scraper.test.ts`), alongside the pure helpers. CI runs them against Node 20, 22, and 24. The scripts in `dev-scripts/` exercise the tools against live sites.\n\n## Credits\n\nInspired by the excellent work of the [Firecrawl](https://firecrawl.com) team at Mendable.ai and their official [Firecrawl MCP Server](https://github.com/firecrawl/firecrawl-mcp-server). Firecrawl Lite is an independent, self-hosted take on the same idea — huge thanks to them for pioneering web scraping for the MCP ecosystem.\n\nLooking for a fully managed, enterprise-grade scraping platform? Check out [firecrawl.com](https://firecrawl.com).\n\n## License\n\nMIT — see [LICENSE](LICENSE).\n","readmeFilename":"README.md"}