{"_id":"@dy531014023/fetcher-mcp","_rev":"2-d646ad1817381862ed03e4880e45a6ac","name":"@dy531014023/fetcher-mcp","dist-tags":{"latest":"0.4.1"},"versions":{"0.4.0":{"name":"@dy531014023/fetcher-mcp","version":"0.4.0","keywords":["mcp","playwright","web-scraping","readability","content-extraction"],"author":"","license":"ISC","_id":"@dy531014023/fetcher-mcp@0.4.0","maintainers":[{"name":"dy531014023","email":"531014023@qq.com"}],"bin":{"fetcher-mcp":"build/index.js"},"dist":{"shasum":"e4c0943cdbb1c8aea6e54f9664b1408e2120de18","tarball":"https://registry.npmjs.org/@dy531014023/fetcher-mcp/-/fetcher-mcp-0.4.0.tgz","fileCount":22,"integrity":"sha512-jGNEaDM5yuh0g41t1vUbf2bqw1HW9y+i2WW8nLsZlce7m9ZqRIyUhsMmFPb9dTSOHNuAIjIvBCZe/SLTKKgs4g==","signatures":[{"sig":"MEQCIB6nyl8XqK053F8geesDsBWKF4caZW13kW+JYhmrDYZIAiBGkJ+ZBkgXqUNdvaaK1HRLyqt+Gk0Qz7IP+eiNAfPP/A==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":73177},"main":"index.js","type":"module","gitHead":"8754aff66e3d9207502207bf82a493f45f556bb8","private":false,"scripts":{"build":"rimraf build && tsc && node -e \"require('fs').chmodSync('build/index.js', '755')\"","watch":"tsc --watch","inspector":"npm run build && npx @modelcontextprotocol/inspector build/index.js --debug","postinstall":"playwright install chromium","install-browser":"npx playwright install chromium"},"_npmUser":{"name":"dy531014023","email":"531014023@qq.com"},"_npmVersion":"11.9.0","description":"MCP server for fetching web content using Playwright browser","directories":{},"_nodeVersion":"22.22.0","dependencies":{"jsdom":"^24.0.0","express":"^4.18.2","turndown":"^7.1.2","playwright":"^1.42.1","turndown-plugin-gfm":"^1.0.2","@mozilla/readability":"^0.5.0","@modelcontextprotocol/sdk":"^1.10.2"},"_hasShrinkwrap":false,"devDependencies":{"rimraf":"^6.1.3","typescript":"^5.3.3","@types/node":"^20.17.24","@types/jsdom":"^21.1.6","@types/express":"^4.17.21","@types/turndown":"^5.0.4"},"_npmOperationalInternal":{"tmp":"tmp/fetcher-mcp_0.4.0_1773493431184_0.5746275085607222","host":"s3://npm-registry-packages-npm-production"}},"0.4.1":{"name":"@dy531014023/fetcher-mcp","version":"0.4.1","description":"MCP server for fetching web content using Playwright browser","private":false,"type":"module","bin":{"fetcher-mcp":"build/index.js"},"scripts":{"build":"rimraf build && tsc && node -e \"require('fs').chmodSync('build/index.js', '755')\"","watch":"tsc --watch","inspector":"npm run build && npx @modelcontextprotocol/inspector build/index.js --debug","install-browser":"npx playwright install chromium","postinstall":"playwright install chromium"},"dependencies":{"@modelcontextprotocol/sdk":"^1.10.2","@mozilla/readability":"^0.5.0","express":"^4.18.2","jsdom":"^24.0.0","playwright":"^1.42.1","turndown":"^7.1.2","turndown-plugin-gfm":"^1.0.2"},"devDependencies":{"@types/express":"^4.17.21","@types/jsdom":"^21.1.6","@types/node":"^20.17.24","@types/turndown":"^5.0.4","rimraf":"^6.1.3","typescript":"^5.3.3"},"main":"index.js","keywords":["mcp","playwright","web-scraping","readability","content-extraction"],"author":"","license":"ISC","gitHead":"8754aff66e3d9207502207bf82a493f45f556bb8","_id":"@dy531014023/fetcher-mcp@0.4.1","_nodeVersion":"22.22.0","_npmVersion":"11.9.0","dist":{"integrity":"sha512-ujQBgRnOrOKnml3xdnzZZzQ4CZNM3/wT08e/Z12Bydk+XMEcMOLV8+FyZ1Oe/mgB1JZg+Hk1St+3ihXbitTlEA==","shasum":"1a46147bde118323e87e46269df3e3a65e0865cb","tarball":"https://registry.npmjs.org/@dy531014023/fetcher-mcp/-/fetcher-mcp-0.4.1.tgz","fileCount":22,"unpackedSize":73952,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQDE+jcszR4/S4HFtiLBc5BcsW3BvEaLcp5cChLWvXP0FQIhAIqfG4fAf2hNTziP68aEsUdxkfJASEX1YyBRwP1wscMh"}]},"_npmUser":{"name":"dy531014023","email":"531014023@qq.com"},"directories":{},"maintainers":[{"name":"dy531014023","email":"531014023@qq.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/fetcher-mcp_0.4.1_1773496066521_0.1716888481844121"},"_hasShrinkwrap":false}},"time":{"created":"2026-03-14T13:03:51.107Z","modified":"2026-03-14T13:47:46.771Z","0.4.0":"2026-03-14T13:03:51.358Z","0.4.1":"2026-03-14T13:47:46.670Z"},"license":"ISC","keywords":["mcp","playwright","web-scraping","readability","content-extraction"],"description":"MCP server for fetching web content using Playwright browser","maintainers":[{"name":"dy531014023","email":"531014023@qq.com"}],"readme":"<div align=\"center\">\r\n  <img src=\"https://raw.githubusercontent.com/jae-jae/fetcher-mcp/refs/heads/main/icon.svg\" width=\"100\" height=\"100\" alt=\"Fetcher MCP Icon\" />\r\n</div>\r\n\r\n[中文](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=zh) |\r\n[Deutsch](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=de) |\r\n[Español](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=es) |\r\n[français](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=fr) |\r\n[日本語](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=ja) |\r\n[한국어](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=ko) |\r\n[Português](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=pt) |\r\n[Русский](https://www.readme-i18n.com/jae-jae/fetcher-mcp?lang=ru)\r\n\r\n# Fetcher MCP\r\n\r\nMCP server for fetch web page content using Playwright headless browser.\r\n\r\n> 🌟 **Recommended**: [OllaMan](https://ollaman.com/) - Powerful Ollama AI Model Manager.\r\n\r\n## Advantages\r\n\r\n- **JavaScript Support**: Unlike traditional web scrapers, Fetcher MCP uses Playwright to execute JavaScript, making it capable of handling dynamic web content and modern web applications.\r\n\r\n- **Intelligent Content Extraction**: Built-in Readability algorithm automatically extracts the main content from web pages, removing ads, navigation, and other non-essential elements.\r\n\r\n- **Flexible Output Format**: Supports both HTML and Markdown output formats, making it easy to integrate with various downstream applications.\r\n\r\n- **Parallel Processing**: The `fetch_urls` tool enables concurrent fetching of multiple URLs, significantly improving efficiency for batch operations.\r\n\r\n- **Resource Optimization**: Automatically blocks unnecessary resources (images, stylesheets, fonts, media) to reduce bandwidth usage and improve performance.\r\n\r\n- **Robust Error Handling**: Comprehensive error handling and logging ensure reliable operation even when dealing with problematic web pages.\r\n\r\n- **Configurable Parameters**: Fine-grained control over timeouts, content extraction, and output formatting to suit different use cases.\r\n\r\n## Quick Start\r\n\r\nRun directly with npx:\r\n\r\n```bash\r\nnpx -y fetcher-mcp\r\n```\r\n\r\nFirst time setup - install the required browser by running the following command in your terminal:\r\n\r\n```bash\r\nnpx playwright install chromium\r\n```\r\n\r\n### HTTP and SSE Transport\r\n\r\nUse the `--transport=http` parameter to start both Streamable HTTP endpoint and SSE endpoint services simultaneously:\r\n\r\n```bash\r\nnpx -y fetcher-mcp --log --transport=http --host=0.0.0.0 --port=3000\r\n```\r\n\r\nAfter startup, the server provides the following endpoints:\r\n\r\n- `/mcp` - Streamable HTTP endpoint (modern MCP protocol)\r\n- `/sse` - SSE endpoint (legacy MCP protocol)\r\n\r\nClients can choose which method to connect based on their needs.\r\n\r\n### Debug Mode\r\n\r\nRun with the `--debug` option to show the browser window for debugging:\r\n\r\n```bash\r\nnpx -y fetcher-mcp --debug\r\n```\r\n\r\n## Configuration MCP\r\n\r\nConfigure this MCP server in Claude Desktop:\r\n\r\nOn MacOS: `~/Library/Application Support/Claude/claude_desktop_config.json`\r\n\r\nOn Windows: `%APPDATA%/Claude/claude_desktop_config.json`\r\n\r\n```json\r\n{\r\n  \"mcpServers\": {\r\n    \"fetcher\": {\r\n      \"command\": \"npx\",\r\n      \"args\": [\"-y\", \"fetcher-mcp\"]\r\n    }\r\n  }\r\n}\r\n```\r\n\r\n## Docker Deployment\r\n\r\n### Running with Docker\r\n\r\n```bash\r\ndocker run -p 3000:3000 ghcr.io/jae-jae/fetcher-mcp:latest\r\n```\r\n\r\n### Deploying with Docker Compose\r\n\r\nCreate a `docker-compose.yml` file:\r\n\r\n```yaml\r\nversion: \"3.8\"\r\n\r\nservices:\r\n  fetcher-mcp:\r\n    image: ghcr.io/jae-jae/fetcher-mcp:latest\r\n    container_name: fetcher-mcp\r\n    restart: unless-stopped\r\n    ports:\r\n      - \"3000:3000\"\r\n    environment:\r\n      - NODE_ENV=production\r\n    # Using host network mode on Linux hosts can improve browser access efficiency\r\n    # network_mode: \"host\"\r\n    volumes:\r\n      # For Playwright, may need to share certain system paths\r\n      - /tmp:/tmp\r\n    # Health check\r\n    healthcheck:\r\n      test: [\"CMD\", \"wget\", \"--spider\", \"-q\", \"http://localhost:3000\"]\r\n      interval: 30s\r\n      timeout: 10s\r\n      retries: 3\r\n```\r\n\r\nThen run:\r\n\r\n```bash\r\ndocker-compose up -d\r\n```\r\n\r\n## Features\r\n\r\n- `fetch_url` - Retrieve web page content from a specified URL\r\n\r\n  - Uses Playwright headless browser to parse JavaScript\r\n  - Supports intelligent extraction of main content and conversion to Markdown\r\n  - Supports the following parameters:\r\n    - `url`: The URL of the web page to fetch (required parameter)\r\n    - `timeout`: Page loading timeout in milliseconds, default is 30000 (30 seconds)\r\n    - `waitUntil`: Specifies when navigation is considered complete, options: 'load', 'domcontentloaded', 'networkidle', 'commit', default is 'load'\r\n    - `extractContent`: Whether to intelligently extract the main content, default is true\r\n    - `maxLength`: Maximum length of returned content (in characters), default is no limit\r\n    - `returnHtml`: Whether to return HTML content instead of Markdown, default is false\r\n    - `waitForNavigation`: Whether to wait for additional navigation after initial page load (useful for sites with anti-bot verification), default is false\r\n    - `navigationTimeout`: Maximum time to wait for additional navigation in milliseconds, default is 10000 (10 seconds)\r\n    - `disableMedia`: Whether to disable media resources (images, stylesheets, fonts, media), default is true\r\n    - `debug`: Whether to enable debug mode (showing browser window), overrides the --debug command line flag if specified\r\n\r\n- `fetch_urls` - Batch retrieve web page content from multiple URLs in parallel\r\n  - Uses multi-tab parallel fetching for improved performance\r\n  - Returns combined results with clear separation between webpages\r\n  - Supports the following parameters:\r\n    - `urls`: Array of URLs to fetch (required parameter)\r\n    - Other parameters are the same as `fetch_url`\r\n\r\n- `browser_install` - Install Playwright Chromium browser binary automatically\r\n\r\n  - Installs required Chromium browser binary when not available\r\n  - Automatically suggested when browser installation errors occur\r\n  - Supports the following parameters:\r\n    - `withDeps`: Install system dependencies required by Chromium browser, default is false\r\n    - `force`: Force installation even if Chromium is already installed, default is false\r\n\r\n## Tips\r\n\r\n### Handling Special Website Scenarios\r\n\r\n#### Dealing with Anti-Crawler Mechanisms\r\n\r\n- **Wait for Complete Loading**: For websites using CAPTCHA, redirects, or other verification mechanisms, include in your prompt:\r\n\r\n  ```\r\n  Please wait for the page to fully load\r\n  ```\r\n\r\n  This will use the `waitForNavigation: true` parameter.\r\n\r\n- **Increase Timeout Duration**: For websites that load slowly:\r\n  ```\r\n  Please set the page loading timeout to 60 seconds\r\n  ```\r\n  This adjusts both `timeout` and `navigationTimeout` parameters accordingly.\r\n\r\n#### Content Retrieval Adjustments\r\n\r\n- **Preserve Original HTML Structure**: When content extraction might fail:\r\n\r\n  ```\r\n  Please preserve the original HTML content\r\n  ```\r\n\r\n  Sets `extractContent: false` and `returnHtml: true`.\r\n\r\n- **Fetch Complete Page Content**: When extracted content is too limited:\r\n\r\n  ```\r\n  Please fetch the complete webpage content instead of just the main content\r\n  ```\r\n\r\n  Sets `extractContent: false`.\r\n\r\n- **Return Content as HTML**: When HTML format is needed instead of default Markdown:\r\n  ```\r\n  Please return the content in HTML format\r\n  ```\r\n  Sets `returnHtml: true`.\r\n\r\n### Debugging and Authentication\r\n\r\n#### Enabling Debug Mode\r\n\r\n- **Dynamic Debug Activation**: To display the browser window during a specific fetch operation:\r\n  ```\r\n  Please enable debug mode for this fetch operation\r\n  ```\r\n  This sets `debug: true` even if the server was started without the `--debug` flag.\r\n\r\n#### Using Custom Cookies for Authentication\r\n\r\n- **Manual Login**: To login using your own credentials:\r\n\r\n  ```\r\n  Please run in debug mode so I can manually log in to the website\r\n  ```\r\n\r\n  Sets `debug: true` or uses the `--debug` flag, keeping the browser window open for manual login.\r\n\r\n- **Interacting with Debug Browser**: When debug mode is enabled:\r\n\r\n  1. The browser window remains open\r\n  2. You can manually log into the website using your credentials\r\n  3. After login is complete, content will be fetched with your authenticated session\r\n\r\n- **Enable Debug for Specific Requests**: Even if the server is already running, you can enable debug mode for a specific request:\r\n  ```\r\n  Please enable debug mode for this authentication step\r\n  ```\r\n  Sets `debug: true` for this specific request only, opening the browser window for manual login.\r\n\r\n## Development\r\n\r\n### Install Dependencies\r\n\r\n```bash\r\nnpm install\r\n```\r\n\r\n### Install Playwright Browser\r\n\r\nInstall the browsers needed for Playwright:\r\n\r\n```bash\r\nnpm run install-browser\r\n```\r\n\r\n### Build the Server\r\n\r\n```bash\r\nnpm run build\r\n```\r\n\r\n## Debugging\r\n\r\nUse MCP Inspector for debugging:\r\n\r\n```bash\r\nnpm run inspector\r\n```\r\n\r\nYou can also enable visible browser mode for debugging:\r\n\r\n```bash\r\nnode build/index.js --debug\r\n```\r\n\r\n## Related Projects\r\n\r\n- [g-search-mcp](https://github.com/jae-jae/g-search-mcp): A powerful MCP server for Google search that enables parallel searching with multiple keywords simultaneously. Perfect for batch search operations and data collection.\r\n\r\n## License\r\n\r\nLicensed under the [MIT License](https://choosealicense.com/licenses/mit/)\r\n\r\n[![Powered by DartNode](https://dartnode.com/branding/DN-Open-Source-sm.png)](https://dartnode.com \"Powered by DartNode - Free VPS for Open Source\")\r\n","readmeFilename":"README.md"}