{"_id":"@advance-labs/crawler","_rev":"3-d7882db5e51f42bbdc1bc0dc02a2b830","name":"@advance-labs/crawler","dist-tags":{"latest":"0.2.2"},"versions":{"0.2.0":{"name":"@advance-labs/crawler","version":"0.2.0","license":"MIT","_id":"@advance-labs/crawler@0.2.0","maintainers":[{"name":"zordhalo","email":"lucas@advancelabs.dev"}],"homepage":"https://github.com/Advance-Labs/aeo-toolkit#readme","bugs":{"url":"https://github.com/Advance-Labs/aeo-toolkit/issues"},"dist":{"shasum":"c199f1a7bc655b7d328f2a6d55edd91e7f7b4f70","tarball":"https://registry.npmjs.org/@advance-labs/crawler/-/crawler-0.2.0.tgz","fileCount":6,"integrity":"sha512-OFcMHCCu1UKdyL5Ny/Yfio6d5c964Vi42N5JLV2UtRAa13T0/nH57rK/Fhvy8ZQQnhmv78+nZP9wA6Uslm9Spw==","signatures":[{"sig":"MEYCIQCNjpDfedgkFacw8JIHjlsaLmD8e2spiR6kv0BNTyyj9QIhAIX5AdRxOjyVZPzYnv+PhiTI9MBnbflTnJHgfJ4SEkkC","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":95830},"main":"./dist/index.js","type":"module","_from":"file:advance-labs-crawler-0.2.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"dev":"tsup --watch","lint":"eslint src","test":"vitest run","build":"tsup","clean":"rimraf dist .turbo","typecheck":"tsc --noEmit"},"_npmUser":{"name":"zordhalo","email":"lucas@advancelabs.dev"},"_resolved":"/private/var/folders/lc/qg6736_d3pxb5ds7q4x5gtf80000gn/T/d88ba2384d886494b49ded87686655f9/advance-labs-crawler-0.2.0.tgz","_integrity":"sha512-OFcMHCCu1UKdyL5Ny/Yfio6d5c964Vi42N5JLV2UtRAa13T0/nH57rK/Fhvy8ZQQnhmv78+nZP9wA6Uslm9Spw==","repository":{"url":"git+https://github.com/Advance-Labs/aeo-toolkit.git","type":"git","directory":"packages/crawler"},"_npmVersion":"11.18.0","description":"Polite, bounded HTTP crawler — sitemap-first discovery, link-following BFS, robots.txt, and site-file detection.","directories":{},"_nodeVersion":"25.8.1","dependencies":{"p-limit":"^6.1.0","robots-parser":"^3.0.1","fast-xml-parser":"^4.5.0","@advance-labs/types":"0.2.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.3.5","rimraf":"^6.0.1","vitest":"^3.2.6","typescript":"^5.7.2","@types/node":"^22.10.2","@advance-labs/config":"0.1.0"},"_npmOperationalInternal":{"tmp":"tmp/crawler_0.2.0_1788059625020_0.24910519579677337","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@advance-labs/crawler","version":"0.2.1","keywords":["crawler","web-crawler","robots-txt","sitemap","scraping","bfs","seo"],"license":"Apache-2.0","_id":"@advance-labs/crawler@0.2.1","maintainers":[{"name":"zordhalo","email":"lucas@advancelabs.dev"}],"homepage":"https://github.com/Advance-Labs/aeo-toolkit#readme","bugs":{"url":"https://github.com/Advance-Labs/aeo-toolkit/issues"},"dist":{"shasum":"5e30d5d6926eafec28c4818840dfbf374dc3a5e7","tarball":"https://registry.npmjs.org/@advance-labs/crawler/-/crawler-0.2.1.tgz","fileCount":6,"integrity":"sha512-IP4Tj5KzmND29JYUq1hIxaWuI+47q2hiuJ7C8WMPB3usfvbfsneBMEnQGzqs11rZim0MDVSentg1RsUl9uCU8A==","signatures":[{"sig":"MEUCIApSXc/kQBzdKy0sJb7FMxAVo4xg1hfZ0jnmLAx+IClEAiEA3Uchlaoyk5SGt21Ak3NNGD1zICl8WB6zyzmC+Pu9RRg=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":106235},"main":"./dist/index.js","type":"module","_from":"file:advance-labs-crawler-0.2.1.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"dev":"tsup --watch","lint":"eslint src","test":"vitest run","build":"tsup","clean":"rimraf dist .turbo","typecheck":"tsc --noEmit"},"_npmUser":{"name":"zordhalo","email":"lucas@advancelabs.dev"},"_resolved":"/tmp/8b5a7b4c63690500dece9fa5929692c9/advance-labs-crawler-0.2.1.tgz","_integrity":"sha512-IP4Tj5KzmND29JYUq1hIxaWuI+47q2hiuJ7C8WMPB3usfvbfsneBMEnQGzqs11rZim0MDVSentg1RsUl9uCU8A==","repository":{"url":"git+https://github.com/Advance-Labs/aeo-toolkit.git","type":"git","directory":"packages/crawler"},"_npmVersion":"11.19.0","description":"Polite, bounded HTTP crawler — sitemap-first discovery, link-following BFS, robots.txt, and site-file detection.","directories":{},"_nodeVersion":"24.20.0","dependencies":{"p-limit":"^6.1.0","robots-parser":"^3.0.1","fast-xml-parser":"^4.5.0","@advance-labs/types":"0.2.1"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.3.5","rimraf":"^6.0.1","vitest":"^3.2.6","typescript":"^5.7.2","@types/node":"^22.10.2","@advance-labs/config":"0.1.0"},"_npmOperationalInternal":{"tmp":"tmp/crawler_0.2.1_1789019425352_0.36281043914032485","host":"s3://npm-registry-packages-npm-production"}},"0.2.2":{"name":"@advance-labs/crawler","version":"0.2.2","type":"module","description":"Polite, bounded HTTP crawler — sitemap-first discovery, link-following BFS, robots.txt, and site-file detection.","keywords":["crawler","web-crawler","robots-txt","sitemap","scraping","bfs","seo"],"license":"Apache-2.0","author":{"name":"Lucas Krawczak","email":"lucas@advancelabs.dev","url":"https://advancelabs.dev/lucas"},"main":"./dist/index.js","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"dependencies":{"fast-xml-parser":"^4.5.0","p-limit":"^6.1.0","robots-parser":"^3.0.1","@advance-labs/types":"0.2.2"},"devDependencies":{"@types/node":"^22.10.2","rimraf":"^6.0.1","tsup":"^8.3.5","typescript":"^5.7.2","vitest":"^3.2.6","@advance-labs/config":"0.1.0"},"publishConfig":{"access":"public"},"repository":{"type":"git","url":"git+https://github.com/Advance-Labs/aeo-toolkit.git","directory":"packages/crawler"},"homepage":"https://docs.advancelabs.dev/aeo-toolkit","scripts":{"build":"tsup","dev":"tsup --watch","typecheck":"tsc --noEmit","lint":"eslint src","test":"vitest run","clean":"rimraf dist .turbo"},"_id":"@advance-labs/crawler@0.2.2","bugs":{"url":"https://github.com/Advance-Labs/aeo-toolkit/issues"},"_integrity":"sha512-7SAdbaoluBh1q4rPM5G27FubAdaBuUpuMzZM/Q9hpO/8yno/PXViVUGTp2XFQDWi+2/fchs460UjjAkMDWRJLA==","_resolved":"/tmp/4cec8e19dba5124e684ee504f73fec55/advance-labs-crawler-0.2.2.tgz","_from":"file:advance-labs-crawler-0.2.2.tgz","_nodeVersion":"24.20.0","_npmVersion":"11.19.0","dist":{"integrity":"sha512-7SAdbaoluBh1q4rPM5G27FubAdaBuUpuMzZM/Q9hpO/8yno/PXViVUGTp2XFQDWi+2/fchs460UjjAkMDWRJLA==","shasum":"4d9fa8c8bc1947ceec262fd29d4ceb182f51d040","tarball":"https://registry.npmjs.org/@advance-labs/crawler/-/crawler-0.2.2.tgz","fileCount":6,"unpackedSize":106311,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDAvzCWyCb/Alk7r99qsjx+RvnheV/a7eK8cuJEmOSWfwIgG0FqoCl8rK6+zdd0dOwckhKVBaSYUEFHJ53Ik1DJk48="}]},"_npmUser":{"name":"zordhalo","email":"lucas@advancelabs.dev"},"directories":{},"maintainers":[{"name":"zordhalo","email":"lucas@advancelabs.dev"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/crawler_0.2.2_1789020711108_0.933322180476607"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-30T03:13:44.837Z","modified":"2026-09-10T06:11:51.445Z","0.2.0":"2026-08-30T03:13:45.152Z","0.2.1":"2026-09-10T05:50:25.489Z","0.2.2":"2026-09-10T06:11:51.237Z"},"bugs":{"url":"https://github.com/Advance-Labs/aeo-toolkit/issues"},"license":"Apache-2.0","homepage":"https://docs.advancelabs.dev/aeo-toolkit","keywords":["crawler","web-crawler","robots-txt","sitemap","scraping","bfs","seo"],"repository":{"type":"git","url":"git+https://github.com/Advance-Labs/aeo-toolkit.git","directory":"packages/crawler"},"description":"Polite, bounded HTTP crawler — sitemap-first discovery, link-following BFS, robots.txt, and site-file detection.","maintainers":[{"name":"zordhalo","email":"lucas@advancelabs.dev"}],"readme":"# @advance-labs/crawler\n\nA polite, bounded HTTP crawler for the AEO Toolkit. It discovers a site's pages sitemap-first,\nthen follows on-site links breadth-first up to a hard page cap. It is robots.txt-aware, rate-limits\nper host, captures redirect chains, and detects the well-known crawl-hint / trust files\n(`robots.txt`, `sitemap.xml`, `llms.txt`, `llms-full.txt`, `favicon.ico`). All network I/O is routed\nthrough an injectable `Fetcher` (default: the global `fetch`), so callers and tests never need a\nlive network.\n\n## Usage\n\n```ts\nimport { crawl, fetchResource, parseRobotsTxt, parseSitemap, AI_BOT_NAMES } from '@advance-labs/crawler';\n\n// Crawl up to 50 pages, respecting robots.txt and spacing requests 500ms per host.\nconst result = await crawl('https://example.com/', {\n  maxPages: 50,\n  respectRobotsTxt: true,\n  perHostRateLimitMs: 500,\n  concurrency: 4,\n});\n\nconsole.log(result.pageCount, 'pages');\nconsole.log(result.robots.aiBotDirectives); // which AI crawlers are allowed at the root\nconsole.log(result.filePresence);           // { robotsTxt, sitemapXml, llmsTxt, llmsFullTxt, favicon }\n\n// Single resource fetch with redirect-chain capture.\nconst page = await fetchResource('https://example.com/');\n\n// Inject a fake fetcher for tests — no real network.\nawait crawl('https://example.com/', { maxPages: 10, fetcher: myFakeFetch });\n```\n\n## Public API\n\n| Export | Kind | Description |\n| --- | --- | --- |\n| `crawl(rootUrl, opts)` | `Promise<CrawlResult>` | Sitemap-first + BFS link-following crawl, capped at `opts.maxPages`. Honors robots.txt, per-host rate limit, and concurrency. |\n| `fetchResource(url, opts?)` | `Promise<PageResource>` | One fetch with status, headers, timing, final URL, content-type, body, and manual redirect-chain capture. |\n| `parseRobotsTxt(raw, url)` | `RobotsTxt` | Parses robots.txt; populates `sitemaps[]`, grouped directives, and an `aiBotDirectives[]` entry for every `AiBotName`. |\n| `emptyRobotsTxt(url)` | `RobotsTxt` | An all-allowed robots result for sites that have none. |\n| `parseSitemap(xml)` | `SitemapEntry[]` | Parses `<urlset>` and `<sitemapindex>` documents; tolerant of malformed input. |\n| `detectSiteFiles(rootUrl, opts?)` | `Promise<SiteFilePresence>` | HEAD/GET probes for robots, sitemap, llms.txt, llms-full.txt, favicon. |\n| `extractLinks(html, baseUrl)` | `Url[]` | Dependency-free regex href extraction, absolutized and de-duplicated. |\n| `sameOrigin`, `sameRegistrableSite`, `hostOf` | helpers | URL scope + host helpers used by the crawl frontier. |\n| `PerHostRateLimiter` | class | Per-host request spacing with an injectable clock/delay. |\n| `resolveFetcher`, `CrawlerError` | runtime | Fetcher resolution and the package's typed error. |\n| `Fetcher`, `FetchOptions`, `CrawlRuntimeOptions`, `FetchResourceOptions`, `DetectSiteFilesOptions` | types | The injectable I/O seam and option shapes. |\n| `AI_BOT_NAMES` | `readonly AiBotName[]` | The known AI/LLM crawler user-agents probed in robots.txt. |\n| `DEFAULT_*`, `MAX_REDIRECT_HOPS` | constants | Tunable defaults (user-agent, concurrency, timeout, depth, redirect cap). |\n\nAll domain shapes (`CrawlResult`, `PageResource`, `RobotsTxt`, `SitemapEntry`, `SiteFilePresence`,\n`AiBotName`, `CrawlOptions`, …) are imported from `@advance-labs/types`; this package never redefines them.\n\n## Status\n\n**Implemented.** No stubs. The single I/O seam is the injectable `Fetcher` (defaults to global\n`fetch` on Node 20+); pass a fake in tests to stay network-free. robots.txt directive parsing is\nprovided by `robots-parser`; sitemap parsing by `fast-xml-parser`.\n","readmeFilename":"README.md","author":{"name":"Lucas Krawczak","email":"lucas@advancelabs.dev","url":"https://advancelabs.dev/lucas"}}