{"_id":"@attribu/ai-crawl","_rev":"3-d45fe970cc8270463d49977948a27d37","name":"@attribu/ai-crawl","dist-tags":{"latest":"1.1.0"},"versions":{"1.0.0":{"name":"@attribu/ai-crawl","version":"1.0.0","keywords":["ai-crawl","bot-traffic","crawler-tracking","attribu","analytics","chatgpt","googlebot","claudebot","ai-agents","seo"],"author":{"name":"attribu.tech"},"license":"MIT","_id":"@attribu/ai-crawl@1.0.0","maintainers":[{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"}],"homepage":"https://attribu.tech/docs/bot-traffic","bugs":{"url":"https://github.com/attribu/ai-crawl/issues"},"dist":{"shasum":"b0736e348e4b7020f7916eece8258126384cb38d","tarball":"https://registry.npmjs.org/@attribu/ai-crawl/-/ai-crawl-1.0.0.tgz","fileCount":5,"integrity":"sha512-wWg/0MGYTBqum8kxHE93haFvMHCkn8wqHnxEvgHpNpIkx11bidkX1tBFesPoiiiJQVvqwyPBCMBMQPlyBmxSaA==","signatures":[{"sig":"MEUCIBliZko2Wl6UUrrvr+vWbKazR1wstW+UHCLuuoR4H3+XAiEA3SsFy22AtK7WIoeuhPFLulWoUywUZJCUMTgHDZmksVw=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":11244},"main":"dist/index.js","types":"dist/index.d.ts","module":"dist/index.mjs","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.js"}},"gitHead":"15c40611d9dd344e40b986e8333342f24e865e75","scripts":{"build":"tsup src/index.ts --format cjs,esm --dts --clean","prepublishOnly":"npm run build"},"_npmUser":{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"},"repository":{"url":"git+https://github.com/attribu/ai-crawl.git","type":"git"},"_npmVersion":"11.6.2","description":"Track AI crawlers, search bots, and training scrapers on your website. Works with Next.js, Cloudflare Workers, Express, Hono, and any server runtime.","directories":{},"_nodeVersion":"24.12.0","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","typescript":"^5.0.0"},"_npmOperationalInternal":{"tmp":"tmp/ai-crawl_1.0.0_1784151047285_0.8354624655532847","host":"s3://npm-registry-packages-npm-production"}},"1.0.1":{"name":"@attribu/ai-crawl","version":"1.0.1","keywords":["ai-crawl","bot-traffic","crawler-tracking","attribu","analytics","chatgpt","googlebot","claudebot","ai-agents","seo"],"author":{"name":"attribu.tech"},"license":"MIT","_id":"@attribu/ai-crawl@1.0.1","maintainers":[{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"}],"homepage":"https://attribu.tech/docs/bot-traffic","bugs":{"url":"https://github.com/attribu/ai-crawl/issues"},"dist":{"shasum":"40aef5210eb1111c52a720d0cfed04e6bc9f80ea","tarball":"https://registry.npmjs.org/@attribu/ai-crawl/-/ai-crawl-1.0.1.tgz","fileCount":6,"integrity":"sha512-59bWJSGd1/nyAZta7gYZH2pkLSGCRq8sVX18ew4lUBZW6qGAmAdfHQLJj3c05N2QPLPCmBP5hy1qBm9mjBPjcg==","signatures":[{"sig":"MEQCID1fqaiuXjszrIJh2EOOWC9L19hqWXkDLoFu8SogJX8TAiBUvTYIdeINkcTHgkTMxKisIyYWjz8gk/OXbZYoZ8uWAA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":15115},"main":"dist/index.js","types":"dist/index.d.ts","module":"dist/index.mjs","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.js"}},"gitHead":"15c40611d9dd344e40b986e8333342f24e865e75","scripts":{"build":"tsup src/index.ts --format cjs,esm --dts --clean","prepublishOnly":"npm run build"},"_npmUser":{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"},"repository":{"url":"git+https://github.com/attribu/ai-crawl.git","type":"git"},"_npmVersion":"11.6.2","description":"Track AI crawlers, search bots, and training scrapers on your website. Works with Next.js, Cloudflare Workers, Express, Hono, and any server runtime.","directories":{},"_nodeVersion":"24.12.0","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","typescript":"^5.0.0"},"_npmOperationalInternal":{"tmp":"tmp/ai-crawl_1.0.1_1784151231257_0.6452577561063264","host":"s3://npm-registry-packages-npm-production"}},"1.1.0":{"name":"@attribu/ai-crawl","version":"1.1.0","description":"Track AI crawlers, search bots, and training scrapers on your website. Works with Next.js, Cloudflare Workers, Express, Hono, and any server runtime.","main":"dist/index.js","module":"dist/index.mjs","types":"dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.js"}},"scripts":{"build":"tsup src/index.ts --format cjs,esm --dts --clean","prepublishOnly":"npm run build"},"keywords":["ai-crawl","bot-traffic","crawler-tracking","attribu","analytics","chatgpt","googlebot","claudebot","ai-agents","seo"],"author":{"name":"attribu.tech"},"license":"MIT","repository":{"type":"git","url":"git+https://github.com/attribu/ai-crawl.git"},"homepage":"https://attribu.tech/docs/bot-traffic","devDependencies":{"tsup":"^8.0.0","typescript":"^5.0.0"},"gitHead":"15c40611d9dd344e40b986e8333342f24e865e75","_id":"@attribu/ai-crawl@1.1.0","bugs":{"url":"https://github.com/attribu/ai-crawl/issues"},"_nodeVersion":"24.12.0","_npmVersion":"11.6.2","dist":{"integrity":"sha512-AgqB9kfaRS4GduiIKgEbgPKboZvKyBmOiF3juQvgmcXXVkXdvffWJsxY+M0FE1/2hYYhTyaX13jpvZ1Mo3GutA==","shasum":"5052fdcf354d486481ced16170f881cf82592786","tarball":"https://registry.npmjs.org/@attribu/ai-crawl/-/ai-crawl-1.1.0.tgz","fileCount":6,"unpackedSize":22747,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQCRiB+2pIBSOy/d3r5zu+bzmBDinWDinw7JzdjPydMWwAIhAPfEJVip26+uNsE1RZypYrd2ppnboyy07oMZp2vgE0aR"}]},"_npmUser":{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"},"directories":{},"maintainers":[{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/ai-crawl_1.1.0_1784152333448_0.20970075429780688"},"_hasShrinkwrap":false}},"time":{"created":"2026-07-15T21:30:47.071Z","modified":"2026-07-15T21:52:13.757Z","1.0.0":"2026-07-15T21:30:47.472Z","1.0.1":"2026-07-15T21:33:51.576Z","1.1.0":"2026-07-15T21:52:13.616Z"},"bugs":{"url":"https://github.com/attribu/ai-crawl/issues"},"author":{"name":"attribu.tech"},"license":"MIT","homepage":"https://attribu.tech/docs/bot-traffic","keywords":["ai-crawl","bot-traffic","crawler-tracking","attribu","analytics","chatgpt","googlebot","claudebot","ai-agents","seo"],"repository":{"type":"git","url":"git+https://github.com/attribu/ai-crawl.git"},"description":"Track AI crawlers, search bots, and training scrapers on your website. Works with Next.js, Cloudflare Workers, Express, Hono, and any server runtime.","maintainers":[{"name":"nikola0020","email":"nikolaperic27.np@gmail.com"}],"readme":"# @attribu/ai-crawl\n\nTrack AI crawlers, search bots, and training scrapers visiting your website. See which AI assistants cite your content, which search engines index your pages, and which crawlers collect your data for model training.\n\nWorks with **Next.js**, **Cloudflare Workers**, **Cloudflare Pages**, **Express**, **Hono**, and any server runtime.\n\n## Install\n\n```bash\nnpm install @attribu/ai-crawl\n```\n\n## Quick start\n\n### Next.js\n\n```ts\n// middleware.ts\nimport { NextRequest, NextResponse, NextFetchEvent } from \"next/server\";\nimport { trackAICrawlerRequest } from \"@attribu/ai-crawl\";\n\nexport function middleware(request: NextRequest, event: NextFetchEvent) {\n  trackAICrawlerRequest(request, event, {\n    websiteId: \"YOUR_SITE_ID\",\n  });\n\n  return NextResponse.next();\n}\n\nexport const config = {\n  matcher: [\"/((?!api|_next/static|_next/image|favicon.ico).*)\"],\n};\n```\n\n### Cloudflare Workers\n\n```ts\nimport { withAICrawlerTracking } from \"@attribu/ai-crawl\";\n\nexport default {\n  fetch: withAICrawlerTracking(\n    async (request, env, ctx) => {\n      return fetch(request);\n    },\n    {\n      websiteId: \"YOUR_SITE_ID\",\n    },\n  ),\n};\n```\n\n### Cloudflare Pages\n\n```ts\n// functions/_middleware.ts\nimport { trackAICrawlerRequest } from \"@attribu/ai-crawl\";\n\nexport async function onRequest(context) {\n  trackAICrawlerRequest(context.request, context, {\n    websiteId: \"YOUR_SITE_ID\",\n  });\n\n  return context.next();\n}\n```\n\n### Express\n\n```js\nimport { createExpressAICrawlerMiddleware } from \"@attribu/ai-crawl\";\n\napp.use(\n  createExpressAICrawlerMiddleware({\n    websiteId: \"YOUR_SITE_ID\",\n  }),\n);\n```\n\n### Hono\n\n```ts\nimport { trackAICrawlerResponse } from \"@attribu/ai-crawl\";\n\napp.use(\"*\", async (c, next) => {\n  await next();\n\n  trackAICrawlerResponse(c.req.raw, c.res, c.executionCtx, {\n    websiteId: \"YOUR_SITE_ID\",\n  });\n});\n```\n\n## How it works\n\nThe package runs in your server middleware. For each request it:\n\n1. Checks the `User-Agent` against 40+ known bot patterns\n2. Skips static assets (`.js`, `.css`, images, fonts)\n3. Skips framework internals (`/_next/`, `/_vercel/`)\n4. Tracks crawler-facing files (`robots.txt`, `llms.txt`, `sitemap.xml`)\n5. Sends a lightweight event to attribu.tech in the background\n\nFull classification (provider, category, IP verification) happens server-side at attribu.tech. Crawler lists stay up to date without requiring package upgrades.\n\n> **Do not await `trackAICrawlerRequest`.** Call it, then return your response. The package uses `waitUntil` to send the event in the background without slowing down your site.\n\n## What gets tracked\n\n| Category | Examples | Why it matters |\n|---|---|---|\n| **AI answers** | ChatGPT, Perplexity, You.com, Phind, Cohere | Fetch your pages live to answer user questions |\n| **Indexing** | Googlebot, Bingbot, DuckDuckBot, Brave, Kagi | Search engines discovering and refreshing your content |\n| **Training** | GPTBot, ClaudeBot, Bytespider, CCBot, Applebot | Crawlers collecting pages for model training or datasets |\n\n## API\n\n### `trackAICrawlerRequest(request, context, options)`\n\nTrack a bot request from middleware before the response exists.\n\n| Param | Type | Description |\n|---|---|---|\n| `request` | `Request` / `NextRequest` / Express `req` | The incoming request object |\n| `context` | `{ waitUntil }` / `null` | Runtime context (NextFetchEvent, ExecutionContext, etc.) |\n| `options` | `TrackOptions` | See options table below |\n\n### `trackAICrawlerResponse(request, response, context, options)`\n\nTrack a bot request with response status code. Use when you have access to the final response.\n\n| Param | Type | Description |\n|---|---|---|\n| `request` | `Request` / `NextRequest` / Express `req` | The incoming request object |\n| `response` | `Response` / Express `res` | The response object (reads `.status` for status code) |\n| `context` | `{ waitUntil }` / `null` | Runtime context |\n| `options` | `TrackOptions` | See options table below |\n\n### `withAICrawlerTracking(handler, options)`\n\nWrap a Cloudflare Worker fetch handler. Captures status code from the response.\n\n```ts\nexport default {\n  fetch: withAICrawlerTracking(handler, { websiteId: \"YOUR_SITE_ID\" }),\n};\n```\n\n### `createExpressAICrawlerMiddleware(options)`\n\nExpress/Connect middleware. Calls `next()` immediately and sends the event after the response finishes via a `finish` listener.\n\n```ts\napp.use(createExpressAICrawlerMiddleware({ websiteId: \"YOUR_SITE_ID\" }));\n```\n\n### Options (`TrackOptions`)\n\n| Option | Type | Default | Description |\n|---|---|---|---|\n| `websiteId` | `string` | Required | Your attribu.tech Website ID |\n| `authToken` | `string` | - | Bot traffic authentication token |\n| `publicOrigin` | `string` | - | Override the public origin for Docker/Cloud Run setups |\n| `disableAnswerFetch` | `boolean` | `false` | Skip AI answer crawlers |\n| `disableSearchCrawlers` | `boolean` | `false` | Skip search indexing crawlers |\n| `disableTrainingCrawlers` | `boolean` | `false` | Skip training crawlers |\n| `disableOtherCrawlers` | `boolean` | `false` | Skip uncategorized crawlers |\n\n## Direct HTTP\n\nIf you can't use the npm package, POST directly:\n\n```bash\ncurl -X POST https://attribu.tech/api/bot-traffic \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"websiteId\": \"YOUR_SITE_ID\",\n    \"domain\": \"yourdomain.com\",\n    \"href\": \"https://yourdomain.com/pricing\",\n    \"ai\": {\n      \"userAgent\": \"Mozilla/5.0 ... ChatGPT-User/1.0\",\n      \"ip\": \"203.0.113.10\",\n      \"statusCode\": 200,\n      \"source\": \"server_middleware\"\n    }\n  }'\n```\n\n## Documentation\n\nFull docs at [attribu.tech/docs/bot-traffic](https://attribu.tech/docs/bot-traffic)\n\n## License\n\nMIT\n","readmeFilename":"README.md"}