{"_id":"@trybyte/robotstxt-parser","_rev":"4-7c76622651df5d63a0122df6818909ae","name":"@trybyte/robotstxt-parser","dist-tags":{"latest":"2.0.0"},"versions":{"1.0.0":{"name":"@trybyte/robotstxt-parser","version":"1.0.0","keywords":["robots.txt","robots","parser","crawler","seo","geo","google","rfc9309"],"author":{"name":"Alireza Esmikhani"},"license":"Apache-2.0","_id":"@trybyte/robotstxt-parser@1.0.0","maintainers":[{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"}],"homepage":"https://github.com/trybyte-app/robotstxt-ts-port#readme","bugs":{"url":"https://github.com/trybyte-app/robotstxt-ts-port/issues"},"dist":{"shasum":"b858a214fb72e6b4703f6490b95ba9fafbee9750","tarball":"https://registry.npmjs.org/@trybyte/robotstxt-parser/-/robotstxt-parser-1.0.0.tgz","fileCount":46,"integrity":"sha512-Mz0dajn3SM4JLwaeBwG/zA1D/6+3ROnolR34PHzGQP9SQep9Y16lezythI3L/rkyaQE9ZWYZ6a/P19Y/207qTA==","signatures":[{"sig":"MEQCIDMCo32Mo1epK5MyCPAp1t0gAXNjXs1qJT6FLgvuIzPMAiB1T6d37ZQmuvBkvd88+hI8A1YoUHCPWn6SovshHT/NPQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":133207},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"2d98906bb03320c733abe1863d7e224dff0cdb38","scripts":{"test":"bun test","build":"tsc","prepublishOnly":"bun run build"},"_npmUser":{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"},"repository":{"url":"git+https://github.com/trybyte-app/robotstxt-ts-port.git","type":"git"},"_npmVersion":"11.4.2","description":"Google's robots.txt parser ported to TypeScript - RFC 9309 compliant","directories":{"test":"tests"},"_nodeVersion":"22.17.0","_hasShrinkwrap":false,"devDependencies":{"@types/bun":"latest","typescript":"^5.0.0"},"_npmOperationalInternal":{"tmp":"tmp/robotstxt-parser_1.0.0_1764768331581_0.31807867076279983","host":"s3://npm-registry-packages-npm-production"}},"1.1.0":{"name":"@trybyte/robotstxt-parser","version":"1.1.0","keywords":["robots.txt","robots","parser","crawler","seo","geo","google","rfc9309"],"author":{"url":"trybyte.app","name":"Byte Team"},"license":"Apache-2.0","_id":"@trybyte/robotstxt-parser@1.1.0","maintainers":[{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"}],"homepage":"https://github.com/trybyte-app/robotstxt-ts-port#readme","bugs":{"url":"https://github.com/trybyte-app/robotstxt-ts-port/issues"},"dist":{"shasum":"8e52f378d063b82d22c4186753c9a1214efcdaca","tarball":"https://registry.npmjs.org/@trybyte/robotstxt-parser/-/robotstxt-parser-1.1.0.tgz","fileCount":47,"integrity":"sha512-bWqvYsj4tGNtxXtorQjpO9wsXxKLp7SZ1aYO2ed0Ws9LBBHjj2bGiu5rz0WlPYU34dYXYBro6FvOAa9vKnUv/w==","signatures":[{"sig":"MEYCIQDw6VB5VqXR/XDVO2aHA+bzHcl6TyV0ofqCw6DM+Ca4gwIhAM/z9XdWSPpG3pfGGULx8m++Xvi/2IRSvmjcFfM4cjY5","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":148874},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"e14a209ac938b847129258c51b3bdae7015cf746","scripts":{"test":"bun test","build":"tsc","format":"prettier \"**/*.{js,jsx,mjs,ts,tsx,json,jsonc}\" --write","prepublishOnly":"bun run build"},"_npmUser":{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"},"repository":{"url":"git+https://github.com/trybyte-app/robotstxt-ts-port.git","type":"git"},"_npmVersion":"11.4.2","description":"Google's robots.txt parser ported to TypeScript - RFC 9309 compliant","directories":{"test":"tests"},"_nodeVersion":"22.17.0","_hasShrinkwrap":false,"devDependencies":{"prettier":"^3.7.4","@types/bun":"latest","typescript":"^5.0.0"},"_npmOperationalInternal":{"tmp":"tmp/robotstxt-parser_1.1.0_1765105018117_0.9889942021141898","host":"s3://npm-registry-packages-npm-production"}},"1.2.0":{"name":"@trybyte/robotstxt-parser","version":"1.2.0","keywords":["robots.txt","robots","parser","crawler","seo","geo","google","rfc9309"],"author":{"url":"trybyte.app","name":"Byte Team"},"license":"Apache-2.0","_id":"@trybyte/robotstxt-parser@1.2.0","maintainers":[{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"}],"homepage":"https://github.com/trybyte-app/robotstxt-ts-port#readme","bugs":{"url":"https://github.com/trybyte-app/robotstxt-ts-port/issues"},"dist":{"shasum":"3ed596cc72a10b1c819a9261d7d49ac6f9f7cd06","tarball":"https://registry.npmjs.org/@trybyte/robotstxt-parser/-/robotstxt-parser-1.2.0.tgz","fileCount":47,"integrity":"sha512-AywmLqMu0ljMLEPVb4oPxwnyFmm3xxOS6LmwRia5jgqvzrBAoXXR0JWXEXaq/IG3tlFSXuAp+XzAabrijYzyUQ==","signatures":[{"sig":"MEUCIQC7EiFEDOZXcu/rmiz87L02laTO3/z13NQoZ3NXOuJ5ngIgcF3w+qLE2BLmtiUrzEQh/M4LjFtqRNcb2fluwxCnqMg=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":149546},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"3005b572337a9186c7e7e2952678b9cd4bfaf349","scripts":{"test":"bun test","build":"tsc","format":"prettier \"**/*.{js,jsx,mjs,ts,tsx,json,jsonc}\" --write","prepublishOnly":"bun run build"},"_npmUser":{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"},"repository":{"url":"git+https://github.com/trybyte-app/robotstxt-ts-port.git","type":"git"},"_npmVersion":"11.4.2","description":"Google's robots.txt parser ported to TypeScript - RFC 9309 compliant","directories":{"test":"tests"},"_nodeVersion":"22.17.0","_hasShrinkwrap":false,"devDependencies":{"prettier":"^3.7.4","@types/bun":"latest","typescript":"^5.0.0"},"_npmOperationalInternal":{"tmp":"tmp/robotstxt-parser_1.2.0_1765192523993_0.5079295602844889","host":"s3://npm-registry-packages-npm-production"}},"2.0.0":{"name":"@trybyte/robotstxt-parser","version":"2.0.0","description":"Compile robots.txt rules for Google-compatible or strict RFC 9309 matching","keywords":["robots.txt","robots","parser","crawler","seo","geo","google","rfc9309"],"homepage":"https://github.com/trybyte-app/robotstxt-ts-port#readme","bugs":{"url":"https://github.com/trybyte-app/robotstxt-ts-port/issues"},"repository":{"type":"git","url":"git+https://github.com/trybyte-app/robotstxt-ts-port.git"},"license":"Apache-2.0","author":{"name":"Byte Team","url":"trybyte.app"},"type":"module","sideEffects":false,"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"main":"./dist/index.js","types":"./dist/index.d.ts","directories":{"test":"tests"},"scripts":{"build":"tsc","test":"bun test","prepublishOnly":"bun run build","format":"prettier \"**/*.{js,jsx,mjs,ts,tsx,json,jsonc}\" --write"},"devDependencies":{"@types/bun":"latest","prettier":"^3.7.4","typescript":"^5.0.0"},"engines":{"node":">=22.0.0"},"gitHead":"54378223490eae43fb3133bed7eb5bd0ebf5425a","_id":"@trybyte/robotstxt-parser@2.0.0","_nodeVersion":"24.19.0","_npmVersion":"11.17.0","dist":{"integrity":"sha512-kqoL5DCRQeTkAbG8xsPGP8/5lEFUtFj/t3hOLvIB7kkFZTIjRXH8rm6blH3FjXEJ44zjzMFYXZMamuTZPCwuMw==","shasum":"45bc8042aad8a982fb6816c741355cde66802ae4","tarball":"https://registry.npmjs.org/@trybyte/robotstxt-parser/-/robotstxt-parser-2.0.0.tgz","fileCount":23,"unpackedSize":117309,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDsKXQpBph5OXmc6CYzmlFDyK1t21NqPrITiqu2JpjcRQIgUP4wnBcJWzO54ghW0WHw+Har1FdoZY+px/jpV9jTkWU="}]},"_npmUser":{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"},"maintainers":[{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/robotstxt-parser_2.0.0_1788434132225_0.4385928417004257"},"_hasShrinkwrap":false}},"time":{"created":"2025-12-03T13:25:31.458Z","modified":"2026-09-03T11:15:32.574Z","1.0.0":"2025-12-03T13:25:31.754Z","1.1.0":"2025-12-07T10:56:58.282Z","1.2.0":"2025-12-08T11:15:24.139Z","2.0.0":"2026-09-03T11:15:32.363Z"},"bugs":{"url":"https://github.com/trybyte-app/robotstxt-ts-port/issues"},"author":{"name":"Byte Team","url":"trybyte.app"},"license":"Apache-2.0","homepage":"https://github.com/trybyte-app/robotstxt-ts-port#readme","keywords":["robots.txt","robots","parser","crawler","seo","geo","google","rfc9309"],"repository":{"type":"git","url":"git+https://github.com/trybyte-app/robotstxt-ts-port.git"},"description":"Compile robots.txt rules for Google-compatible or strict RFC 9309 matching","maintainers":[{"name":"mresmikhani","email":"mresmikhani.work@gmail.com"}],"readme":"# @trybyte/robotstxt-parser\n\n`@trybyte/robotstxt-parser` is a dependency-free TypeScript parser and matcher for robots.txt files.\n\nParse the file once with `compileRobotsText()`. Bind one crawler identity with `forCrawler()`. The resulting matcher can check one URL or consume a lazy iterable of URLs.\n\nCompatibility with Google's [open-source parser and matcher](https://github.com/google/robotstxt) is the default. Callers can choose strict [RFC 9309](https://www.rfc-editor.org/rfc/rfc9309.html) behavior when they need the standard instead of Google's parser rules.\n\n## Install\n\nUse one package manager.\n\n| Package manager | Command                                 |\n| --------------- | --------------------------------------- |\n| npm             | `npm install @trybyte/robotstxt-parser` |\n| pnpm            | `pnpm add @trybyte/robotstxt-parser`    |\n| Bun             | `bun add @trybyte/robotstxt-parser`     |\n\nThe package supports Node.js 22 LTS and newer.\n\n## Match URLs\n\n```ts\nimport { compileRobotsText } from \"@trybyte/robotstxt-parser\";\n\nconst robotsText = `\nUser-agent: *\nDisallow: /private/\n\nUser-agent: Googlebot\nAllow: /\n`;\n\nconst robots = compileRobotsText(robotsText);\nconst googlebot = robots.forCrawler(\"Googlebot/2.1\");\n\ngooglebot.isAllowed(\"https://example.com/private/report.pdf\"); // true\n\nconst decision = googlebot.match(\"https://example.com/private/report.pdf\");\n\nconsole.log(decision);\n// {\n//   url: \"https://example.com/private/report.pdf\",\n//   path: \"/private/report.pdf\",\n//   allowed: true,\n//   selectedRules: \"specific\",\n//   evidence: {\n//     kind: \"rule\",\n//     directive: \"allow\",\n//     pattern: \"/\",\n//     lineNumber: 6\n//   }\n// }\n```\n\n`forCrawler()` accepts an RFC product token such as `Googlebot` or a `Product/Version` value such as `Googlebot/2.1`. It stores the lowercase product token and matches group names without regard to case. An invalid value throws `InvalidCrawlerIdentityError`.\n\n## Choose a matching policy\n\n```ts\nconst google = compileRobotsText(robotsText);\nconst strict = compileRobotsText(robotsText, { policy: \"rfc9309\" });\n```\n\n| Policy    | What it does                                                                                                                                                                                                       |\n| --------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |\n| `google`  | This is the default. It follows Google's open-source parser and matcher, including accepted field-name misspellings, field-name prefix matching, the Google line-length limit, and `index.htm` directory matching. |\n| `rfc9309` | It requires exact directive syntax, compares percent-encoded paths by RFC rules, rejects invalid product tokens and path patterns, and always allows `/robots.txt`.                                                |\n\nBoth policies use the same public types, crawler identity normalization, URL path extraction, and lazy bulk methods. Each policy keeps its own parsing and matching behavior where the two specifications disagree.\n\nGoogle mode covers decisions made by the open-source parser and matcher. It does not fetch robots.txt files or model HTTP status codes, redirects, caches, download limits, or crawl schedules. Supply URI-encoded URLs because that is what Google's matcher expects.\n\n## Process large URL sets\n\n`isAllowedMany()` and `matchMany()` accept any `Iterable<string>`. Both return lazy, single-pass iterators. The package does not collect the input or output.\n\n```ts\nconst crawler = compileRobotsText(robotsText).forCrawler(\"Googlebot\");\n\nfunction* urls() {\n\tfor (let index = 0; index < 1_000_000; index++) {\n\t\tyield `https://example.com/page/${index}`;\n\t}\n}\n\nfor (const allowed of crawler.isAllowedMany(urls())) {\n\t// Store, count, or stream this result before requesting the next one.\n}\n```\n\nUse `isAllowedMany()` for booleans. Use `matchMany()` when the caller also needs the chosen rule set and match evidence.\n\nCalling `Array.from()` stores every result. Iterate over the return value directly when memory use matters.\n\n## Inspect a robots file\n\nInspection is separate from compilation. Normal matching therefore does not keep source-line metadata that it never reads.\n\n```ts\nimport { inspectRobotsText } from \"@trybyte/robotstxt-parser\";\n\nconst report = inspectRobotsText(robotsText, { policy: \"google\" });\n\nconsole.log(report.lineCount);\nconsole.log(report.recognizedDirectiveCount);\nconsole.log(report.lines);\n```\n\nThe report is structured data, not a formatted text summary.\n\n| Field                       | Meaning                                                                                                         |\n| --------------------------- | --------------------------------------------------------------------------------------------------------------- |\n| `policy`                    | The policy used to interpret the source.                                                                        |\n| `lineCount`                 | The number of source lines scanned.                                                                             |\n| `recognizedDirectiveCount`  | The number of `user-agent`, `allow`, `disallow`, and `sitemap` directives recognized under the selected policy. |\n| `unsupportedDirectiveCount` | The number of known but unsupported directives.                                                                 |\n| `unknownDirectiveCount`     | The number of other named directives.                                                                           |\n| `lines`                     | One `ParsedLine` entry for each source line.                                                                    |\n\nEach `ParsedLine` contains its one-based `lineNumber`, an interpreted `directive` or `null`, and zero or more diagnostic codes. A directive records its normalized name and value plus `effective`, which tells you whether it can affect matching. Diagnostic codes include `missing-colon`, `acceptable-typo`, `line-too-long`, `comment`, `whole-line-comment`, and `empty`.\n\n## Read match evidence\n\n`match()` returns the checked URL, the extracted path, the boolean decision, the selected rule set, and evidence for that decision.\n\nWhen a rule wins, `evidence.kind` is `rule`. The evidence names the `allow` or `disallow` directive, preserves its source pattern, and gives its one-based source line number.\n\nWhen no rule decides the result, `evidence.kind` is `default-allow`. Its `reason` is one of these values.\n\n| Reason        | Meaning                                             |\n| ------------- | --------------------------------------------------- |\n| `no-group`    | No specific or global user-agent group matched.     |\n| `empty-group` | A matching group exists but has no effective rules. |\n| `no-match`    | Rules were selected, but none matched the URL path. |\n| `robots-txt`  | RFC 9309 grants access to `/robots.txt`.            |\n\n`selectedRules` is `specific`, `global`, or `none`. A matching specific group suppresses global rules even when the specific group contains no effective rules.\n\n## Public API\n\n```ts\ncompileRobotsText(source, options?): CompiledRobotsText\ninspectRobotsText(source, options?): RobotsReport\n```\n\n`CompiledRobotsText` has this interface.\n\n```ts\nreadonly policy: \"google\" | \"rfc9309\";\nforCrawler(identity: string): CrawlerRules;\n```\n\n`CrawlerRules` has this interface.\n\n```ts\nreadonly identity: string;\nisAllowed(url: string): boolean;\nmatch(url: string): MatchDecision;\nisAllowedMany(urls: Iterable<string>): IterableIterator<boolean>;\nmatchMany(urls: Iterable<string>): IterableIterator<MatchDecision>;\n```\n\n## Migrate from version 1\n\nVersion 2 replaces the version 1 class and callback API. It has no compatibility aliases. An automated migration should remove old imports instead of wrapping or recreating them.\n\nUse this checklist as the migration contract.\n\n1. Require Node.js 22 or newer in the consuming project.\n2. Replace every version 1 package import with the version 2 exports listed below.\n3. Call `compileRobotsText()` once for each robots text and policy.\n4. Call `forCrawler()` once for each crawler identity used with that compiled file.\n5. Replace scalar checks with `isAllowed()` or `match()`.\n6. Replace array-based bulk checks with the lazy `isAllowedMany()` or `matchMany()` iterator.\n7. Replace parser callbacks and `RobotsParsingReporter` with `inspectRobotsText()`.\n8. Remove code that depends on deleted matcher state, internal utilities, constants, or custom matching strategies.\n9. Run the consuming project's type checker and robots matching tests. Review the behavior changes at the end of this section before accepting new results.\n\n### Replace imports\n\nThese version 1 exports no longer exist.\n\n```ts\nRobotsMatcher;\nParsedRobots;\nRobotsParsingReporter;\nRobotsParseHandler;\nRobotsMatchStrategy;\nLongestMatchRobotsMatchStrategy;\nUrlCheckResult;\nParsedRule;\nLineMetadata;\nRobotsParsedLine;\nparseRobotsTxt;\nmatches;\ngetPathParamsQuery;\nmaybeEscapePattern;\nKeyType;\nRobotsTagName;\ncreateLineMetadata;\ncreateRobotsParsedLine;\nK_MAX_LINE_LEN;\nK_ALLOW_FREQUENT_TYPOS;\nK_UNSUPPORTED_TAGS;\n```\n\nImport only the version 2 operations and types that the application uses.\n\n```ts\nimport {\n\tcompileRobotsText,\n\tinspectRobotsText,\n\tInvalidCrawlerIdentityError,\n\ttype CrawlerRules,\n\ttype MatchDecision,\n\ttype RobotsReport,\n} from \"@trybyte/robotstxt-parser\";\n```\n\n### Replace matching calls\n\n| Version 1                                                       | Version 2                                                                                                                      |\n| --------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `new RobotsMatcher().oneAgentAllowedByRobots(text, agent, url)` | `compileRobotsText(text).forCrawler(agent).isAllowed(url)`                                                                     |\n| `matcher.allowedByRobots(text, agents, url)`                    | Compile once, then call `agents.some()` with one crawler matcher per identity.                                                 |\n| `RobotsMatcher.parse(text)`                                     | `compileRobotsText(text)`                                                                                                      |\n| `ParsedRobots.parse(text)`                                      | `compileRobotsText(text)`                                                                                                      |\n| `RobotsMatcher.batchCheck(text, agent, urls)`                   | `compileRobotsText(text).forCrawler(agent).matchMany(urls)`                                                                    |\n| `parsed.checkUrl(agent, url)`                                   | `compiled.forCrawler(agent).match(url)`                                                                                        |\n| `parsed.checkUrls(agent, urls)`                                 | `compiled.forCrawler(agent).matchMany(urls)`                                                                                   |\n| `matcher.disallow()`                                            | Negate `crawler.isAllowed(url)`, or read `decision.allowed` from `crawler.match(url)`.                                         |\n| `matcher.everSeenSpecificAgent()`                               | Read `decision.selectedRules === \"specific\"`. Rule selection is fixed for a bound crawler, so this value does not vary by URL. |\n| `matcher.matchingLine()`                                        | When `decision.evidence.kind === \"rule\"`, read `decision.evidence.lineNumber`.                                                 |\n| `RobotsMatcher.isValidUserAgentToObey(agent)`                   | Call `compiled.forCrawler(agent)` and catch `InvalidCrawlerIdentityError`.                                                     |\n\nThe direct replacement for a single crawler looks like this.\n\n```ts\n// Version 1\nconst allowed = new RobotsMatcher().oneAgentAllowedByRobots(\n\trobotsText,\n\t\"Googlebot\",\n\turl,\n);\n\n// Version 2\nconst allowed = compileRobotsText(robotsText)\n\t.forCrawler(\"Googlebot\")\n\t.isAllowed(url);\n```\n\nVersion 1 `allowedByRobots()` returned `true` when any supplied crawler identity was allowed. Preserve that behavior explicitly.\n\n```ts\nconst compiled = compileRobotsText(robotsText);\nconst allowed = crawlerIdentities.some((identity) =>\n\tcompiled.forCrawler(identity).isAllowed(url),\n);\n```\n\nCompile and bind outside repeated URL loops.\n\n```ts\nconst compiled = compileRobotsText(robotsText);\nconst crawler = compiled.forCrawler(\"Googlebot\");\n\nfor (const decision of crawler.matchMany(urls)) {\n\tconsume(decision);\n}\n```\n\n`matchMany()` returns a lazy iterator. Version 1 `checkUrls()` returned an array. If the caller still needs an array, collect it at the application boundary.\n\n```ts\nconst decisions = Array.from(crawler.matchMany(urls));\n```\n\n### Replace bulk result fields\n\nVersion 1 returned `UrlCheckResult`. Version 2 returns `MatchDecision` from `match()` and `matchMany()`.\n\n| Version 1 field   | Version 2 field                                      |\n| ----------------- | ---------------------------------------------------- |\n| `url`             | `url`                                                |\n| `allowed`         | `allowed`                                            |\n| `matchingLine`    | `evidence.lineNumber` when `evidence.kind` is `rule` |\n| `matchedPattern`  | `evidence.pattern` when `evidence.kind` is `rule`    |\n| `matchedRuleType` | `evidence.directive` when `evidence.kind` is `rule`  |\n\nVersion 2 also returns the extracted `path`, the `selectedRules` scope, and a typed default-allow reason when no rule wins. Do not invent a line number or empty pattern for a default allow. Branch on `evidence.kind`.\n\n```ts\nconst decision = crawler.match(url);\n\nif (decision.evidence.kind === \"rule\") {\n\tconsole.log(\n\t\tdecision.evidence.directive,\n\t\tdecision.evidence.pattern,\n\t\tdecision.evidence.lineNumber,\n\t);\n} else {\n\tconsole.log(decision.evidence.reason);\n}\n```\n\n### Replace parsing reports\n\nDelete custom `RobotsParseHandler` subclasses when they only collect line metadata or directive counts. Call `inspectRobotsText()` instead.\n\n```ts\nconst report = inspectRobotsText(robotsText, { policy: \"google\" });\n```\n\n| Version 1 reporter value   | Version 2 report value                                   |\n| -------------------------- | -------------------------------------------------------- |\n| `lastLineSeen()`           | `lineCount`                                              |\n| `validDirectives()`        | `recognizedDirectiveCount`                               |\n| `unusedDirectives()`       | `unsupportedDirectiveCount + unknownDirectiveCount`      |\n| `parseResults()`           | `lines`                                                  |\n| `RobotsParsedLine.lineNum` | `ParsedLine.lineNumber`                                  |\n| `RobotsParsedLine.tagName` | `ParsedLine.directive?.name`                             |\n| `LineMetadata` booleans    | `ParsedLine.diagnostics` and `ParsedDirective.effective` |\n\nThe report uses string directive names and diagnostic codes instead of the `RobotsTagName` enum and `LineMetadata` booleans. Review any code that serializes the old report shape.\n\n### Handle APIs with no direct replacement\n\n- `disallowIgnoreGlobal()` has no version 2 equivalent. Version 2 always follows the selected policy's group rules. Redesign code that bypasses global groups.\n- `getExplicitAgents()` and `hasSpecificAgent()` are not compiled-document queries in version 2. Use `selectedRules` when the application only needs to know whether a bound crawler selected specific rules. Use `inspectRobotsText()` to build a line-level user-agent inventory.\n- Custom `RobotsMatchStrategy` implementations have no extension point. Choose `google` or `rfc9309` when compiling.\n- Parser callbacks, matching helpers, URL helpers, and parser constants are internal. Move unrelated utility behavior into application code. Use the public compiler and inspector for robots decisions.\n\n### Review behavior changes\n\n- Google matching remains the default. Pass `{ policy: \"rfc9309\" }` when the application requires strict RFC 9309 parsing and matching.\n- `forCrawler()` accepts a valid `Product` or `Product/Version` identity, normalizes it to a lowercase product token, and throws `InvalidCrawlerIdentityError` for malformed input. Version 1 did not normalize caller identities before matching. Review results for `Product/Version` values, and replace malformed values such as `Foo Bar` with an explicit product token.\n- A matching specific group suppresses global rules even when the specific group has no effective rules.\n- `match()` reports the original source pattern and its one-based line number. Default access reports `no-group`, `empty-group`, `no-match`, or `robots-txt` instead of a synthetic line zero.\n- Bulk methods preserve input order and consume the iterable once. They produce results only as the caller requests them.\n- Compilation does not retain inspection metadata. Call `inspectRobotsText()` separately when the application needs a report.\n\nAfter migration, search the consuming repository for every removed export named above. A clean search plus a passing type check is the minimum evidence that the version 1 API is gone.\n\n## Develop the package\n\n```bash\nbun install --frozen-lockfile\nbun test\nbun run build\n```\n\nThe Google regression suite adapts cases from [`google/robotstxt`](https://github.com/google/robotstxt). The RFC suite covers strict syntax, group selection, rule priority, percent encoding, and the required `/robots.txt` allowance. See [TESTS.md](./TESTS.md) for the pinned upstream revision and the full test map.\n\n## License\n\nApache-2.0\n","readmeFilename":"README.md"}