{"_id":"@codecai/maps-cli","_rev":"5-edfa59e99fe7b3414431e5d81ae7256b","name":"@codecai/maps-cli","dist-tags":{"latest":"0.5.0"},"versions":{"0.1.0":{"name":"@codecai/maps-cli","version":"0.1.0","keywords":["codec","tokenizer","llm","huggingface","dialect","schema","cli"],"license":"MIT","_id":"@codecai/maps-cli@0.1.0","maintainers":[{"name":"wdunn001","email":"wdunn001@gmail.com"}],"homepage":"https://github.com/wdunn001/codec-maps","bugs":{"url":"https://github.com/wdunn001/Codec/issues"},"bin":{"codecai-maps":"dist/cli.js"},"dist":{"shasum":"d76dc408c5633504ee15c63d6fd1a3e5737303f9","tarball":"https://registry.npmjs.org/@codecai/maps-cli/-/maps-cli-0.1.0.tgz","fileCount":14,"integrity":"sha512-QMNSMkYnUnJzDsegu/iwgLh1o5UpFw1QV8nNnleG9MNFmS00S2iyb2u4uKDY2erUbt9j5BB6hOtumqvY6G8krA==","signatures":[{"sig":"MEUCIQCz4kWoyphvrbQ+p6PVAEuFJA6+kPCQHODNQhySZA4I4QIgPo+4y1UTQJ1qstK+31hhNGC16H3f/IvYGrD9pZbd/fY=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":35182},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./convert":{"types":"./dist/convert.d.ts","import":"./dist/convert.js"}},"gitHead":"eb3dcc1e02a284ba0e8df69a924e0878fb41f2d2","scripts":{"test":"node --test --import tsx test/*.test.ts","build":"tsc -p tsconfig.json && node -e \"require('fs').chmodSync('dist/cli.js', 0o755)\""},"_npmUser":{"name":"wdunn001","email":"wdunn001@gmail.com"},"repository":{"url":"git+https://github.com/wdunn001/Codec.git","type":"git","directory":"packages/maps-cli"},"_npmVersion":"11.8.0","description":"Generate Codec tokenizer dialect maps from HuggingFace tokenizer.json files. The 'tsc --declaration' for LLM token vocabularies.","directories":{},"_nodeVersion":"25.5.0","dependencies":{"@codecai/web":"^0.2.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.2","typescript":"^5.7.3","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/maps-cli_0.1.0_1778052712983_0.7028600228937978","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@codecai/maps-cli","version":"0.1.1","keywords":["codec","tokenizer","llm","huggingface","dialect","schema","cli"],"license":"MIT","_id":"@codecai/maps-cli@0.1.1","maintainers":[{"name":"wdunn001","email":"wdunn001@gmail.com"}],"homepage":"https://github.com/wdunn001/codec-maps","bugs":{"url":"https://github.com/wdunn001/Codec/issues"},"bin":{"codecai-maps":"dist/cli.js"},"dist":{"shasum":"8a73f5aa73358ea76543dcf38b3f28cd4cf54dc5","tarball":"https://registry.npmjs.org/@codecai/maps-cli/-/maps-cli-0.1.1.tgz","fileCount":14,"integrity":"sha512-fJD3UnbqZplxvDunySq7mZkiH4152FgFmStYHGwqRNYaBxK3vKXAt3nXHJhHvp8TCWZjF+D9VWYOzFlzeLMu8g==","signatures":[{"sig":"MEUCIQDzF8sW7xEOh6je9f3uVb7ly3toHJ4RZHctDjKyHNiqRAIgKTRxqU/G+jMZcG4z+ojCJeuiVrh0Rqy9E95aMr7Mgtk=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":35182},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./convert":{"types":"./dist/convert.d.ts","import":"./dist/convert.js"}},"gitHead":"628b4f7d350aed7566df85efe2127e2f7d00a380","scripts":{"test":"node --test --import tsx test/*.test.ts","build":"tsc -p tsconfig.json && node -e \"require('fs').chmodSync('dist/cli.js', 0o755)\""},"_npmUser":{"name":"wdunn001","email":"wdunn001@gmail.com"},"repository":{"url":"git+https://github.com/wdunn001/Codec.git","type":"git","directory":"packages/maps-cli"},"_npmVersion":"11.8.0","description":"Generate Codec tokenizer dialect maps from HuggingFace tokenizer.json files. The 'tsc --declaration' for LLM token vocabularies.","directories":{},"_nodeVersion":"25.5.0","dependencies":{"@codecai/web":"^0.2.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.2","typescript":"^5.7.3","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/maps-cli_0.1.1_1778065711393_0.7549267641646151","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@codecai/maps-cli","version":"0.2.0","keywords":["codec","tokenizer","llm","huggingface","dialect","schema","cli"],"license":"MIT","_id":"@codecai/maps-cli@0.2.0","maintainers":[{"name":"wdunn001","email":"wdunn001@gmail.com"}],"homepage":"https://github.com/wdunn001/codec-maps","bugs":{"url":"https://github.com/wdunn001/Codec/issues"},"bin":{"codecai-maps":"dist/cli.js"},"dist":{"shasum":"857fef6832f9c70ff1a65cce284c79ab47c4ffb2","tarball":"https://registry.npmjs.org/@codecai/maps-cli/-/maps-cli-0.2.0.tgz","fileCount":14,"integrity":"sha512-NbbeegJdNy5yesHmVYHeOl3vHBlWksPesTwEFj91f7ZBOa82Fou6f5hFvEuTzl63qFix3PH5AVcv3J4vj6fpnw==","signatures":[{"sig":"MEUCIQDbb42UziFfrRVS3T8WjKEoMCU16LgkNzrkrTN2JuyfNgIgbDO+tja20brTvZu2OeWFqBZStRwxRiyJE9lEnL94v3g=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":45062},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./convert":{"types":"./dist/convert.d.ts","import":"./dist/convert.js"}},"gitHead":"429a31fa147a117acb237d55534d4b8842a4c898","scripts":{"test":"node --test --import tsx test/*.test.ts","build":"tsc -p tsconfig.json && node -e \"require('fs').chmodSync('dist/cli.js', 0o755)\""},"_npmUser":{"name":"wdunn001","email":"wdunn001@gmail.com"},"repository":{"url":"git+https://github.com/wdunn001/Codec.git","type":"git","directory":"packages/maps-cli"},"_npmVersion":"11.8.0","description":"Generate Codec tokenizer dialect maps from HuggingFace tokenizer.json files. The 'tsc --declaration' for LLM token vocabularies.","directories":{},"_nodeVersion":"25.5.0","dependencies":{"@codecai/web":"^0.3.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.2","typescript":"^5.7.3","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/maps-cli_0.2.0_1778071749134_0.7422591783364143","host":"s3://npm-registry-packages-npm-production"}},"0.4.1":{"name":"@codecai/maps-cli","version":"0.4.1","keywords":["codec","tokenizer","llm","huggingface","dialect","schema","cli"],"license":"MIT","_id":"@codecai/maps-cli@0.4.1","maintainers":[{"name":"wdunn001","email":"wdunn001@gmail.com"}],"homepage":"https://github.com/wdunn001/codec-maps","bugs":{"url":"https://github.com/wdunn001/Codec/issues"},"bin":{"codecai-maps":"dist/cli.js"},"dist":{"shasum":"cb06828172cbb6cfe3a4f56fb4ec8bc305595928","tarball":"https://registry.npmjs.org/@codecai/maps-cli/-/maps-cli-0.4.1.tgz","fileCount":22,"integrity":"sha512-P4XiaWXgIZO5xlD5M8tHATuwS3w3Qey/pg8uO+M5GgUguBzfTJnQQ8iKUJRS8jW5mYti6p1jQrZD08E9sWb4wQ==","signatures":[{"sig":"MEUCIE6M4l+Hz4D1gIrIhwQ1vEC8fwgG85VxKqRwbTZ44XVlAiEA/fYdQ36SIh6yEIBDooKhfcnYRxlMsnsR4iYcF3JISO0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":146319},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./convert":{"types":"./dist/convert.d.ts","import":"./dist/convert.js"}},"gitHead":"53a71503f33fca91d353d7dfbeddfcf050c2d37a","scripts":{"test":"node --test --import tsx test/*.test.ts","build":"tsc -p tsconfig.json && node -e \"require('fs').chmodSync('dist/cli.js', 0o755)\""},"_npmUser":{"name":"wdunn001","email":"wdunn001@gmail.com"},"repository":{"url":"git+https://github.com/wdunn001/Codec.git","type":"git","directory":"packages/maps-cli"},"_npmVersion":"10.8.2","description":"Generate Codec tokenizer dialect maps from HuggingFace tokenizer.json files. The 'tsc --declaration' for LLM token vocabularies.","directories":{},"_nodeVersion":"20.20.2","dependencies":{"@codecai/web":"^0.4.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.2","typescript":"^5.7.3","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/maps-cli_0.4.1_1778954333000_0.498549733253586","host":"s3://npm-registry-packages-npm-production"}},"0.5.0":{"name":"@codecai/maps-cli","version":"0.5.0","description":"Generate Codec tokenizer dialect maps from HuggingFace tokenizer.json files. The 'tsc --declaration' for LLM token vocabularies.","license":"MIT","type":"module","bin":{"codecai-maps":"dist/cli.js"},"main":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./convert":{"types":"./dist/convert.d.ts","import":"./dist/convert.js"}},"scripts":{"build":"tsc -p tsconfig.json && node -e \"require('fs').chmodSync('dist/cli.js', 0o755)\"","test":"node --test --import tsx test/*.test.ts"},"dependencies":{"@codecai/web":"^0.5.0"},"devDependencies":{"@types/node":"^20.0.0","tsx":"^4.19.2","typescript":"^5.7.3"},"engines":{"node":">=18"},"keywords":["codec","tokenizer","llm","huggingface","dialect","schema","cli"],"repository":{"type":"git","url":"git+https://github.com/wdunn001/Codec.git","directory":"packages/maps-cli"},"homepage":"https://github.com/wdunn001/codec-maps","_id":"@codecai/maps-cli@0.5.0","gitHead":"3acffb9c70bacc876e64ee42a63beab26cca5f26","bugs":{"url":"https://github.com/wdunn001/Codec/issues"},"_nodeVersion":"20.20.2","_npmVersion":"10.8.2","dist":{"integrity":"sha512-URAFLf8ZGYcILvsl6E2HODmj8OzGDMzOZ5X/I/VMI0WNOqHeV1wRI5PMN2hq3xtPiXp6dnfDrUA+U6CG4oofFA==","shasum":"0621c736fa9bb29bb364a225b57a89480d21e362","tarball":"https://registry.npmjs.org/@codecai/maps-cli/-/maps-cli-0.5.0.tgz","fileCount":22,"unpackedSize":157225,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIGokxpuo6Dk9f8ARxacDdTwIJP9J71pMPCQombw7MXktAiEAyotcDvYQx23yU/ku4pJv3tIR6OdmIgBoHF0eoIv3XP8="}]},"_npmUser":{"name":"wdunn001","email":"wdunn001@gmail.com"},"directories":{},"maintainers":[{"name":"wdunn001","email":"wdunn001@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/maps-cli_0.5.0_1779081201869_0.4759042718467479"},"_hasShrinkwrap":false}},"time":{"created":"2026-05-06T07:31:52.908Z","modified":"2026-05-18T05:13:22.210Z","0.1.0":"2026-05-06T07:31:53.140Z","0.1.1":"2026-05-06T11:08:31.559Z","0.2.0":"2026-05-06T12:49:09.322Z","0.4.1":"2026-05-16T17:58:53.163Z","0.5.0":"2026-05-18T05:13:22.059Z"},"bugs":{"url":"https://github.com/wdunn001/Codec/issues"},"license":"MIT","homepage":"https://github.com/wdunn001/codec-maps","keywords":["codec","tokenizer","llm","huggingface","dialect","schema","cli"],"repository":{"type":"git","url":"git+https://github.com/wdunn001/Codec.git","directory":"packages/maps-cli"},"description":"Generate Codec tokenizer dialect maps from HuggingFace tokenizer.json files. The 'tsc --declaration' for LLM token vocabularies.","maintainers":[{"name":"wdunn001","email":"wdunn001@gmail.com"}],"readme":"# @codecai/maps-cli\n\n**The `tsc --declaration` for LLM token vocabularies.**\n\nGenerate [Codec](https://github.com/wdunn001/Codec) tokenizer dialect maps from HuggingFace `tokenizer.json` files. Maps are content-addressed, immutable JSON files that any [`@codecai/web`](https://www.npmjs.com/package/@codecai/web) client can use to encode/decode token streams.\n\n## Install\n\n```bash\nnpm install -g @codecai/maps-cli\n```\n\nOr run without installing:\n\n```bash\nnpx @codecai/maps-cli build Qwen/Qwen2.5-7B-Instruct --id=qwen/qwen2\n```\n\n## CLI\n\n### `build` — fetch from HuggingFace and convert\n\n```bash\ncodecai-maps build <hf-model> [--id=<id>] [--out=<path>] [--token=<hf-token>]\n```\n\nFetches `tokenizer.json` from `https://huggingface.co/<hf-model>`, converts to a Codec `TokenizerMap`, writes JSON to disk, and prints the canonical sha256 hash.\n\n```bash\n$ codecai-maps build Qwen/Qwen2.5-7B-Instruct --id=qwen/qwen2\n▶ fetching Qwen/Qwen2.5-7B-Instruct from HuggingFace…\n✓ written  qwen_qwen2.json\n  id           qwen/qwen2\n  vocab_size   151665\n  encoder      byte_level\n  merges       151387\n  hash         sha256:c73972f7a580…\n```\n\nFor gated models (Llama, Gemma) pass a HuggingFace access token: `--token=hf_xxx`.\n\n### `convert` — local file in, map out\n\n```bash\ncodecai-maps convert ./tokenizer.json --id=my-org/my-model --out=./my-model.json\n```\n\n### `validate` — schema check\n\n```bash\ncodecai-maps validate ./qwen_qwen2.json\n```\n\n### `hash` — print canonical sha256\n\n```bash\ncodecai-maps hash ./qwen_qwen2.json\n# → sha256:c73972f7a580936d724ffd8df9df2ce546d255c543e9d09b6d75e5bf69b1a64d\n```\n\nUse this value when pinning a map: `loadMap({ url, hash })` will reject any map that doesn't match.\n\n### `preview` — sanity check round-trip\n\n```bash\ncodecai-maps preview ./qwen_qwen2.json --text=\"Explain entropy.\"\n# map:           qwen/qwen2\n# tokenizer:     BPETokenizer\n# input:         \"Explain entropy.\"\n# token IDs:     [840, 20772, 47502, 13]\n# token count:   4\n# round-trip:    \"Explain entropy.\"\n# exact match:   YES\n```\n\n### `translate` — cross-vocab token stream conversion\n\nPipe one tokenizer's IDs through another's vocab with streaming-safe\nword-boundary buffering. Useful for previewing what an agent-to-agent\nhandoff actually emits at the token level.\n\n```bash\ncodecai-maps translate --from=qwen2.json --to=llama-3.json \\\n  --text=\"The quick brown fox.\"\n\n# from:    qwen/qwen2\n# to:      meta-llama/llama-3\n# input:   \"The quick brown fox.\"\n# src ids: [785, 4937, 13876, 38835, 13]   (5 tokens, qwen-2)\n# dst ids: [791, 4062, 14198, 39935, 13]   (5 tokens, llama-3)\n# decoded: \"The quick brown fox.\"          (round-trip via llama-3 detok)\n```\n\nOr with raw IDs:\n\n```bash\ncodecai-maps translate --from=qwen2.json --to=llama-3.json --ids=785,4937\n```\n\n### `translation-table` — context-free V_A → V_B[] lookup\n\n```bash\ncodecai-maps translation-table --from=qwen2.json --to=llama-3.json \\\n  --out=qwen-to-llama.json\n```\n\nEmits a JSON file mapping every non-special source ID to the sequence\nof target IDs its rendered text encodes to. Context-free (BPE merges\ndepend on context), so prefer the streaming `translate` for runtime\nuse; the static table is for analysis (vocab overlap, cost estimation).\n\n## Programmatic API\n\n```ts\nimport { convertHFTokenizer, fetchAndConvert, hashMap } from '@codecai/maps-cli/convert';\n\n// From a parsed tokenizer.json object\nconst map = convertHFTokenizer(hfJson, { id: 'my-org/my-model' });\n\n// Or fetch from HuggingFace directly\nconst map = await fetchAndConvert({\n  hfModel: 'Qwen/Qwen2.5-7B-Instruct',\n  id: 'qwen/qwen2',\n});\n\n// Compute the hash for pinning\nconst hash = await hashMap(map);\n```\n\n## What gets generated\n\nThe output is a JSON file matching the `TokenizerMap` schema from `@codecai/web` (v2.1):\n\n```json\n{\n  \"id\": \"qwen/qwen2\",\n  \"version\": \"2\",\n  \"vocab_size\": 151665,\n  \"vocab\": { \"Hello\": 9707, \"Ġworld\": 1879, \"...\": 0 },\n  \"encoder\": \"byte_level\",\n  \"merges\": [\"Ġ Ġ\", \"ĠĠ ĠĠ\", \"i n\", \"...\"],\n  \"pre_tokenizer_pattern\": \"(?i:'s|'t|'re|...)| ?\\\\p{L}+|...\",\n  \"pre_tokenizer_program\": {\n    \"version\": 1,\n    \"ops\": [\n      { \"op\": \"literals_ci\", \"patterns\": [\"'s\",\"'t\",\"'re\",\"'ve\",\"'m\",\"'ll\",\"'d\"] },\n      { \"op\": \"letters\",     \"lead_other\": true },\n      { \"op\": \"numbers\",     \"max_run\": 1 },\n      { \"op\": \"punct_run\",   \"lead_space\": true, \"trailing_newlines\": true },\n      { \"op\": \"newline_block\" },\n      { \"op\": \"trailing_ws\" },\n      { \"op\": \"ws_run\" }\n    ]\n  },\n  \"special_tokens\": {\n    \"<|endoftext|>\": 151643,\n    \"<|im_start|>\": 151644\n  },\n  \"published_at\": \"2026-05-06T12:00:00.000Z\"\n}\n```\n\nThe schema covers three tokenizer families that span ~95% of open models:\n\n- **`byte_level`** — GPT-2 byte→unicode BPE (Llama-3, Qwen, Phi-3, Mistral-Nemo, DeepSeek-V3, …).\n- **`metaspace`** — `▁`-prefix BPE with byte fallback (Llama-2, Mistral-v3, Mixtral, Gemma).\n- **identity** — vocab-only tokenizers without merges (canonical-IR / closed vocabs).\n\n### Pre-tokenizer program (v2.1, additive)\n\nBoth `pre_tokenizer_pattern` and `pre_tokenizer_program` describe the\nsame splitter. The program is the regex compiled into a named-op list;\nruntimes prefer it when present so they can encode without a Unicode\nregex engine. The CLI emits it automatically for any pre-tokenizer\nregex it recognises (currently the GPT-2-family canonical form used by\nLlama-3, Qwen, Phi-3, DeepSeek-V3, Mistral-Nemo, Falcon, SmolLM2,\nCodestral byte_level). Maps with unrecognised regexes still build\nnormally — `pre_tokenizer_program` is just omitted, and runtimes fall\nback to the regex string.\n\nSee [`spec/PRETOKENIZER_PROGRAM.md`](https://github.com/wdunn001/Codec/blob/main/spec/PRETOKENIZER_PROGRAM.md)\nfor the full op set and equivalence rules.\n\n## Hosting your map\n\nOnce generated, host the JSON anywhere static:\n\n- **GitHub + jsDelivr** (free CDN): commit to a public repo, then  \n  `https://cdn.jsdelivr.net/gh/<user>/<repo>/path/to/map.json`\n- **Hugging Face**: push to a Space or alongside your model weights.\n- **S3 / Cloudflare R2**: standard static hosting.\n- **Codec community registry**: contribute via PR to [`codec-maps`](https://github.com/wdunn001/codec-maps).\n\nThen any client can pin against your hash:\n\n```ts\nimport { loadMap } from '@codecai/web';\n\nconst map = await loadMap({\n  url: 'https://your-host/your-model.json',\n  hash: 'sha256:abcd1234…',\n});\n```\n\n### `well-known` — publish for `.well-known/codec/` discovery\n\nGenerate the static directory tree clients need to find your map by `(origin, id)` alone, so consumers don't have to hard-code your CDN URL:\n\n```bash\ncodecai-maps well-known --map=./qwen_qwen2.json \\\n  --url=https://cdn.example/qwen2.json \\\n  --out-dir=./public\n```\n\nThis writes:\n\n```\npublic/.well-known/codec/maps/qwen/qwen2.json   ← pointer { id, url, hash }\npublic/.well-known/codec/index.json             ← directory of all your maps\n```\n\nDrop `./public` onto any static host (GitHub Pages, S3, Vercel) under the origin you control, and any client can do:\n\n```ts\nimport { discoverMap } from '@codecai/web';\nconst map = await discoverMap({ origin: 'https://qwen.io', id: 'qwen/qwen2' });\n```\n\nPass `--inline` instead of `--url` to embed the full map at the well-known location (skips the CDN indirection — recommended only for small maps). Re-running with the same id replaces the existing index entry. See [`spec/WELL_KNOWN_DISCOVERY.md`](https://github.com/wdunn001/Codec/blob/main/spec/WELL_KNOWN_DISCOVERY.md) for the publishing contract.\n\n### `policies-*` — safety-policy descriptor lifecycle (v0.4)\n\nThe v0.4 [safety-policy negotiation spec](https://github.com/wdunn001/Codec/blob/main/spec/versions/v0.4.md#safety-policy-negotiation) ships four CLI subcommands that mirror the tokenizer-map shape exactly:\n\n```bash\n# Validate that an operator-internal policy is well-formed.\ncodecai-maps policies-validate ./internal-config.json\n\n# Strip internal-only fields (banned_token_ids, regex_patterns,\n# grammar_constraints, multi_token_patterns, classifier thresholds /\n# weights) and emit the publishable descriptor — what the world sees\n# at .well-known/codec/policies/<id>.json. Internal-field counts\n# survive as rules_summary.* for auditors.\ncodecai-maps policies-sanitize --internal=./internal-config.json \\\n  --out=./acme-strict-v3.policy.json\n\n# Canonical sha256 over the sanitized descriptor — bit-identical\n# across @codecai/web, codecai (Python), codec-rs, Codec.Net,\n# codec (Java), libcodec, and codec-supervisor.\ncodecai-maps policies-hash ./acme-strict-v3.policy.json\n\n# Emit both .well-known/codec/policies/<id>.json (mutable pointer or\n# inline) AND .well-known/codec/policies/sha256/<hex>.json (immutable\n# content-addressed sibling) so clients that received a hash in READY\n# can fetch + verify without a redirect hop.\ncodecai-maps policies-well-known --descriptor=./acme-strict-v3.policy.json \\\n  --inline --out-dir=./public\n\n# v0.5 (resolves v0.4-OQ4): productize the offline enumerator scripts.\n# Reads a JSON array of literal strings, generates surface variants\n# (verbatim / leading-space / leading-newline / lowercase / titlecase /\n# uppercase / trimmed), tokenizes each variant through the supplied\n# tokenizer map, deduplicates by token sequence, and writes a JSON file\n# ready to paste into your internal policy's 'multi_token_patterns'\n# field. The output pins the tokenizer-map sha256 so the enumeration is\n# verifiably tied to the exact map bytes the runtime will tokenize\n# against.\ncodecai-maps policies-enumerate --map=./qwen_qwen2.json \\\n  --literals=./adversarial-strings.json \\\n  --out=./enumerated-patterns.json\n```\n\nThe descriptor never contains operator-internal contents — that's the\ndisclosure-boundary contract (an attacker who can fetch the\n`.well-known` page learns the *shape* of enforcement, not the contents\nof banned-token lists or classifier thresholds). The internal-config\nside lives in [`codec-supervisor`](https://github.com/wdunn001/codec-supervisor)\nunder `policies_dir/` and is never published.\n\n## License\n\nMIT.\n","readmeFilename":"README.md"}