{"_id":"@dwight-ship-it/context-cost","_rev":"3-467ae7a4f05c73ac975e362d5bcf21de","name":"@dwight-ship-it/context-cost","dist-tags":{"latest":"0.9.0"},"versions":{"0.6.0":{"name":"@dwight-ship-it/context-cost","version":"0.6.0","keywords":["claude-code","claude","plugin","mcp","tokens","context-window","cost","cli"],"license":"MIT","_id":"@dwight-ship-it/context-cost@0.6.0","maintainers":[{"name":"andrfaria","email":"atlasallyai@gmail.com"}],"homepage":"https://github.com/Dwight-ship-it/context-cost#readme","bugs":{"url":"https://github.com/Dwight-ship-it/context-cost/issues"},"bin":{"context-cost":"dist/cli/index.js"},"dist":{"shasum":"c0dd6354bd391c7f085db0f2ce8f71f18143df0d","tarball":"https://registry.npmjs.org/@dwight-ship-it/context-cost/-/context-cost-0.6.0.tgz","fileCount":72,"integrity":"sha512-K65Q5mu6xu6gXepx35DVsxvR164c7BSrA22dDXdCUyrNuA0aXM+KAhIRGerFx3XzA/EnaeM/yAo74ct+HkgYHg==","signatures":[{"sig":"MEYCIQCEyLaQkkjlKV/k7zTHM+NSANwSNCbns1sLiq+7VH1TPAIhAJMrhfCXlZrJ3qK+V704LAu2Ajeyo9mrMyqkHVVSG7zk","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":96671},"type":"module","engines":{"node":">=18"},"gitHead":"a8822154e7a152b948b6c115cb252001b02b14b1","scripts":{"dev":"tsx src/cli/index.ts","test":"vitest run","build":"tsc","prepare":"npm run build","version":"node scripts/changelog-release.mjs && git add CHANGELOG.md","test:watch":"vitest","prepublishOnly":"npm run build && npm test"},"_npmUser":{"name":"andrfaria","email":"atlasallyai@gmail.com"},"repository":{"url":"git+https://github.com/Dwight-ship-it/context-cost.git","type":"git"},"_npmVersion":"11.16.0","description":"Estimate the per-session token cost a Claude Code plugin adds to your context window.","directories":{},"_nodeVersion":"22.22.3","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.7.0","vitest":"^1.4.0","typescript":"^5.4.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/context-cost_0.6.0_1780697186261_0.8445543272383733","host":"s3://npm-registry-packages-npm-production"}},"0.7.0":{"name":"@dwight-ship-it/context-cost","version":"0.7.0","keywords":["claude-code","claude","plugin","mcp","tokens","context-window","cost","cli"],"license":"MIT","_id":"@dwight-ship-it/context-cost@0.7.0","maintainers":[{"name":"andrfaria","email":"atlasallyai@gmail.com"}],"homepage":"https://github.com/Dwight-ship-it/context-cost#readme","bugs":{"url":"https://github.com/Dwight-ship-it/context-cost/issues"},"bin":{"context-cost":"dist/cli/index.js"},"dist":{"shasum":"c1a932957c7f4356fee682bec3af4ac42540cbf9","tarball":"https://registry.npmjs.org/@dwight-ship-it/context-cost/-/context-cost-0.7.0.tgz","fileCount":84,"integrity":"sha512-JcA4ZgrpiiFq8dGCQAedeiW726x+pGw+Encmv2hzbOCdl9fWU7xeEqzMYDjUBqNx2+Ag0Up6e7PQMfyrRcmXvQ==","signatures":[{"sig":"MEYCIQDu8k13OlPecVDaMR6mts2plSZJeJefNo9/X6Lyofoe2QIhAIJOxKaOSxgH0VulaBFIGpRbck+6aPX8t3x+f8sWFg8p","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":124023},"type":"module","engines":{"node":">=18"},"gitHead":"e4bee98f3e1b8d1e8d2d7208770e929e84339eb6","scripts":{"dev":"tsx src/cli/index.ts","test":"vitest run","build":"tsc","smoke":"npm run build && node scripts/smoke.mjs","prepare":"npm run build","version":"node scripts/changelog-release.mjs && git add CHANGELOG.md","test:watch":"vitest","prepublishOnly":"npm run build && npm test"},"_npmUser":{"name":"andrfaria","email":"atlasallyai@gmail.com"},"repository":{"url":"git+https://github.com/Dwight-ship-it/context-cost.git","type":"git"},"_npmVersion":"11.16.0","description":"Estimate the per-session token cost a Claude Code plugin adds to your context window.","directories":{},"_nodeVersion":"22.22.3","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.7.0","vitest":"^1.4.0","typescript":"^5.4.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/context-cost_0.7.0_1780769062012_0.20496212684538184","host":"s3://npm-registry-packages-npm-production"}},"0.9.0":{"name":"@dwight-ship-it/context-cost","version":"0.9.0","description":"Estimate the per-session token cost a Claude Code plugin adds to your context window.","license":"MIT","type":"module","bin":{"context-cost":"dist/cli/index.js"},"keywords":["claude-code","claude","plugin","mcp","tokens","context-window","cost","cli"],"repository":{"type":"git","url":"git+https://github.com/Dwight-ship-it/context-cost.git"},"homepage":"https://github.com/Dwight-ship-it/context-cost#readme","bugs":{"url":"https://github.com/Dwight-ship-it/context-cost/issues"},"publishConfig":{"access":"public"},"engines":{"node":">=18"},"scripts":{"build":"tsc","dev":"tsx src/cli/index.ts","test":"vitest run","test:watch":"vitest","smoke":"npm run build && node scripts/smoke.mjs","prepare":"npm run build","prepublishOnly":"npm run build && npm test","version":"node scripts/changelog-release.mjs && git add CHANGELOG.md","benchmark":"npm run build && node scripts/benchmark.mjs"},"peerDependencies":{"tiktoken":">=1.0.0"},"peerDependenciesMeta":{"tiktoken":{"optional":true}},"devDependencies":{"@types/node":"^20.0.0","tsx":"^4.7.0","typescript":"^5.4.0","vitest":"^1.4.0"},"gitHead":"f94057340f6a545c8cc46de701f3a2d4b7f1983e","_id":"@dwight-ship-it/context-cost@0.9.0","_nodeVersion":"22.22.3","_npmVersion":"11.16.0","dist":{"integrity":"sha512-vYKdXpV+416N3ayAQ0qKBUDelLCf9gkfDoGdX0E6aqPbdEkrzR1IddBQMv7FfbEBYLX/H0jA6NN1wJtgpoNvZQ==","shasum":"68f4193b30622acd62d828cb04f33058d8f6cae5","tarball":"https://registry.npmjs.org/@dwight-ship-it/context-cost/-/context-cost-0.9.0.tgz","fileCount":84,"unpackedSize":157874,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDwpyUAtrPPVQId4DlU9LYhtG7m8ywpdyF+828SYa8kVAIgKo2/Y9flTkETbTvjAEk0Jq6r92Nmtc9JOFJC/KdVcuY="}]},"_npmUser":{"name":"andrfaria","email":"atlasallyai@gmail.com"},"directories":{},"maintainers":[{"name":"andrfaria","email":"atlasallyai@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/context-cost_0.9.0_1780788743182_0.6087953041175489"},"_hasShrinkwrap":false}},"time":{"created":"2026-06-05T22:06:26.076Z","modified":"2026-06-06T23:32:23.434Z","0.6.0":"2026-06-05T22:06:26.416Z","0.7.0":"2026-06-06T18:04:22.148Z","0.9.0":"2026-06-06T23:32:23.323Z"},"bugs":{"url":"https://github.com/Dwight-ship-it/context-cost/issues"},"license":"MIT","homepage":"https://github.com/Dwight-ship-it/context-cost#readme","keywords":["claude-code","claude","plugin","mcp","tokens","context-window","cost","cli"],"repository":{"type":"git","url":"git+https://github.com/Dwight-ship-it/context-cost.git"},"description":"Estimate the per-session token cost a Claude Code plugin adds to your context window.","maintainers":[{"name":"andrfaria","email":"atlasallyai@gmail.com"}],"readme":"# context-cost\n\n[![CI](https://github.com/Dwight-ship-it/context-cost/actions/workflows/ci.yml/badge.svg)](https://github.com/Dwight-ship-it/context-cost/actions/workflows/ci.yml)\n[![npm](https://img.shields.io/npm/v/@dwight-ship-it/context-cost)](https://www.npmjs.com/package/@dwight-ship-it/context-cost)\n[![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)\n\n**What it is:** a zero-dependency CLI that **estimates** the per-session token\ncost a Claude Code plugin adds to your context window — broken down by skills,\nMCP servers, hooks, and injected context — *before* you install it, and measures\nwhat's already installed from your own session logs.\n\n**Why it exists:** installing plugins quietly inflates the context window of\nevery new session. The unit you install is the **plugin**, but the cost hides\ninside it: MCP tool schemas and — most easily missed — **hooks that inject\ncontext every turn**. `context-cost` makes that cost visible, and is honest about\nwhat it can and can't know statically (estimated / variable / unknown are never\ndressed up as exact).\n\n## Install & run (15 seconds)\n\nFrom npm (scoped package, `context-cost` command — published from v0.6.0):\n\n    npx @dwight-ship-it/context-cost inspect ./path/to/plugin\n    # or install it on your PATH:\n    npm install -g @dwight-ship-it/context-cost\n\nStraight from GitHub — always works, no npm needed (used in CI too):\n\n    npx github:Dwight-ship-it/context-cost inspect ./path/to/plugin\n\nOr clone for local/contributor use:\n\n    git clone https://github.com/Dwight-ship-it/context-cost\n    cd context-cost && npm install && npm run build\n    npm link        # optional — puts `context-cost` on your PATH\n\nWithout `npm link`, run `node dist/cli/index.js …` in place of `context-cost`.\nThe npm package is **scoped** (`@dwight-ship-it/context-cost`) because the\nunscoped name `context-cost` belongs to an unrelated package — see\n[installation paths](#installation-paths).\n\n## Example output\n\nPredict a plugin's cost before installing it — fully static, never runs its code:\n\n```console\n$ context-cost inspect examples/bad-plugin\n  bad-plugin  v0.9.0                      grade: D\n  ─────────────────────────────────────────────────────────────\n  confidence: partial · risk: high · 4 unmeasured (3 variable, 1 unknown)\n  PER SESSION (fixed)   ~140 tok\n  PER TURN (added)      ~0 tok\n  ON-DEMAND (if used)   ~0 tok\n\n  Breakdown\n    skill       mega                     ~140 tok   per-session  metadata only; body loads on-demand\n    mcp-server  everything               unknown    per-session  MCP tools not statically declared — run `--deep` or `audit` to measure\n    hook        SessionStart (startup)   variable   per-session  injects context / runs code per-session — run `audit` to measure\n    hook        UserPromptSubmit         variable   per-turn  injects context / runs code per-turn — run `audit` to measure\n    hook        PreToolUse (*)           variable   per-tool-call  injects context / runs code per-tool-call — run `audit` to measure\n\n  ⚠ Flags\n    • 3 hook(s) inject context or run every turn — real cost not knowable without installing. Run `context-cost audit` after install to measure.\n    • 2 hook(s) run every turn or tool call (e.g. UserPromptSubmit, Pre/PostToolUse) — cost recurs unbounded per turn; `audit` to measure the real total.\n    • 1 MCP server(s) don't declare tools statically — run `--deep` to measure.\n```\n\nThe `confidence / risk / unmeasured` line is the **trust summary**: how much of\nthe grade rests on real numbers, and how much hidden cost the unmeasured parts\ncould add. Here, four components can't be sized statically — so a `D` grade\nbuilt from `~140` known tokens is flagged `partial` confidence and `high` risk,\nnot taken at face value. See [Trust signals](#trust-signals).\n\nTurn that into ranked, actionable advice:\n\n```console\n$ context-cost optimize examples/bad-plugin\n  bad-plugin  v0.9.0                      grade: D\n  ...\n  Recommendations  (4 total, 4 high)\n    HIGH   INJECTING_HOOK    UserPromptSubmit   [cost unknown] Hook injects context every turn — real cost unknown until measured.\n    HIGH   UNDECLARED_MCP    everything         [cost unknown] MCP server does not declare tools statically — opaque until run.\n    ...\n```\n\nMeasure what's **already installed** — this reads your real session transcripts and\nattributes the context injecting hooks actually added:\n\n```console\n$ context-cost audit --plugin claude-mem\n  claude-mem  v13.4.0                      grade: B\n  ─────────────────────────────────────────────────────────────\n  confidence: partial · risk: medium · 1 unmeasured (1 variable, 0 unknown)\n  PER SESSION (fixed)   ~1098 tok\n  PER TURN (added)      ~0 tok\n  ON-DEMAND (if used)   ~0 tok\n  ...\n```\n\nDiff two plugins (or two revisions) and fail CI on a regression:\n\n```console\n$ context-cost compare examples/clean-plugin examples/bad-plugin --fail-on-grade-drop\n  compare  clean-plugin v1.0.0  →  bad-plugin v0.9.0\n  ─────────────────────────────────────────────────────────────\n  PER SESSION (fixed)    ~16 → ~140 tok   (+124)  ▲\n  GRADE                  A → D  ▼ worse\n  ...\ncontext-cost: gate failed — grade dropped A → D     # ← stderr\n$ echo $?\n1\n```\n\n## Current coverage\n\nWhat `context-cost` measures today, and what it doesn't yet. \"Partial\" means it\ncontributes a number but with a caveat worth reading.\n\n| Capability | Status | Notes |\n|------------|--------|-------|\n| `skills/SKILL.md` | ✅ Supported | Front-matter (name + description) estimated; the body loads on-demand and isn't counted as fixed cost. |\n| `hooks/hooks.json` | ✅ Supported | Events detected and classified; injecting events (SessionStart, UserPromptSubmit, Pre/PostToolUse) flagged `variable`. |\n| `.mcp.json` static detection | ✅ Supported | Launch config is read statically; tool **schemas** aren't declared there, so servers are flagged `unknown` until measured. |\n| Local stdio MCP deep measurement | ✅ Supported | `inspect --deep --allow-exec` runs each local stdio server, enumerates tools, reports `estimated`. |\n| Remote HTTP/SSE MCP | ❌ Not yet | Remote servers are never contacted and are not measured. |\n| Commands (`commands/*.md`) | ✅ Supported | Each command's full markdown is costed as **on-demand** (injected only when invoked) — `estimated`, never per-session fixed. |\n| Agents (`agents/*.md`) | ✅ Supported | Name + description costed as **per-session** metadata (mirrors skills); the body runs in a separate subagent context and isn't counted. |\n| Injected session-context attribution | ⚠️ Partial | `audit` measures injected tokens from transcripts, but attributes them **by hook event, not by plugin** — if two plugins inject on the same event the total is shared. Only `hook_success` transcript entries are attributed. |\n| Malformed config files | ✅ Supported | An unparseable `.mcp.json` / `hooks.json` is reported as an `unknown` blind spot (never crashes, never guessed), lowering `confidence` and capping the grade. |\n\n## When should I use this?\n\n**Use it if you:**\n\n- install Claude Code plugins and want to know what each one costs your context\n  window *before* committing to it;\n- maintain several plugins/MCP servers and suspect one is quietly bloating every\n  session;\n- author a plugin and want a committable cost gate (`manifest --fail-on`) or a\n  before/after check on a change (`compare`);\n- need a quick, honest order-of-magnitude read rather than exact token billing.\n\n**Skip it (for now) if you:**\n\n- need exact, billable token counts — this uses a heuristic `~chars/4` tokenizer\n  tuned for *relative ranking*, not Claude's exact tokenizer;\n- want costs for remote HTTP/SSE MCP servers — only local stdio servers can be\n  measured today;\n- expect it to measure a plugin's cost without running it: undeclared MCP tool\n  schemas are genuinely unknowable statically and are reported as `unknown`/\n  flagged, not guessed.\n\nIt is most valuable as a **decision and regression tool**, least valuable as a\nbilling meter.\n\n## Safety model\n\n`context-cost` is built so the *default* path can never run untrusted code. Here\nis exactly what each command does:\n\n| Command | Reads | Executes plugin code? |\n|---------|-------|-----------------------|\n| `inspect` (default) | plugin manifest, `SKILL.md` front-matter, `.mcp.json`, `hooks.json` — as **text** | **No.** Never. |\n| `optimize` / `manifest` / `badge` / `compare` | the same static report as `inspect` | **No.** Pure consumers of the report. |\n| `audit` | your own session transcripts under `~/.claude/projects` | **No.** It reads logs of sessions that already ran; it launches nothing. |\n| `inspect --deep --allow-exec` | the above, **plus** live MCP tool schemas | **Yes — local stdio MCP servers only**, and only with the explicit flag. |\n\nThe only command that executes anything is **`inspect --deep --allow-exec`**, and\nit is doubly gated:\n\n1. `--deep` alone **refuses** and instead prints the exact commands it *would*\n   run, so you can read them first:\n\n   ```console\n   $ context-cost inspect ./some-plugin --deep\n   context-cost: --deep would EXECUTE the following MCP server command(s) from this plugin:\n     search: node server.js\n   This runs code from the plugin you are inspecting.\n   Re-run with --allow-exec to proceed.\n   ```\n\n2. Adding `--allow-exec` launches each declared **local stdio** MCP server,\n   enumerates its tools, and shuts it down. Each server runs under a wall-clock\n   timeout (`--deep-timeout`, default 10s) and a stdout byte cap\n   (`--deep-max-bytes`, default 5MB); if it crashes, the tail of its **stderr** is\n   surfaced in the error so you can see why. Failures are isolated per server.\n   Remote (HTTP/SSE) servers are never contacted and are **not** deeply measured\n   yet. Measured servers are reported `estimated` (real schemas, heuristic\n   tokenizer) — never `exact`.\n\n   **Inherited environment:** a measured server inherits your **full shell\n   environment** (`process.env`) with any per-server `env` from `.mcp.json`\n   layered on top — exactly how Claude Code launches it, so the measurement is\n   realistic. That also means the plugin's code sees your `PATH`, tokens, and\n   other secrets, which is the main reason `--deep --allow-exec` should be run in\n   a disposable environment.\n\nAll `variable`/`unknown` costs are reported **separately** from the token totals\nand are never converted into exact numbers — see the tier table under\n[What it reports](#what-it-reports).\n\n### Recommended safe workflow\n\n1. **Static first.** `context-cost inspect <target>` — no code runs. This is\n   enough to decide whether to install.\n2. **Review the execution plan.** If you want deep MCP numbers, see exactly what\n   would run first — it executes nothing:\n\n   ```console\n   $ context-cost inspect <target> --print-exec-plan\n   context-cost: deep MCP measurement would EXECUTE these local stdio command(s):\n     search: node server.js\n   Nothing was executed (--print-exec-plan). ...\n   ```\n\n   `--print-exec-plan` is always safe — it overrides `--allow-exec`.\n3. **Only then, deep — in a disposable environment.** Run\n   `inspect <target> --deep --allow-exec` inside a sandbox/container you can throw\n   away, never against untrusted code on your main machine.\n\nSo: deciding *whether* to install is always static and safe. Running code is an\nopt-in step you take only after reading what it will run.\n\n### inspect vs audit\n\n- **`inspect`** predicts cost from static files — use it *before* installing.\n- **`audit`** measures the **real** injected-context cost already paid, from your\n  session transcripts, upgrading injecting hooks from `variable` to `estimated`\n  where it can attribute them. Attribution is **by hook event, not by plugin**\n  (if two plugins inject on the same event the measured total is shared), which\n  is why it is reported as `estimated` rather than `exact`. Use it *after*\n  installing.\n\n## Commands\n\n    context-cost inspect github:owner/repo        # predict cost before install\n    context-cost inspect github:owner/repo#v1.2.3  # pin to a tag or branch\n    context-cost inspect ./path/to/plugin         # predict from a local path\n    context-cost audit                       # measure installed plugins\n    context-cost audit --plugin claude-mem   # one plugin (prefix match)\n    context-cost optimize <target>           # ranked cost-reduction advice\n    context-cost manifest <target>           # committable cost-manifest.json\n    context-cost badge <target>              # Shields.io badge for your README\n    context-cost compare <before> <after>    # diff two plugins / two versions\n\n    context-cost inspect <target> --json     # raw PluginCostReport JSON\n\nFlags:\n\n    --json                          emit raw JSON (report / comparison / manifest)\n    --plugin <name>                 audit: filter installed plugins by id prefix\n    --output <path>                 manifest: write the manifest JSON to a file\n    --fail-on <grade>               manifest: exit 1 if grade is WORSE than <grade>\n    --deep                          inspect: measure MCP servers by running them (needs --allow-exec)\n    --allow-exec                    permit --deep to execute the plugin's MCP server commands\n    --print-exec-plan               inspect: show what --deep would run, then exit (runs nothing)\n    --deep-timeout <ms>             inspect: per-server timeout for --deep (default 10000)\n    --deep-max-bytes <n>            inspect: per-server stdout byte cap for --deep (default 5000000)\n    --fail-on-grade-drop            compare: exit 1 if the grade got worse\n    --fail-on-unmeasured-increase   compare: exit 1 if variable/unknown component count rose\n    --fail-on-session-increase <n>  compare: exit 1 if per-session fixed tokens rose by > n\n\n### What it reports\n\nEvery plugin is broken into components, each with a **cadence** (per-session /\nper-turn / per-tool-call / on-demand) and a **confidence tier**:\n\n| Tier | Meaning |\n|------|---------|\n| `exact` | a value known with certainty (e.g. a non-injecting hook that adds **0** tokens) |\n| `estimated` | a counted number with heuristic-tokenizer/attribution caveats (skill metadata; deep-measured MCP tools; hook tokens measured from session logs and attributed by event, not plugin) |\n| `variable` | injects context / runs every turn — not knowable without installing |\n| `unknown` | can't be sized statically (e.g. MCP tools not declared in `.mcp.json`) |\n\n| Component | How it's costed |\n|-----------|-----------------|\n| Skills | metadata only (name + description); the body loads on-demand |\n| MCP servers | flagged `unknown` — `.mcp.json` declares launch config, not tool schemas |\n| Hooks | injecting events (SessionStart, UserPromptSubmit, Pre/PostToolUse) flagged `variable`; others 0; unrecognised events `unknown` |\n| Commands | `commands/*.md` body costed `on-demand` (`estimated`) — injected only when the command is invoked, so never a per-session cost |\n| Agents | `agents/*.md` name + description costed `per-session` (`estimated`); the body runs in a separate subagent context and isn't counted |\n\nTotals are split three ways — **per-session fixed** (paid every session just by\nhaving it installed), **per-turn added**, and **on-demand potential** (only if\nused) — plus an A–F grade and flags for the things that hurt most (injecting\nhooks, undeclared MCP servers).\n\n> ⚠️ **The A–F grade is a decision aid, not a billable or safety score.** It is\n> derived from a heuristic token estimate (`~chars/4`), not Claude's exact\n> tokenizer, and says nothing about a plugin's security, correctness, or quality.\n> Don't use it for billing or as a safety gate.\n\n**How the grade is computed.** The starting point is the static token total, but\nthe grade is then *capped* by how much of the plugin is unmeasurable, so hidden\ncost can't hide behind a near-zero static count:\n\n- a per-turn / per-tool-call injecting hook, 3+ injecting hooks, or an injecting\n  hook combined with an undeclared MCP server (i.e. `risk: high`) → **at most D**;\n- a single undeclared MCP server, a lone session-only injecting hook, or another\n  isolated unmeasured surface (`risk: medium`) → **at most C**;\n- a report where *everything* found is unmeasurable (`confidence: low`) → **at\n  most C**.\n\nCaps only ever pull a grade **down**, never up, and unmeasured costs are still\nnever converted into invented token numbers — they cap the letter grade and are\ncalled out in the flags and the `unmeasured` count instead.\n\n#### Trust signals\n\nSo the grade is never read as more certain than it is, every report carries two\nextra axes alongside it (in the human output and in `--json`):\n\n| Signal | Values | Answers |\n|--------|--------|---------|\n| `confidence` | `full` · `partial` · `low` | How much of the report is real numbers? `full` = nothing unmeasured; `partial` = a mix; `low` = everything found was unsizable. |\n| `risk` | `low` · `medium` · `high` | How much hidden context-bloat could the unmeasured parts add? `high` = a per-turn injecting hook, 3+ injecting hooks, or injecting-hook + undeclared MCP together. |\n| `unmeasured` | `{ variable, unknown, total }` | How many components couldn't be sized statically, split by reason. |\n\n`confidence` and `risk` are a separate axis from the token totals — they describe\n*how much you can trust the numbers*, not the numbers themselves. They also feed\nthe grade caps above: a plugin that is *nothing but* an undeclared MCP server has\na `~0` known-token total, but because its real cost is entirely unmeasured it\nlands at `low` confidence / `medium` risk and is capped at `C`, never read as a\nclean `A`. The totals themselves stay honest — only the letter grade is capped,\nand no unmeasured cost is ever turned into a fabricated token number.\n\n### optimize, manifest, badge & compare\n\nThese build on the same report — no new analysis, no new dependencies.\n\n- **`optimize <target>`** turns a report into **ranked, actionable advice**:\n  which components cost the most and what to do about them (gate an injecting\n  hook, audit an undeclared MCP server, trim heavy per-session metadata).\n  Recommendations are ordered by token impact; anything whose cost is `unknown`\n  or `variable` is flagged `[cost unknown]` and never ranked as if it were exact.\n\n- **`manifest <target>`** emits a stable, versioned **`cost-manifest.json`** a\n  plugin author can commit to their repo — `schemaVersion`, totals, grade,\n  honesty `flags`, and recommendation severity counts (not the full text, so\n  minor optimizer tweaks don't churn the file). Write it with `--output`:\n\n      context-cost manifest ./my-plugin --output cost-manifest.json\n\n  Use **`--fail-on <grade>`** as a CI gate — it exits non-zero when the plugin\n  grades worse than your threshold, so a regression fails the build:\n\n      context-cost manifest ./my-plugin --fail-on B   # exit 1 if C/D/F\n\n- **`badge <target>`** prints a [Shields.io](https://shields.io) static badge\n  URL plus a ready-to-paste Markdown snippet:\n\n      ![context-cost](https://img.shields.io/badge/context--cost-A_%C2%B7_~16_tok-brightgreen)\n\n  Grade drives the colour (A→brightgreen … F→red); the `~` marks the count as\n  approximate.\n\n- **`compare <before> <after>`** diffs two reports — two plugins, or the same\n  plugin at two revisions. It shows the per-bucket token delta and the grade\n  direction, and (crucially) reports any change in **unmeasured** (`variable`/\n  `unknown`) components *separately*, so adding an injecting hook never hides\n  behind a `+0` token delta:\n\n      context-cost compare examples/clean-plugin examples/bad-plugin\n\n  ```console\n    compare  clean-plugin v1.0.0  →  bad-plugin v0.9.0\n    ─────────────────────────────────────────────────────────────\n    PER SESSION (fixed)    ~16 → ~140 tok   (+124)  ▲\n    GRADE                  A → D  ▼ worse\n\n    Unmeasured components (variable/unknown — cost not captured above)\n      before: 0   after: 4   (+4)\n      ⚠ numeric deltas above exclude variable/unknown costs — measure with `audit`.\n  ```\n\n  For plugin authors: run it in CI against the last released revision to catch a\n  cost regression in review, before it ships to users. Three optional gates make\n  `compare` exit non-zero so a regression fails the build:\n\n  | Flag | Exits 1 when |\n  |------|--------------|\n  | `--fail-on-grade-drop` | the grade got worse (e.g. B → C) |\n  | `--fail-on-unmeasured-increase` | the count of `variable`/`unknown` components rose — catches an added injecting hook or undeclared MCP server even at `+0` tokens |\n  | `--fail-on-session-increase <tokens>` | per-session fixed tokens rose by **more than** `<tokens>` (use `0` to fail on *any* increase) |\n\n  Gates combine with AND — any tripped gate fails the run, and each prints one\n  reason to stderr:\n\n  ```console\n  $ context-cost compare /tmp/base . --fail-on-grade-drop --fail-on-session-increase 200\n  context-cost: gate failed — grade dropped A → C\n  context-cost: gate failed — per-session fixed cost increased by 240 tok (110 → 350), over the allowed +200\n  $ echo $?\n  1\n  ```\n\n  The `--fail-on-unmeasured-increase` gate is the honest counterpart to the\n  token gate: it reports a **count** of unmeasured components, never a fabricated\n  token figure for costs that can't be sized statically.\n\n## CI for plugin authors\n\nCatch a context-cost regression in pull requests. This workflow checks out the\nPR's base revision into a git worktree and diffs the working tree against it —\nno published action, just the CLI:\n\n```yaml\n# .github/workflows/context-cost.yml\nname: context-cost\non: pull_request\njobs:\n  cost:\n    runs-on: ubuntu-latest\n    steps:\n      - uses: actions/checkout@v4\n        with:\n          fetch-depth: 0          # need the base ref to diff against\n      - uses: actions/setup-node@v4\n        with:\n          node-version: 20\n      - name: Fail on a cost regression\n        run: |\n          git worktree add /tmp/base \"origin/${{ github.base_ref }}\"\n          npx --yes github:Dwight-ship-it/context-cost compare /tmp/base . \\\n            --fail-on-grade-drop \\\n            --fail-on-unmeasured-increase \\\n            --fail-on-session-increase 200\n```\n\n`npx github:Dwight-ship-it/context-cost` installs straight from GitHub. The repo\nships a `prepare` script (`npm run build`) so that a git/`npx` install compiles\n`dist/` automatically. You can equally use the published scoped npm package,\n`npx @dwight-ship-it/context-cost` (published from v0.6.0 — see\n[installation paths](#installation-paths)). Pin to a tag for reproducible CI,\ne.g. `github:Dwight-ship-it/context-cost#v0.6.0`.\n\n**Exit codes** (so CI behaves predictably):\n\n| Code | Meaning |\n|------|---------|\n| `0` | all gates passed |\n| `1` | a gate failed (the reason is printed to **stderr**; stdout/JSON stays valid) |\n| `2` | invalid CLI usage/config (e.g. a non-numeric `--fail-on-session-increase`) |\n\n### Installation paths\n\n- **Scoped npm (from v0.6.0):** `npx @dwight-ship-it/context-cost <command>` or\n  `npm install -g @dwight-ship-it/context-cost`. The package is **scoped** because\n  the unscoped name **`context-cost` on npm belongs to an unrelated package** —\n  this project is **not** it. The installed command stays `context-cost` (the\n  `bin` name is independent of the scoped package name).\n- **GitHub / `npx` (always works):** `npx github:Dwight-ship-it/context-cost\n  <command>`. Builds on install via the `prepare` script, so it works without\n  going through npm at all. Pin a branch or tag for CI, e.g.\n  `…/context-cost#v0.6.0`.\n- **Clone + `npm link` (contributors):** see [Install & run](#install--run-15-seconds).\n\n## Honesty\n\nNumbers are approximate (a heuristic ~chars/4 tokenizer), tuned for **relative\nranking and order-of-magnitude truth, not exact billing**. Every number is\nlabelled measured / estimated / variable / unknown — the tool never silently\nguesses and presents it as fact. For a detailed explanation of the three\ntokenizer modes and why counts are not billing-exact, see\n[`docs/token-count-calibration.md`](docs/token-count-calibration.md).\n\n## Limitations\n\n- `audit` attributes injected content that appears as `hook_success` entries in\n  transcripts; content injected through other mechanisms isn't attributed yet.\n- `inspect --deep --allow-exec` measures MCP tool schemas by running each stdio\n  server locally and enumerating its tools. It **executes code from the plugin\n  under inspection**, so it is gated behind `--allow-exec`; without that flag\n  `--deep` prints the commands it would run and refuses. Remote (HTTP/SSE)\n  servers aren't measured yet. Measured servers are reported `estimated` (real\n  schemas, heuristic tokenizer).\n- The tokenizer is a pluggable approximation, not Claude's exact tokenizer.\n- `github:owner/repo#ref` targets resolve a branch, tag, or commit SHA.\n  Branch/tag refs use a fast shallow clone (`--depth 1`); commit SHAs trigger a\n  full clone + `git checkout`, which is slower. Pin to a tag for reproducibility\n  and performance in CI.\n- `SKILL.md` front-matter is read by a minimal token-estimation parser. It handles\n  plain values and block scalars (folded `>`/`>-`/`>+`, literal `|`/`|-`/`|+`),\n  but is not a spec-complete YAML loader — flow collections and anchors aren't\n  modelled.\n\n## Development\n\n    npm install\n    npm test          # run the vitest suite\n    npm run build     # compile to dist/\n\nPure cost engine under `src/core/` (no CLI deps); thin CLI under `src/cli/`. The\nengine emits a `PluginCostReport` (see `--json`); `optimize`, `manifest`,\n`badge`, and `compare` are pure consumers of that report (`src/core/optimizer/`,\n`src/core/manifest/`, `src/core/compare/`), so the whole tool stays\nzero-dependency.\n\nThe [`examples/`](examples/) directory holds small, self-contained plugin\nfixtures (clean / hook-heavy / mcp-heavy / command-heavy / agent-heavy / bad /\nmalformed) used by both the docs above and `tests/examples.test.ts`. They are\ncontrolled fixtures — `inspect` never runs them; don't pass `--deep --allow-exec`\nagainst them.\n\n## Docs\n\n- [`docs/architecture.md`](docs/architecture.md) — analyzer pipeline, confidence\n  tiers, parser boundaries, static vs deep MCP, and why zero dependencies.\n- [`docs/report-schema.md`](docs/report-schema.md) — the JSON shapes for\n  `inspect`, `compare`, and `manifest`, with stability notes.\n- [`docs/real-world-validation.md`](docs/real-world-validation.md) — results of\n  running `context-cost` against real public plugins.\n- [`docs/token-count-calibration.md`](docs/token-count-calibration.md) — the\n  three tokenizer modes (`heuristic`, `words`, `tiktoken`), why counts are\n  estimates rather than billing-exact, when to trust grade/risk signals vs raw\n  numbers, and a worked example.\n\n## License\n\n[MIT](LICENSE) © Dwight-ship-it\n","readmeFilename":"README.md"}