{"_id":"tokensniff","_rev":"2-4e207cf80c5944ab9b77a2adfe12cbd7","name":"tokensniff","dist-tags":{"latest":"0.1.1"},"versions":{"0.1.0":{"name":"tokensniff","version":"0.1.0","keywords":["claude-code","claude code","claude-code-statusline","claude-code-hud","claude","anthropic","token-tracker","cost-tracker","telemetry","statusline","status-bar","hud","reverse-proxy","heatmap-dashboard","zero-dependency","TTFT","TPS","Turns","Input Tokens","Output Tokens","Cache %","Tool Calls","Context window","Gemini","Antigravity","proxy"],"license":"MIT","_id":"tokensniff@0.1.0","maintainers":[{"name":"neeeraj03","email":"neerajsahu6644@gmail.com"}],"homepage":"https://github.com/neeraj0304/tokensniff#readme","bugs":{"url":"https://github.com/neeraj0304/tokensniff/issues"},"bin":{"tokensniff":"bin/tokensniff.js","tokensniff-status":"bin/tokensniff-status.js"},"dist":{"shasum":"bcf61161bc770db2493bc84c5fe88c8f0242635c","tarball":"https://registry.npmjs.org/tokensniff/-/tokensniff-0.1.0.tgz","fileCount":39,"integrity":"sha512-/3vTCbRsLePCoLG0zkeAPL0reCZymNnvJOqgua4aSnPpiaC2g9K+lHwKL0lAYM4U8oA8PzCqJ/+7Xekk0uwNhw==","signatures":[{"sig":"MEUCIQDjLUfK6FeRus8M9JFB6uYPoudvFo1ju0fGxtRxGzQJogIgZ0vrfk0DbHOEKRgudVbf/frA7QBqSHGMXkOMbfMa8/Q=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":925915},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./schema":{"types":"./dist/shared/schema.d.ts","import":"./dist/schema.js"},"./pricing":{"types":"./dist/shared/pricing.d.ts","import":"./dist/pricing.js"},"./package.json":"./package.json"},"scripts":{"dev":"tsup --watch","lint":"biome check .","test":"tsx --test test/*.test.ts","build":"tsup && tsc --emitDeclarationOnly","prepack":"pnpm build","lint:fix":"biome check --write .","typecheck":"tsc --noEmit","check:package":"publint && attw --pack . --ignore-rules cjs-resolves-to-esm","prepublishOnly":"pnpm typecheck && pnpm test"},"_npmUser":{"name":"neeeraj03","email":"neerajsahu6644@gmail.com"},"repository":{"url":"git+https://github.com/neeraj0304/tokensniff.git","type":"git"},"_npmVersion":"11.2.0","description":"Claude Code telemetry proxy and live terminal status line HUD. Real-time token tracking, API costs, TTFT, and calendar heatmap dashboard. Zero dependencies.","directories":{},"sideEffects":false,"_nodeVersion":"22.17.1","publishConfig":{"access":"public"},"typesVersions":{"*":{"schema":["./dist/shared/schema.d.ts"],"pricing":["./dist/shared/pricing.d.ts"]}},"_hasShrinkwrap":false,"packageManager":"pnpm@11.25.0","devDependencies":{"tsx":"^4.23.13","tsup":"^8.5.1","publint":"^0.3.24","typescript":"^7.0.2","@types/node":"^26.5.0","@biomejs/biome":"^2.5.12","@arethetypeswrong/cli":"^0.18.5"},"_npmOperationalInternal":{"tmp":"tmp/tokensniff_0.1.0_1789286455372_0.937052266400525","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"tokensniff","version":"0.1.1","description":"Claude Code telemetry proxy and live terminal status line HUD. Real-time token tracking, API costs, TTFT, and calendar heatmap dashboard. Zero dependencies.","license":"MIT","type":"module","main":"./dist/index.js","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./schema":{"types":"./dist/shared/schema.d.ts","import":"./dist/schema.js"},"./pricing":{"types":"./dist/shared/pricing.d.ts","import":"./dist/pricing.js"},"./package.json":"./package.json"},"typesVersions":{"*":{"schema":["./dist/shared/schema.d.ts"],"pricing":["./dist/shared/pricing.d.ts"]}},"bin":{"tokensniff":"bin/tokensniff.js","tokensniff-status":"bin/tokensniff-status.js"},"sideEffects":false,"publishConfig":{"access":"public"},"engines":{"node":">=22.0.0"},"packageManager":"pnpm@11.25.0","scripts":{"build":"tsup && tsc --emitDeclarationOnly","dev":"tsup --watch","typecheck":"tsc --noEmit","lint":"biome check .","lint:fix":"biome check --write .","test":"tsx --test test/*.test.ts","check:package":"publint && attw --pack . --ignore-rules cjs-resolves-to-esm","prepack":"pnpm build","prepublishOnly":"pnpm typecheck && pnpm test"},"devDependencies":{"@arethetypeswrong/cli":"^0.18.5","@biomejs/biome":"^2.5.12","@types/node":"^26.5.0","publint":"^0.3.24","tsup":"^8.5.1","tsx":"^4.23.13","typescript":"^7.0.2"},"keywords":["claude-code","claude code","claude-code-statusline","claude-code-hud","claude","anthropic","token-tracker","cost-tracker","telemetry","statusline","status-bar","hud","reverse-proxy","heatmap-dashboard","zero-dependency","TTFT","TPS","Turns","Input Tokens","Output Tokens","Cache %","Tool Calls","Context window","Gemini","Antigravity","proxy"],"repository":{"type":"git","url":"git+https://github.com/neerajsahu0306/tokensniff.git"},"_id":"tokensniff@0.1.1","gitHead":"71fefd59c02eaef145f8e46403a793b5459021f1","bugs":{"url":"https://github.com/neerajsahu0306/tokensniff/issues"},"homepage":"https://github.com/neerajsahu0306/tokensniff#readme","_nodeVersion":"22.17.1","_npmVersion":"11.2.0","dist":{"integrity":"sha512-ISclKhPbRxWa/Uu06r4uDm2hdUPnkb1vwe7kxJTCRDW3+Bt5haFxGUhEOZ7jeXwX/cd7Qc+oxOUke74EGUDsdg==","shasum":"7825cc576f322962fee5fe79d68504b2ebd0eb56","tarball":"https://registry.npmjs.org/tokensniff/-/tokensniff-0.1.1.tgz","fileCount":39,"unpackedSize":925923,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIBlB7DzLcMkKZRAUFAjdw8DOlIUdBCGCgLbzoKmVrwCLAiB4dxt9hsXKaU+qAQcQ154PqoEjuHbhbCglg4MyqxvEyw=="}]},"_npmUser":{"name":"neeeraj03","email":"neerajsahu6644@gmail.com"},"directories":{},"maintainers":[{"name":"neeeraj03","email":"neerajsahu6644@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/tokensniff_0.1.1_1789290356312_0.6700947152016423"},"_hasShrinkwrap":false}},"time":{"created":"2026-09-13T08:00:55.125Z","modified":"2026-09-13T09:05:56.642Z","0.1.0":"2026-09-13T08:00:55.511Z","0.1.1":"2026-09-13T09:05:56.470Z"},"bugs":{"url":"https://github.com/neerajsahu0306/tokensniff/issues"},"license":"MIT","homepage":"https://github.com/neerajsahu0306/tokensniff#readme","keywords":["claude-code","claude code","claude-code-statusline","claude-code-hud","claude","anthropic","token-tracker","cost-tracker","telemetry","statusline","status-bar","hud","reverse-proxy","heatmap-dashboard","zero-dependency","TTFT","TPS","Turns","Input Tokens","Output Tokens","Cache %","Tool Calls","Context window","Gemini","Antigravity","proxy"],"repository":{"type":"git","url":"git+https://github.com/neerajsahu0306/tokensniff.git"},"description":"Claude Code telemetry proxy and live terminal status line HUD. Real-time token tracking, API costs, TTFT, and calendar heatmap dashboard. Zero dependencies.","maintainers":[{"name":"neeeraj03","email":"neerajsahu6644@gmail.com"}],"readme":"# tokensniff\n\n**Local telemetry sidecar for AI coding CLIs.**\n\nTrack every token, every turn, every dollar — live in your terminal and on a calendar heatmap dashboard. Zero runtime dependencies.\n\n[![npm version](https://img.shields.io/npm/v/tokensniff)](https://www.npmjs.com/package/tokensniff)\n[![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](./LICENSE)\n[![Node.js](https://img.shields.io/badge/node-%3E%3D22.0.0-brightgreen)](https://nodejs.org)\n\n---\n\n## What Is tokensniff?\n\ntokensniff is a **transparent HTTP reverse proxy and local telemetry sidecar** designed specifically for the [Claude Code](https://docs.anthropic.com/en/docs/claude-code) CLI harness running alongside the local [`antigravity-claude-proxy`](https://github.com/badri-s2001/antigravity-claude-proxy).\n\nBy default, tokensniff orchestrates and connects to `antigravity-claude-proxy` to route Claude Code API requests through Google Antigravity / Cloud Code, allowing you to utilize your Antigravity Gemini quota directly inside Claude Code.\n\n---\n\n### ⚠️ Important Notice: Use At Your Own Risk & Account Warning\n\n> [!CAUTION]\n> **Use at your own risk.** By default, tokensniff routes traffic through [`antigravity-claude-proxy`](https://github.com/badri-s2001/antigravity-claude-proxy), which accesses Google Antigravity / Cloud Code endpoints using unofficial proxying techniques.\n>\n> - **Risk of Account Suspension / Bans**: Google actively monitors and enforces Terms of Service (ToS) restrictions. Accounts connected to unofficial Cloud Code and Antigravity reverse proxies risk being **shadow-banned, quota-restricted, or permanently banned**.\n> - **No Guarantees & Zero Liability**: The creators and maintainers of `tokensniff` provide **no guarantees or warranties** of any kind, express or implied. We are **not responsible or liable** for any account bans, suspensions, quota penalties, data loss, or any other consequences resulting from the use of this software.\n> - **Safety Recommendation**: **Do not use your primary, personal, or corporate Google account.** If you choose to use this integration, use a dedicated burner/disposable Google account.\n> - **Learn More**: Read the upstream [antigravity-claude-proxy GitHub Repository](https://github.com/badri-s2001/antigravity-claude-proxy) and its [Safety, Usage, and Risk Notices](https://github.com/badri-s2001/antigravity-claude-proxy#readme) to understand how the underlying authentication and proxying operate.\n\n---\n\nIt captures **every API turn** flowing through it and gives you:\n\n- **Per-turn token counts** — input, output, cache-read, cache-write, thinking tokens\n- **Real-time cost tracking** — per turn, per session, per day (in USD)\n- **Performance metrics** — Time to First Token (TTFT), tokens per second (TPS), turn duration\n- **Live terminal status line** — a single-line HUD rendered directly inside Claude Code's status bar\n- **Calendar heatmap dashboard** — a beautiful, self-contained HTML page showing your daily token spend over the entire year\n- **Multi-terminal session management** — run multiple terminals against the same proxy without conflicts\n- **Tool call tracking** — which tools the model called, how many times, argument token sizes\n- **Context window monitoring** — how full your context is, with color-coded warnings (green → yellow → red)\n\nAll of this happens **transparently**. Your AI tool doesn't know tokensniff exists. It just thinks it's talking to the normal API. tokensniff forwards every byte with zero latency overhead, while quietly recording the telemetry.\n\n---\n\n## How It Works\n\n```mermaid\nflowchart TD\n    CLI[\"Claude Code CLI<br/><code>ANTHROPIC_BASE_URL:4000</code>\"]\n    TS[\"tokensniff<br/><code>Reverse Proxy :4000</code>\"]\n    Proxy[\"antigravity-claude-proxy<br/><code>Local Upstream :8085</code>\"]\n    Google[\"Google Cloud Code API<br/><code>Gemini Backend / Quota</code>\"]\n\n    subgraph Telemetry[\"Telemetry Engine\"]\n        T1[\"latest.json\"]\n        T2[\"totals.json\"]\n        T3[\"history.ndjson\"]\n        T4[\"/dashboard (HTML)\"]\n    end\n\n    Statusline[\"Claude Code Statusline HUD\"]\n    Dashboard[\"Live Calendar Heatmap<br/><code>http://localhost:4000/dashboard</code>\"]\n\n    CLI <-->|\"1. Anthropic API Requests / Streams\"| TS\n    TS <-->|\"2. Forward\"| Proxy\n    TS -->|\"3. Telemetry (Async)\"| Telemetry\n    Proxy <-->|\"4. Protocol Translation\"| Google\n\n    T1 --> Statusline\n    T4 --> Dashboard\n```\n\n1. **Harness Redirection**: You point Claude Code to `http://localhost:4000` by setting `ANTHROPIC_BASE_URL` in `~/.claude/settings.json`. Claude Code continues sending standard Anthropic Messages API requests normally.\n2. **Upstream Proxy Startup**: When launched, tokensniff checks port 8085 and automatically starts the local upstream proxy command (`npx antigravity-claude-proxy@latest start`) if it is not already running.\n3. **Protocol & Quota Translation**: The local [`antigravity-claude-proxy`](https://github.com/badri-s2001/antigravity-claude-proxy) receives the Anthropic-formatted request, authenticates with your Google Antigravity / Cloud Code OAuth credentials, translates the schema to Google Generative AI format, and submits it to Google Cloud Code's backend to consume your Antigravity Gemini quota.\n4. **Streaming Response Pass-Through**: As Google's response streams back, `antigravity-claude-proxy` transforms it into Anthropic-compatible SSE events or buffered JSON. tokensniff instantly relays every raw byte back to Claude Code (`res.write(chunk)`) with zero latency overhead.\n5. **Telemetry Extraction**: Concurrently and non-blockingly, tokensniff's parser inspects the response payload. It extracts Google's `usageMetadata` (input, output, cache-read, cache-write, and thinking/reasoning tokens), measures Time to First Token (TTFT) and token generation speed (TPS), and tracks turn duration.\n6. **Live Pricing Calculation**: tokensniff queries live pricing from [OpenRouter's model catalog](https://openrouter.ai), computing turn cost, session cumulative spend, and daily totals.\n7. **Statusline HUD & Calendar Heatmap**: Telemetry snapshots are written atomically to disk (`latest.json`, `totals.json`, `history.ndjson`). Claude Code's status bar runs `tokensniff-status` to display the live single-line HUD, and tokensniff serves an interactive calendar heatmap dashboard at `http://localhost:4000/dashboard`.\n\n---\n\n## Quick Start\n\n### 1. Install\n\n```bash\nnpm install -g tokensniff\n```\n\nOr with pnpm:\n\n```bash\npnpm add -g tokensniff\n```\n\n### 2. Initialize Configuration\n\n```bash\ntokensniff init\n```\n\nThis creates a configuration file at `~/.tokensniff/config.json` with sensible defaults:\n\n```json\n{\n  \"upstreamCommand\": \"npx antigravity-claude-proxy@latest start\",\n  \"upstreamHost\": \"localhost\",\n  \"upstreamPort\": 8085,\n  \"listenPort\": 4000,\n  \"harnessCommand\": \"claude\"\n}\n```\n\n> **Default Upstream:** Out of the box, tokensniff is pre-configured to launch [`antigravity-claude-proxy`](https://github.com/badri-s2001/antigravity-claude-proxy) on port `8085`, allowing you to route Claude Code prompts through your Antigravity Gemini quota. Please review the [Risk & Terms of Service Warning](#️-important-notice-use-at-your-own-risk--account-warning) before running with this configuration.\n>\n\n\n### 3. Configure Claude Code\n\nAdd the following to your **`~/.claude/settings.json`** file:\n\n```json\n{\n  \"env\": {\n    \"ANTHROPIC_BASE_URL\": \"http://localhost:4000\"\n  },\n  \"statusLine\": {\n    \"type\": \"command\",\n    \"command\": \"tokensniff-status\",\n    \"padding\": 0\n  }\n}\n```\n\nThis does two things:\n\n- **`ANTHROPIC_BASE_URL`** — Tells Claude Code to send all API requests through tokensniff's proxy on port 4000 instead of directly to Anthropic\n- **`statusLine`** — Tells Claude Code to run `tokensniff-status` and display its output as a live status bar at the bottom of the terminal\n\n### 4. Run\n\n```bash\ntokensniff\n```\n\nThat's it. tokensniff will:\n\n1. Start the upstream proxy (if configured)\n2. Start the telemetry proxy on port 4000\n3. Launch Claude Code (or whatever harness command you configured)\n4. Show you a live status line and dashboard URL\n\nOpen the dashboard in your browser:\n\n```\nhttp://localhost:4000/dashboard\n```\n\n---\n\n## CLI Commands\n\n### `tokensniff`\n\nRuns the full telemetry pipeline: starts the upstream proxy, starts the tokensniff proxy, and launches your AI coding tool.\n\n```bash\ntokensniff\n```\n\n### `tokensniff init`\n\nGenerates the global configuration file at `~/.tokensniff/config.json` and prints the Claude Code settings.json instructions.\n\n```bash\ntokensniff init\n```\n\n### `tokensniff dashboard`\n\nPrints the URL of the live calendar heatmap dashboard.\n\n```bash\ntokensniff dashboard\n# Output: [tokensniff] live heatmap dashboard: http://127.0.0.1:4000/dashboard\n```\n\n### `tokensniff-status`\n\nThis is the statusline renderer. You don't run this directly — Claude Code runs it automatically via the `statusLine` setting. It reads the latest telemetry snapshot from disk and outputs a formatted single-line status bar.\n\nIf you want to test it manually:\n\n```bash\necho '{}' | tokensniff-status\n```\n\nTo get raw JSON output instead of the formatted status line:\n\n```bash\nTOKENSNIFF_JSON=1 echo '{}' | tokensniff-status\n```\n\n---\n\n## The Status Line\n\nWhen running inside Claude Code, you'll see a live status line at the bottom of your terminal that looks like this:\n\n**Wide terminals (≥ 180 columns) — single line:**\n\n```\nturn 3 [3.8-flash] | ctx: 38.5k/1M (3.9%) [cache: 25k (64.9%)] | input: +4.5k tok, output: 42 tok, think: 180 tok | ttft: 1.2s, 110 tok/s | cost: turn $0.0042, sess $0.058, today $0.245\n```\n\n**Standard terminals (< 180 columns) — clean 2-line stack:**\n\n```\nturn 3 [3.8-flash] | ctx: 38.5k/1M (3.9%) [cache: 25k (64.9%)]\ninput: +4.5k tok, output: 42 tok, think: 180 tok | ttft: 1.2s, 110 tok/s | cost: turn $0.0042, sess $0.058, today $0.245\n```\n\n### What Each Segment Means\n\n| Segment | Example | Meaning |\n|---------|---------|---------|\n| **Turn** | `turn 3` | Which turn number this is in the current session |\n| **Model** | `[3.8-flash]` | The AI model being used (vendor prefix stripped for readability) |\n| **[bg]** | `[bg]` | Shown when this is a background agent turn (via `x-app: cli-bg` header) |\n| **Context** | `ctx: 38.5k/1M (3.9%)` | Current context usage / max window size (percentage full) |\n| **Cache** | `[cache: 25k (64.9%)]` | How many tokens were served from cache and the cache hit ratio |\n| **Input** | `input: +4.5k tok` | New tokens added to context this turn (delta). Shows `freed` when context shrinks |\n| **Output** | `output: 42 tok` | Tokens generated by the model this turn |\n| **Thinking** | `think: 180 tok` | Reasoning/thinking tokens used (for models with extended thinking) |\n| **Tool** | `tool: Read (45 tok)` | Which tool was called and how many argument tokens it used |\n| **TTFT** | `ttft: 1.2s` | Time to First Token — how long before the model started generating |\n| **Speed** | `110 tok/s` | Generation velocity in tokens per second |\n| **Turn Cost** | `turn $0.0042` | How much this specific turn cost in USD |\n| **Session Cost** | `sess $0.058` | Total spend for this entire session |\n| **Today Cost** | `today $0.245` | Total spend across all sessions today (resets at local midnight) |\n| **[idle]** | `[idle]` | Shown when the last turn was more than 120 seconds ago |\n\n### Color Coding\n\nThe context percentage is color-coded based on how full your context window is:\n\n- 🟢 **Green** — Under 60% (plenty of room)\n- 🟡 **Yellow** — 60-80% (getting full, consider starting a new session)\n- 🔴 **Red** — Over 80% (context is nearly full, model may start forgetting earlier context)\n\nThese thresholds are configurable via `warnPct` and `critPct`.\n\n---\n\n## The Dashboard\n\ntokensniff serves a **live calendar heatmap dashboard** directly on the proxy port. Open it in any browser:\n\n```\nhttp://localhost:4000/dashboard\n```\n\nThe dashboard shows:\n\n- **Total Spend** — cumulative USD spent across all sessions\n- **Total Tokens** — cumulative token volume processed\n- **Cache Ratio** — what percentage of tokens were served from cache\n- **Total Turns** — how many API turns have been recorded\n- **Calendar Heatmap** — a GitHub-contributions-style grid showing daily token volume across the year\n\n### Heatmap Tiers\n\nThe calendar tiles are colored using a 5-tier emerald luminosity scale:\n\n| Tier | Daily Volume | Color |\n|------|-------------|-------|\n| 0 | No activity | Dark (nearly invisible) |\n| 1 | < 5M tokens | Dark emerald |\n| 2 | 5M – 25M tokens | Medium emerald |\n| 3 | 25M – 100M tokens | Bright emerald |\n| 4 | > 100M tokens | Vivid emerald (glowing) |\n\nHover over any tile to see a detailed tooltip with exact token counts, cost, cache leverage, and turns logged for that day.\n\n### Dashboard API\n\nThere's also a JSON API endpoint for programmatic access:\n\n```bash\ncurl http://localhost:4000/api/daily\n```\n\nReturns the raw daily rollup data as JSON.\n\n---\n\n## Configuration\n\ntokensniff uses a strict hierarchical configuration system:\n\n```\nEnvironment Variables  >  Config File  >  Embedded Defaults\n          (highest)                          (medium)                   (lowest)\n```\n\n### Config File Location\n\n```\n~/.tokensniff/config.json\n```\n\nYou can override this path with the `TOKENSNIFF_CONFIG` environment variable:\n\n```bash\nTOKENSNIFF_CONFIG=/path/to/custom/config.json tokensniff\n```\n\n### All Configuration Options\n\n| Config Key | Env Variable | Default | Description |\n|-----------|-------------|---------|-------------|\n| `listenPort` | `TOKENSNIFF_PORT` | `4000` | Port the tokensniff proxy listens on |\n| `listenHost` | `TOKENSNIFF_HOST` | `127.0.0.1` | Host/IP the proxy binds to |\n| `upstreamHost` | `TOKENSNIFF_UPSTREAM_HOST` | `localhost` | Hostname of the upstream API server |\n| `upstreamPort` | `TOKENSNIFF_UPSTREAM_PORT` | `8085` | Port of the upstream API server |\n| `upstreamTimeoutMs` | `TOKENSNIFF_UPSTREAM_TIMEOUT_MS` | `300000` (5 min) | Timeout for upstream requests in milliseconds (range: 1,000 – 3,600,000) |\n| `maxBodyBytes` | `TOKENSNIFF_MAX_BODY_BYTES` | `10485760` (10 MB) | Maximum request/response body size. Requests exceeding this get a 413 error (range: 1,024 – 104,857,600) |\n| `statusDirs` | `TOKENSNIFF_STATUS_DIRS` | `~/.tokensniff/status` | Comma-separated list of directories where telemetry snapshots are written |\n| `maxSessions` | `TOKENSNIFF_MAX_SESSIONS` | `50` | Maximum number of tracked sessions before oldest are evicted (range: 1 – 1,000) |\n| `deadSessionMs` | `TOKENSNIFF_DEAD_SESSION_MS` | `259200000` (72 hrs) | How long to keep stale session files before pruning (range: 1 hour minimum) |\n| `staleAfterS` | `TOKENSNIFF_STALE_AFTER_S` | `120` | Seconds of inactivity before a session shows `[idle]` in the status line (range: 5 – 86,400) |\n| `labelMaxIn` | `TOKENSNIFF_LABEL_MAX_IN` | `2000` | Maximum input tokens for a turn to be classified as a micro-label |\n| `labelMaxOut` | `TOKENSNIFF_LABEL_MAX_OUT` | `60` | Maximum output tokens for a turn to be classified as a micro-label |\n| `warnPct` | `TOKENSNIFF_WARN_PCT` | `60` | Context utilization % threshold for yellow warning color (range: 1 – 99) |\n| `critPct` | `TOKENSNIFF_CRIT_PCT` | `80` | Context utilization % threshold for red critical color (range: 2 – 100) |\n| `color` | `TOKENSNIFF_COLOR` | `true` | Enable/disable ANSI color output. Automatically disabled when `NO_COLOR` env var is set (per [no-color.org](https://no-color.org/)) |\n| `upstreamCommand` | `TOKENSNIFF_UPSTREAM_CMD` | `npx antigravity-claude-proxy@latest start` | Shell command to start the upstream proxy server |\n| `upstreamStopCommand` | `TOKENSNIFF_UPSTREAM_STOP_CMD` | _(empty)_ | Shell command to stop the upstream proxy on shutdown |\n| `harnessCommand` | `TOKENSNIFF_HARNESS_CMD` | `claude` | The AI coding tool to launch (e.g., `claude`, `codex`, or any executable) |\n\n### Example Config File\n\n```json\n{\n  \"listenPort\": 4000,\n  \"listenHost\": \"127.0.0.1\",\n  \"upstreamHost\": \"localhost\",\n  \"upstreamPort\": 8085,\n  \"upstreamCommand\": \"npx antigravity-claude-proxy@latest start\",\n  \"upstreamStopCommand\": \"\",\n  \"harnessCommand\": \"claude\",\n  \"maxSessions\": 50,\n  \"staleAfterS\": 120,\n  \"warnPct\": 60,\n  \"critPct\": 80,\n  \"color\": true\n}\n```\n\n### Environment Variable Examples\n\n```bash\n# Change the proxy port\nTOKENSNIFF_PORT=5000 tokensniff\n\n# Point to a different upstream\nTOKENSNIFF_UPSTREAM_HOST=api.anthropic.com TOKENSNIFF_UPSTREAM_PORT=443 tokensniff\n\n# Disable colors\nNO_COLOR=1 tokensniff\n\n# Use a custom config file\nTOKENSNIFF_CONFIG=./my-config.json tokensniff\n\n# Write status to multiple directories\nTOKENSNIFF_STATUS_DIRS=\"/path/a,/path/b\" tokensniff\n```\n\n---\n\n## Multi-Terminal Support\n\ntokensniff supports **multiple terminal sessions** running simultaneously against the same proxy. This is how it works:\n\n1. When you run `tokensniff`, it first checks if a proxy is **already running** on port 4000\n2. If yes, it **attaches** to the existing proxy (no duplicate servers) and just launches your harness\n3. Each terminal session registers itself in `~/.tokensniff/sessions/` with a PID lock file\n4. When you close a terminal (Ctrl+C or exit):\n   - If **other terminals are still active** → only the local harness is terminated; the proxy stays alive\n   - If **this was the last terminal** → the proxy and upstream are cleanly shut down\n5. Dead PID lock files (from crashed terminals) are automatically pruned\n\nThis means you can have 5 Claude Code windows all routing through the same tokensniff proxy, and the telemetry stays unified. Session costs are tracked independently, but `today` cost accumulates across all sessions.\n\n---\n\n## Supported Providers\n\ntokensniff's parser understands multiple API response formats:\n\n| Provider | Format | Detection |\n|----------|--------|-----------|\n| **Anthropic** (Claude) | SSE streams (`text/event-stream`) | `event: message_start` framing |\n| **Anthropic** (Claude) | Buffered JSON (`application/json`) | `usage.input_tokens` / `usage.output_tokens` |\n| **Google Gemini** (Antigravity) | Buffered JSON | `candidates[].content.parts[]` + `usageMetadata` |\n\n\nThe parser automatically detects the format from the response content-type and payload structure. You don't need to configure anything.\n\n### Pricing\n\ntokensniff fetches live per-token pricing from [OpenRouter's public model catalog](https://openrouter.ai/api/v1/models). This means:\n\n- Pricing is **always up to date** — no hardcoded rate tables to maintain\n- **Any model** listed on OpenRouter is automatically priced correctly\n- Pricing includes prompt, completion, cache-read, and cache-write tiers\n- Context window sizes are also pulled dynamically from the catalog\n\nIf a model isn't found on OpenRouter, costs default to `$0.00` (zero-rate boundary) rather than guessing wrong. The context window defaults to 128K tokens for unknown models.\n\nThe pricing cache is populated on-demand (first turn using a new model triggers a background fetch) and persists in memory for the lifetime of the proxy process.\n\n---\n\n## Telemetry Data Files\n\ntokensniff stores all telemetry in `~/.tokensniff/status/`:\n\n| File | Format | Purpose |\n|------|--------|---------|\n| `latest.json` | JSON | Most recent telemetry snapshot (any session) |\n| `latest-<session_id>.json` | JSON | Most recent snapshot for a specific session |\n| `totals.json` | JSON | Cumulative spend per session |\n| `history.ndjson` | Newline-delimited JSON | Append-only turn-by-turn ledger (feeds the heatmap) |\n\n### latest.json Schema (v1)\n\n```json\n{\n  \"v\": 1,\n  \"session_id\": \"abc-123\",\n  \"turn_index\": 5,\n  \"ts\": 1726056000000,\n  \"model\": \"gemini-3.8-flash-tiered\",\n  \"is_bg\": false,\n  \"ctx\": 38500,\n  \"window\": 1000000,\n  \"pct\": 3.9,\n  \"cache_read\": 25000,\n  \"cache_create\": 0,\n  \"delta_in\": 4500,\n  \"in_tokens\": 13500,\n  \"out_tokens\": 42,\n  \"thinking\": true,\n  \"thinking_tokens\": 180,\n  \"tool\": \"Read\",\n  \"tool_tk\": 45,\n  \"tools_summary\": [{ \"name\": \"Read\", \"count\": 1, \"arg_tk\": 45 }],\n  \"stop_reason\": \"end_turn\",\n  \"is_label\": false,\n  \"label\": \"\",\n  \"ttft_ms\": 1200,\n  \"tps\": 110,\n  \"dur_s\": 1.6,\n  \"cost_turn\": 0.0042,\n  \"cost_session\": 0.058,\n  \"cost_today\": 0.245,\n  \"error\": null,\n  \"status\": 200,\n  \"quota_pct\": null,\n  \"quota_reset\": null,\n  \"tier\": null\n}\n```\n\n### history.ndjson Record Format\n\nEach line is a compact JSON object:\n\n```json\n{\"d\":\"2026-09-11\",\"ts\":1726056000000,\"s\":\"abc-123\",\"t\":5,\"m\":\"gemini-3.8-flash\",\"tk\":38542,\"c\":0.0042,\"cr\":25000,\"th\":180}\n```\n\n| Field | Meaning |\n|-------|---------|\n| `d` | Local calendar date (YYYY-MM-DD) |\n| `ts` | Unix timestamp in milliseconds |\n| `s` | Session ID |\n| `t` | Turn number |\n| `m` | Model name |\n| `tk` | Total tokens (context + output) |\n| `c` | Cost in USD |\n| `cr` | Cache-read tokens |\n| `th` | Thinking tokens |\n\n---\n\n## Programmatic API\n\ntokensniff exports its entire engine as a library. You can use it in your own Node.js projects:\n\n```bash\nnpm install tokensniff\n```\n\n### Start the Proxy Programmatically\n\n```js\nimport { startProxy } from 'tokensniff';\n\nconst server = startProxy({\n  listenPort: 4000,\n  listenHost: '127.0.0.1',\n  upstreamHost: 'localhost',\n  upstreamPort: 8085,\n});\n\n// server is a standard Node.js http.Server\nserver.on('listening', () => {\n  console.log('tokensniff proxy is running');\n});\n```\n\n### Calculate Costs\n\n```js\nimport { costFor, ratesFor, resolveModelRates } from 'tokensniff/pricing';\n\n// Synchronous (from cache, or zero if not yet fetched)\nconst rates = ratesFor('claude-3-7-sonnet');\n\n// Async (fetches from OpenRouter if needed)\nconst rates2 = await resolveModelRates('gemini-2.5-pro');\n\n// Calculate cost\nconst cost = costFor(rates, {\n  input: 10000,\n  output: 500,\n  cacheRead: 8000,\n  cacheCreate: 0,\n});\n\nconsole.log(`Turn cost: $${cost.toFixed(4)}`);\n```\n\n### Use the Schema\n\n```js\nimport { buildLatest, isTokenSniffLatest, SCHEMA_VERSION } from 'tokensniff/schema';\n\n// Build a telemetry snapshot with safe defaults\nconst snapshot = buildLatest({\n  session_id: 'my-session',\n  model: 'gemini-3.8-flash',\n  ctx: 25000,\n  window: 1000000,\n  pct: 2.5,\n});\n\n// Validate unknown data\nif (isTokenSniffLatest(someData)) {\n  console.log('Valid telemetry snapshot');\n}\n```\n\n### Full API Exports\n\n```js\nimport {\n  // CLI Orchestrator\n  runCli,\n  countActiveSessions,\n  getSessionsDir,\n  isPortActive,\n  isProcessAlive,\n  registerSession,\n  waitForTcp,\n  writeInitFile,\n  printClaudeInstructions,\n\n  // Configuration\n  loadConfig,\n  getGlobalDir,\n  getGlobalConfigPath,\n\n  // Proxy Server\n  startProxy,\n\n  // Dashboard & Analytics\n  importCapturesDirectory,\n  loadDailyRollup,\n  renderHeatmapHtml,\n\n  // Statusline Renderer\n  formatStatus,\n  formatTokenCount,\n  formatToolSegment,\n  runStatusRenderer,\n  readStdin,\n  pickLatest,\n  extractModel,\n\n  // Pricing Engine\n  costFor,\n  ratesFor,\n  resolveModelRates,\n  normalizeModelName,\n  syncOpenRouterCatalog,\n\n  // Schema & Types\n  buildLatest,\n  isTokenSniffLatest,\n  extractCleanLabel,\n  SCHEMA_VERSION,\n} from 'tokensniff';\n```\n\n### Sub-path Exports\n\n```js\n// Just the schema types and guards\nimport { buildLatest, isTokenSniffLatest } from 'tokensniff/schema';\n\n// Just the pricing engine\nimport { costFor, resolveModelRates } from 'tokensniff/pricing';\n```\n\n---\n\n## Architecture\n\n```\ntokensniff/\n├── bin/\n│   ├── tokensniff.js          # CLI entrypoint → dist/cli.js\n│   └── tokensniff-status.js   # Statusline entrypoint → dist/status.js\n├── src/\n│   ├── index.ts               # Public API facade (re-exports everything)\n│   ├── cli/\n│   │   └── run.ts             # Master CLI orchestrator & multi-terminal lifecycle\n│   ├── collector/\n│   │   ├── config.ts          # Hierarchical config loader (env > file > defaults)\n│   │   ├── index.ts           # HTTP reverse proxy server & telemetry capture\n│   │   ├── parse.ts           # SSE stream & JSON payload parser (multi-provider)\n│   │   └── store.ts           # Atomic file persistence engine (Windows-safe)\n│   ├── dashboard/\n│   │   └── heatmap.ts         # Calendar heatmap HTML renderer & capture importer\n│   ├── renderer/\n│   │   ├── format.ts          # Responsive statusline formatter\n│   │   └── index.ts           # Statusline CLI renderer (stdin consumer)\n│   └── shared/\n│       ├── pricing.ts         # Dynamic pricing engine (OpenRouter catalog sync)\n│       └── schema.ts          # Domain schemas, type guards, label extraction\n└── test/\n    ├── fixtures.ts            # Synthetic SSE/JSON payload generators\n    ├── collector.test.ts      # Integration: proxy, streaming, 502, 413, dashboard\n    ├── format.test.ts         # Statusline formatting & responsive layouts\n    ├── heatmap.test.ts        # Daily rollups, capture import, HTML rendering\n    ├── parse.test.ts          # SSE/JSON parsing, multi-provider, tool grouping\n    ├── pricing.test.ts        # Cost math, OpenRouter sync, model normalization\n    ├── run.test.ts            # CLI routing, TCP probing, session lifecycle\n    ├── schema.test.ts         # Type guards, buildLatest, label extraction\n    └── store.test.ts          # Atomic writes, totals, history, pruning, config\n```\n\n### Key Design Decisions\n\n- **Zero runtime dependencies** — The entire package uses only Node.js built-in modules (`http`, `fs`, `net`, `path`, `os`, `child_process`). No Express, no Axios, no anything. This keeps the install tiny and avoids supply chain risk.\n\n- **Atomic file writes** — All file persistence uses a temp-file + OS rename pattern. This means readers (the statusline renderer) never see a half-written JSON file. On Windows, retries with exponential backoff handle EPERM/EBUSY file lock contention.\n\n- **Streaming-first proxy** — Response bytes are forwarded to the client as they arrive (`res.write(chunk)`). tokensniff never buffers the full response before forwarding. This means zero latency overhead. The telemetry parsing happens on the buffered copy.\n\n- **LRU-bounded memory** — Internal maps (context history, turn indices) are capped at 500 entries using LRU eviction. The proxy can run for weeks without leaking memory.\n\n- **Defensive parsing** — The parser never throws. Corrupt payloads, truncated SSE streams, binary noise — everything returns a safe fallback result. This is critical because the proxy sits in the hot path of your AI tool.\n\n---\n\n## Development\n\n### Prerequisites\n\n- Node.js ≥ 22.0.0\n- pnpm 11.x\n\n### Setup\n\n```bash\ngit clone https://github.com/neerajsahu0306/tokensniff.git\ncd tokensniff\npnpm install\n```\n\n### Build\n\n```bash\npnpm build\n```\n\n### Run Tests\n\n```bash\npnpm test\n```\n\n### Type Check\n\n```bash\npnpm typecheck\n```\n\n### Lint\n\n```bash\npnpm lint\n```\n\n### Auto-fix Lint Issues\n\n```bash\npnpm lint:fix\n```\n\n### Watch Mode (Development)\n\n```bash\npnpm dev\n```\n\n### Validate Package Structure\n\n```bash\npnpm check:package\n```\n\n---\n\n## Troubleshooting\n\n### \"waiting for first turn...\"\n\nThe status line shows this message when tokensniff hasn't received any API requests yet. Make sure:\n\n1. Your `ANTHROPIC_BASE_URL` is set to `http://localhost:4000` in `~/.claude/settings.json`\n2. The tokensniff proxy is actually running (check terminal output)\n3. You've made at least one prompt in Claude Code\n\n### Port 4000 is already in use\n\nAnother tokensniff instance (or another program) is using port 4000. Either:\n\n- Let tokensniff attach to it (it will do this automatically if the existing proxy is tokensniff)\n- Change the port: `TOKENSNIFF_PORT=5000 tokensniff`\n- Kill the existing process: find and terminate whatever is using port 4000\n\n### Upstream proxy failed health check\n\ntokensniff couldn't connect to the upstream API server. Check that:\n\n1. Your `upstreamCommand` is valid and the upstream server starts correctly\n2. The `upstreamHost` and `upstreamPort` match where the upstream is listening\n3. The upstream server is not firewalled or blocked\n\ntokensniff will continue running even if the upstream health check fails — downstream requests will get 502 errors until the upstream becomes available.\n\n### Costs showing $0.0000\n\nThis means tokensniff couldn't find pricing for the model you're using on OpenRouter. This can happen if:\n\n- The model is brand new and not yet listed on OpenRouter\n- The OpenRouter API was unreachable when tokensniff tried to fetch pricing\n- You're using a custom/private model that isn't publicly listed\n\nThe proxy and telemetry still work perfectly — only the cost calculation defaults to zero.\n\n### Status line not appearing in Claude Code\n\nMake sure your `~/.claude/settings.json` has the exact `statusLine` block:\n\n```json\n{\n  \"statusLine\": {\n    \"type\": \"command\",\n    \"command\": \"tokensniff-status\",\n    \"padding\": 0\n  }\n}\n```\n\nAlso verify that `tokensniff-status` is accessible in your PATH (it should be if you installed tokensniff globally).\n\n---\n\n## Requirements\n\n- **Node.js** ≥ 22.0.0\n- **OS**: Windows, macOS, or Linux\n- **Terminal**: Any terminal that supports ANSI colors (for the status line color coding)\n\n---\n\n## License\n\n[MIT](./LICENSE) © 2026 tokensniff contributors\n","readmeFilename":"README.md"}