{"_id":"@advik1228/llm-cost-guard","_rev":"3-ab307ef7a735e9502f8289b9ccfcc5da","name":"@advik1228/llm-cost-guard","dist-tags":{"latest":"0.1.2"},"versions":{"0.1.0":{"name":"@advik1228/llm-cost-guard","version":"0.1.0","license":"MIT","_id":"@advik1228/llm-cost-guard@0.1.0","maintainers":[{"name":"advik1228","email":"advik.hingmire12@gmail.com"}],"dist":{"shasum":"bc70bea21d2e230b78e9e2810510ddea42bf3f58","tarball":"https://registry.npmjs.org/@advik1228/llm-cost-guard/-/llm-cost-guard-0.1.0.tgz","fileCount":38,"integrity":"sha512-0YKWSYIhZQNuU1ztBb+SuSOZRKCBwrgZ8ZUhU5zEF3zH8CWwP/nsjae5xdNC9P7JtWT++K13Aebo3UhFGlqHbg==","signatures":[{"sig":"MEYCIQCbzl+2R2aQht1dHIyWZ2E8VKelk8wE+ZJsiRot2Me5fAIhAJlGFfelCyIk0eYfSdhtq2FnW5foH7ZFdV8IeFmUNP9H","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":60791},"main":"dist/index.js","types":"dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.js"},"./adapters":{"types":"./dist/adapters/redis.d.ts","require":"./dist/adapters/redis.js"}},"gitHead":"559d7628bd5f193c35edc969723d20d244dfbfbd","scripts":{"dev":"tsc --watch","test":"jest","build":"tsc","prepublishOnly":"npm run build && npm test"},"_npmUser":{"name":"advik1228","email":"advik.hingmire12@gmail.com"},"_npmVersion":"11.12.1","directories":{},"_nodeVersion":"24.15.0","dependencies":{"redis":"^4.7.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"jest":"^29.0.0","ts-jest":"^29.0.0","typescript":"^5.4.0","@types/jest":"^29.0.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/llm-cost-guard_0.1.0_1781339742390_0.04602028060774721","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@advik1228/llm-cost-guard","version":"0.1.1","keywords":["llm","openai","anthropic","claude","cost","budget","middleware","token","api","guard"],"license":"MIT","_id":"@advik1228/llm-cost-guard@0.1.1","maintainers":[{"name":"advik1228","email":"advik.hingmire12@gmail.com"}],"dist":{"shasum":"5474721f870ffcda4cd0ba0cea8544a1de16cc59","tarball":"https://registry.npmjs.org/@advik1228/llm-cost-guard/-/llm-cost-guard-0.1.1.tgz","fileCount":39,"integrity":"sha512-A4L1qkzadxJTK2dbGamkVE6vpXNQDHGJ5e82JeEwtckFhoOI3pc13CpixGvlVD5/rvTgZfFNNNb4ni5EDZ2UWw==","signatures":[{"sig":"MEUCIEp/KbK2roRgmRQqbyH6NEQ3jLAuTzWmbQiOJjc9f6rqAiEAz2hvdoiuqrsqLV2eZUwidV0lvfMVqutCmaxtOmpFmgc=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":72986},"main":"dist/index.js","types":"dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.js"},"./adapters":{"types":"./dist/adapters/redis.d.ts","require":"./dist/adapters/redis.js"}},"gitHead":"c56e7d1ab1913776af0212efcd4c23996c220b46","scripts":{"dev":"tsc --watch","test":"jest","build":"tsc","prepublishOnly":"npm run build && npm test"},"_npmUser":{"name":"advik1228","email":"advik.hingmire12@gmail.com"},"_npmVersion":"11.12.1","description":"[![npm version](https://img.shields.io/npm/v/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228/llm-cost-guard)\r [![npm downloads](https://img.shields.io/npm/dm/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228","directories":{},"_nodeVersion":"24.15.0","dependencies":{"redis":"^4.7.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"jest":"^29.0.0","ts-jest":"^29.0.0","typescript":"^5.4.0","@types/jest":"^29.0.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/llm-cost-guard_0.1.1_1781340736910_0.9985162498731421","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@advik1228/llm-cost-guard","version":"0.1.2","main":"dist/index.js","types":"dist/index.d.ts","exports":{".":{"require":"./dist/index.js","import":"./dist/index.js","types":"./dist/index.d.ts"},"./adapters":{"require":"./dist/adapters/redis.js","types":"./dist/adapters/redis.d.ts"}},"publishConfig":{"access":"public"},"scripts":{"build":"tsc","test":"jest","dev":"tsc --watch","prepublishOnly":"npm run build && npm test"},"dependencies":{"redis":"^4.7.0"},"devDependencies":{"typescript":"^5.4.0","@types/node":"^20.0.0","jest":"^29.0.0","ts-jest":"^29.0.0","@types/jest":"^29.0.0"},"license":"MIT","keywords":["llm","openai","anthropic","claude","cost","budget","middleware","token","api","guard"],"gitHead":"2e61c9c8dba471d44f7e1429932e5cd41ba06164","_id":"@advik1228/llm-cost-guard@0.1.2","description":"[![npm version](https://img.shields.io/npm/v/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228/llm-cost-guard)\r [![npm downloads](https://img.shields.io/npm/dm/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228","_nodeVersion":"24.15.0","_npmVersion":"11.12.1","dist":{"integrity":"sha512-VPe04d3CkRqtH1mrennbusjgFnl2yYKW+bADqlNGEIvCo8twQvgw2whJ4zgJKYqjH0uLepXtfsEKFgVmeaAkzw==","shasum":"d5abe48a12edee9b2ab0db30aeb21644dd38f0a8","tarball":"https://registry.npmjs.org/@advik1228/llm-cost-guard/-/llm-cost-guard-0.1.2.tgz","fileCount":39,"unpackedSize":77315,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIGAzNYRzjUMKu6/SP38O6BwKw6XhZ20mNHRK6Q2I+JJPAiEAzgTk876LydmupYrRFgHLhZjCz51+4foOYjAslXqPuFw="}]},"_npmUser":{"name":"advik1228","email":"advik.hingmire12@gmail.com"},"directories":{},"maintainers":[{"name":"advik1228","email":"advik.hingmire12@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/llm-cost-guard_0.1.2_1781341205926_0.2217011628903538"},"_hasShrinkwrap":false}},"time":{"created":"2026-06-13T08:35:42.237Z","modified":"2026-06-13T09:00:06.240Z","0.1.0":"2026-06-13T08:35:42.514Z","0.1.1":"2026-06-13T08:52:17.047Z","0.1.2":"2026-06-13T09:00:06.069Z"},"license":"MIT","keywords":["llm","openai","anthropic","claude","cost","budget","middleware","token","api","guard"],"description":"[![npm version](https://img.shields.io/npm/v/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228/llm-cost-guard)\r [![npm downloads](https://img.shields.io/npm/dm/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228","maintainers":[{"name":"advik1228","email":"advik.hingmire12@gmail.com"}],"readme":"# @advik1228/llm-cost-guard\r\n\r\n[![npm version](https://img.shields.io/npm/v/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228/llm-cost-guard)\r\n[![npm downloads](https://img.shields.io/npm/dm/%40advik1228%2Fllm-cost-guard)](https://www.npmjs.com/package/@advik1228/llm-cost-guard)\r\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\r\n\r\nDrop-in cost guard for LLM API clients. Wrap your Anthropic, OpenAI, or Google Gemini client with a JavaScript Proxy — track spend, enforce limits, and get alerts without changing your application code.\r\n\r\n## Why This Exists\r\n\r\nLLM API costs can spike silently. A single runaway loop or oversized prompt can burn through a daily budget in minutes. Most teams discover overspend only after the invoice arrives.\r\n\r\n**@advik1228/llm-cost-guard** sits between your app and the LLM provider. It reads real token counts from API responses, calculates cost in USD, enforces configurable limits, and optionally fires webhook alerts — all with one wrapper call.\r\n\r\n## Features\r\n\r\n- **Zero code changes** — wrap existing clients via Proxy; intercepts `messages.create`, `chat.completions.create`, and Gemini `generateContent`\r\n- **Real token counts** — reads `usage` from API responses (not tiktoken estimates at runtime)\r\n- **Daily, monthly, and per-request limits** — throw, warn, or silently ignore on breach\r\n- **Per-user budget tracking** — isolate spend by `userId` with optional `userDailyLimit`\r\n- **Pre-flight estimation** — estimate cost before sending (char/4 heuristic)\r\n- **Streaming support** — Anthropic and OpenAI streaming with usage captured after stream completes\r\n- **Webhook alerts** — Slack/Discord/custom webhook on warn threshold or limit breach\r\n- **Pluggable storage** — in-memory (default) or Redis for multi-instance deployments\r\n- **Custom pricing** — override built-in model rates for private or fine-tuned models\r\n- **TypeScript-first** — full type definitions included\r\n\r\n## Installation\r\n\r\n```bash\r\nnpm install @advik1228/llm-cost-guard\r\n```\r\n\r\n## Quick Start\r\n\r\n### Anthropic\r\n\r\n```typescript\r\nimport Anthropic from \"@anthropic-ai/sdk\";\r\nimport { guard } from \"@advik1228/llm-cost-guard\";\r\n\r\nconst client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });\r\n\r\nconst guarded = guard(client, {\r\n  dailyLimit: 5.0,\r\n  warnAt: 4.0,\r\n  onLimit: \"throw\",\r\n});\r\n\r\nconst response = await guarded.messages.create({\r\n  model: \"claude-sonnet-4-6\",\r\n  max_tokens: 1024,\r\n  messages: [{ role: \"user\", content: \"Hello!\" }],\r\n});\r\n\r\nconsole.log(response.content);\r\n```\r\n\r\n### OpenAI\r\n\r\n```typescript\r\nimport OpenAI from \"openai\";\r\nimport { guard } from \"@advik1228/llm-cost-guard\";\r\n\r\nconst client = new OpenAI({ apiKey: process.env.OPENAI_API_KEY });\r\n\r\nconst guarded = guard(client, {\r\n  dailyLimit: 10.0,\r\n  monthlyLimit: 100.0,\r\n  perRequestLimit: 0.50,\r\n  onLimit: \"throw\",\r\n});\r\n\r\nconst response = await guarded.chat.completions.create({\r\n  model: \"gpt-4o\",\r\n  messages: [{ role: \"user\", content: \"Hello!\" }],\r\n});\r\n\r\nconsole.log(response.choices[0].message.content);\r\n```\r\n\r\n## Configuration\r\n\r\nPass a `GuardConfig` object as the second argument to `guard()`:\r\n\r\n| Option | Type | Description |\r\n|--------|------|-------------|\r\n| `dailyLimit` | `number` | Max USD spend per UTC day |\r\n| `monthlyLimit` | `number` | Max USD spend per UTC month |\r\n| `perRequestLimit` | `number` | Max USD per single API call |\r\n| `warnAt` | `number` | USD threshold to trigger a warning (and webhook if configured) |\r\n| `onLimit` | `\"throw\" \\| \"warn\" \\| \"silent\"` | Behavior when a limit is exceeded (default: `\"throw\"`) |\r\n| `userId` | `string` | Track spend for a specific user |\r\n| `userDailyLimit` | `number` | Per-user daily USD cap |\r\n| `alertWebhook` | `string` | URL to POST alert payloads on warn/limit events |\r\n| `preflight` | `boolean` | Reserved for pre-flight enforcement |\r\n| `storage` | `\"memory\" \\| StorageAdapter` | Storage backend (default: in-memory) |\r\n| `customPricing` | `Record<string, { inputPerMillion, outputPerMillion }>` | Override built-in model pricing |\r\n\r\n```typescript\r\nconst guarded = guard(client, {\r\n  dailyLimit: 5.0,\r\n  monthlyLimit: 50.0,\r\n  perRequestLimit: 1.0,\r\n  warnAt: 4.0,\r\n  onLimit: \"throw\",\r\n  alertWebhook: \"https://hooks.slack.com/services/...\",\r\n  userId: \"user_abc123\",\r\n  userDailyLimit: 0.50,\r\n});\r\n```\r\n\r\n## Per-User Budget Tracking\r\n\r\nTrack spend per end-user in multi-tenant apps:\r\n\r\n```typescript\r\nconst guarded = guard(client, {\r\n  dailyLimit: 100.0,       // org-wide cap\r\n  userId: req.user.id,\r\n  userDailyLimit: 2.0,     // per-user cap\r\n  onLimit: \"throw\",\r\n});\r\n```\r\n\r\nWhen a user's daily spend exceeds `userDailyLimit`, a `LimitExceededError` is thrown with `limitType: \"user\"`.\r\n\r\n## Pre-Flight Cost Estimation\r\n\r\nEstimate cost before making a call (uses char/4 token heuristic, assumes output = 25% of input):\r\n\r\n```typescript\r\nimport { estimateCost } from \"@advik1228/llm-cost-guard\";\r\n\r\nconst estimate = await estimateCost(\r\n  {\r\n    provider: \"anthropic\",\r\n    model: \"claude-sonnet-4-6\",\r\n    messages: [{ role: \"user\", content: \"Summarize this 10-page document...\" }],\r\n  },\r\n  2.5,  // dailySpentSoFar (USD)\r\n  5.0   // dailyLimit (USD)\r\n);\r\n\r\nconsole.log(estimate.estimatedCostUSD);   // e.g. 0.003\r\nconsole.log(estimate.willBreachLimit);    // false\r\n```\r\n\r\nStreaming requests run a pre-flight check via `estimateCost()` before the stream starts.\r\n\r\n## Webhook Alerts\r\n\r\nSet `alertWebhook` to receive JSON POST payloads on warn threshold or limit breach:\r\n\r\n```typescript\r\nconst guarded = guard(client, {\r\n  dailyLimit: 10.0,\r\n  warnAt: 8.0,\r\n  alertWebhook: \"https://your-webhook.example.com/alerts\",\r\n});\r\n```\r\n\r\n**Payload shape:**\r\n\r\n```json\r\n{\r\n  \"event\": \"warn_threshold_reached\",\r\n  \"currentSpendUSD\": 8.12,\r\n  \"limitUSD\": 10.0,\r\n  \"warnAtUSD\": 8.0,\r\n  \"timestamp\": \"2026-06-13T12:00:00.000Z\",\r\n  \"provider\": \"anthropic\",\r\n  \"userId\": \"user_abc123\"\r\n}\r\n```\r\n\r\nEvents: `warn_threshold_reached`, `limit_reached`, `per_request_limit_reached`, `user_limit_reached`.\r\n\r\nAlerts are fire-and-forget — webhook failures are logged but never throw.\r\n\r\n## Usage Stats\r\n\r\n```typescript\r\nimport { getStats } from \"@advik1228/llm-cost-guard\";\r\n\r\nconst stats = await getStats();\r\nconsole.log(stats.todayUSD);      // today's spend\r\nconsole.log(stats.monthUSD);      // this month's spend\r\nconsole.log(stats.requestCount);  // total requests tracked\r\nconsole.log(stats.byModel);       // spend by model (future)\r\nconsole.log(stats.byUser);        // spend by user (future)\r\n```\r\n\r\n> **Note:** `getStats()` uses a module-level in-memory tracker. For production, pass a shared `StorageAdapter` via `guard({ storage })` and query it directly.\r\n\r\n## Error Reference\r\n\r\n### `LimitExceededError`\r\n\r\nThrown when a configured limit is exceeded (when `onLimit: \"throw\"`).\r\n\r\n```typescript\r\nimport { LimitExceededError } from \"@advik1228/llm-cost-guard\";\r\n\r\ntry {\r\n  await guarded.messages.create({ ... });\r\n} catch (err) {\r\n  if (err instanceof LimitExceededError) {\r\n    console.log(err.limitType);     // \"daily\" | \"monthly\" | \"perRequest\" | \"user\"\r\n    console.log(err.currentSpend);  // current USD spend\r\n    console.log(err.limit);         // configured limit\r\n  }\r\n}\r\n```\r\n\r\n### `PreflightError`\r\n\r\nReserved for pre-flight enforcement when estimated cost exceeds remaining budget.\r\n\r\n```typescript\r\nimport { PreflightError } from \"@advik1228/llm-cost-guard\";\r\n// err.estimatedCost, err.remainingBudget\r\n```\r\n\r\n## Supported Models & Pricing\r\n\r\nPrices in USD per 1 million tokens (input / output):\r\n\r\n| Model | Input | Output |\r\n|-------|------:|-------:|\r\n| `claude-opus-4-6` | $15.00 | $75.00 |\r\n| `claude-sonnet-4-6` | $3.00 | $15.00 |\r\n| `claude-haiku-4-5` | $0.80 | $4.00 |\r\n| `gpt-4o` | $5.00 | $15.00 |\r\n| `gpt-4o-mini` | $0.15 | $0.60 |\r\n| `gpt-3.5-turbo` | $0.50 | $1.50 |\r\n| `gemini-1.5-pro` | $3.50 | $10.50 |\r\n| `gemini-1.5-flash` | $0.075 | $0.30 |\r\n| `gemini-2.0-flash` | $0.10 | $0.40 |\r\n\r\n## Custom Pricing Override\r\n\r\nOverride or add models not in the built-in table:\r\n\r\n```typescript\r\nconst guarded = guard(client, {\r\n  dailyLimit: 10.0,\r\n  customPricing: {\r\n    \"my-fine-tuned-model\": {\r\n      inputPerMillion: 1.0,\r\n      outputPerMillion: 3.0,\r\n    },\r\n  },\r\n});\r\n```\r\n\r\n## Redis Storage\r\n\r\nFor multi-instance deployments, use Redis instead of in-memory storage:\r\n\r\n```typescript\r\nimport { createClient } from \"redis\";\r\nimport { guard, RedisAdapter } from \"@advik1228/llm-cost-guard\";\r\n\r\nconst redis = createClient({ url: process.env.REDIS_URL });\r\nawait redis.connect();\r\n\r\nconst guarded = guard(client, {\r\n  dailyLimit: 10.0,\r\n  storage: new RedisAdapter(redis),\r\n});\r\n```\r\n\r\nOr import the adapter directly:\r\n\r\n```typescript\r\nimport { RedisAdapter } from \"@advik1228/llm-cost-guard/adapters\";\r\n```\r\n\r\nAll Redis keys are prefixed with `llmguard:` to avoid collisions.\r\n\r\n## Custom Storage Adapter\r\n\r\nImplement the `StorageAdapter` interface for your own backend:\r\n\r\n```typescript\r\nimport { guard, StorageAdapter } from \"@advik1228/llm-cost-guard\";\r\n\r\nclass PostgresAdapter implements StorageAdapter {\r\n  async get(key: string): Promise<number> { /* ... */ return 0; }\r\n  async set(key: string, value: number, ttlSeconds?: number): Promise<void> { /* ... */ }\r\n  async increment(key: string, by: number): Promise<number> { /* ... */ return 0; }\r\n}\r\n\r\nconst guarded = guard(client, {\r\n  dailyLimit: 10.0,\r\n  storage: new PostgresAdapter(),\r\n});\r\n```\r\n\r\n## Environment Variables\r\n\r\nThe library does not read environment variables directly. Pass credentials and config from your app:\r\n\r\n```bash\r\nANTHROPIC_API_KEY=sk-ant-...\r\nOPENAI_API_KEY=sk-...\r\nREDIS_URL=redis://localhost:6379\r\n```\r\n\r\n```typescript\r\nconst guarded = guard(client, {\r\n  dailyLimit: parseFloat(process.env.DAILY_LLM_BUDGET ?? \"5\"),\r\n  alertWebhook: process.env.LLM_ALERT_WEBHOOK,\r\n});\r\n```\r\n\r\n## TypeScript Support\r\n\r\nFull type definitions are included. Import types as needed:\r\n\r\n```typescript\r\nimport {\r\n  guard,\r\n  GuardConfig,\r\n  UsageStats,\r\n  LimitExceededError,\r\n  PreflightError,\r\n  AlertPayload,\r\n} from \"@advik1228/llm-cost-guard\";\r\n```\r\n\r\n## Contributing\r\n\r\n### Setup\r\n\r\n```bash\r\ngit clone https://github.com/advikhingmire12-oss/llm-cost-guard.git\r\ncd llm-cost-guard\r\nnpm install\r\nnpm run build\r\nnpm test\r\n```\r\n\r\n### Project Structure\r\n\r\n```\r\nsrc/\r\n  index.ts          # Public exports\r\n  guard.ts          # Proxy wrapper + limit enforcement\r\n  tracker.ts        # SpendTracker — records and queries spend\r\n  pricing.ts        # Model pricing table + calculateCost()\r\n  estimator.ts      # Pre-flight cost estimation\r\n  errors.ts         # LimitExceededError, PreflightError\r\n  alerts.ts         # Webhook alert sender\r\n  adapters/\r\n    memory.ts       # In-memory StorageAdapter (default)\r\n    redis.ts        # Redis StorageAdapter\r\ntests/\r\n  guard.test.ts\r\n  estimator.test.ts\r\n```\r\n\r\n### How to Add a Provider\r\n\r\n1. In `src/guard.ts`, detect the provider's API call pattern in the Proxy `get` trap (method name + parent object path).\r\n2. Add token extraction in `extractTokens()` using the provider's `usage` response shape.\r\n3. For nested clients (like Gemini's `getGenerativeModel`), wrap returned objects in a secondary Proxy.\r\n4. Add model pricing to `PRICING` in `src/pricing.ts`.\r\n5. Add tests in `tests/guard.test.ts` with a mocked client.\r\n\r\n### PR Rules\r\n\r\n- All tests must pass (`npm test`) before submitting\r\n- Keep changes focused — one feature or fix per PR\r\n- Add tests for new behavior\r\n- Do not change public API without discussion\r\n- Follow existing TypeScript style (strict mode, no `any` unless unavoidable)\r\n\r\n## Roadmap\r\n\r\n- [ ] Full `byModel` and `byUser` breakdown in `getStats()`\r\n- [ ] Preflight enforcement via `preflight: true` config flag\r\n- [ ] Dashboard / CLI for live spend monitoring\r\n- [ ] AWS Bedrock and Azure OpenAI adapters\r\n- [ ] Rate limiting (requests per minute, not just cost)\r\n\r\n## License\r\n\r\nMIT — see [LICENSE](LICENSE).\r\n\r\n## Author\r\n\r\n**Advik** — [github.com/advikhingmire12-oss](https://github.com/advikhingmire12-oss)\r\n","readmeFilename":"README.md"}