{"_id":"@aid-on/llm-throttle","_rev":"2-bbd6c6d37c12d07eaeb10b8148246d91","name":"@aid-on/llm-throttle","dist-tags":{"latest":"1.0.1"},"versions":{"1.0.0":{"name":"@aid-on/llm-throttle","version":"1.0.0","keywords":["rate-limit","throttle","llm","api","rpm","tpm","token-bucket","rate-limiting","openai","anthropic","typescript"],"author":{"name":"aid-on"},"license":"MIT","_id":"@aid-on/llm-throttle@1.0.0","maintainers":[{"name":"aid-on","email":"hiromi.motodera@aid-on.org"}],"homepage":"https://aid-on.github.io/llm-throttle/","bugs":{"url":"https://github.com/Aid-On/llm-throttle/issues"},"dist":{"shasum":"a7ed20908b2184daf40ad93960ac9e3a7bdcb349","tarball":"https://registry.npmjs.org/@aid-on/llm-throttle/-/llm-throttle-1.0.0.tgz","fileCount":9,"integrity":"sha512-TLfT1DQDTVblXgdlE5Q6fV4fLq2Kkf2fKEqLRbPMtghNhWp6yclNgKFFAnGNvX/mfLjTvqigvf9aRkzpBS7dQw==","signatures":[{"sig":"MEUCIQDeLE8bMz7/8fPzbFqo5SNTGkGq1TD4/0l9lR9HfmCaYQIgLDlNQL/Pc2/ZFn4VF8FYAEh3oUtHCpIj9S+e0t0f7T4=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":311349},"main":"dist/index.js","types":"dist/index.d.ts","module":"dist/index.mjs","engines":{"node":">=16.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.js"}},"gitHead":"b17a501429d9d5843df314979811782e7cbf4b37","scripts":{"dev":"tsup --watch","lint":"eslint src --ext .ts","test":"vitest","build":"tsup","demo:dev":"vite demo","demo:build":"vite build","type-check":"tsc --noEmit","demo:preview":"vite preview demo","deploy:manual":"npm run build && npm run demo:build && gh-pages -d demo-dist","test:coverage":"vitest --coverage","prepublishOnly":"npm run build && npm run type-check"},"_npmUser":{"name":"aid-on","email":"hiromi.motodera@aid-on.org"},"repository":{"url":"git+https://github.com/Aid-On/llm-throttle.git","type":"git"},"_npmVersion":"10.2.3","description":"高精度なLLMレート制限ライブラリ - Precise dual rate limiting for LLM APIs (RPM + TPM)","directories":{},"_nodeVersion":"20.10.0","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","vite":"^5.4.19","react":"^19.1.1","eslint":"^8.0.0","vitest":"^1.0.0","gh-pages":"^6.3.0","react-dom":"^19.1.1","typescript":"^5.0.0","@types/node":"^20.0.0","@types/react":"^19.1.9","@types/react-dom":"^19.1.7","@vitest/coverage-v8":"^1.0.0","@vitejs/plugin-react":"^4.7.0","@typescript-eslint/parser":"^6.0.0","@typescript-eslint/eslint-plugin":"^6.0.0"},"optionalDependencies":{"@aid-on/fuzztok":"^1.0.0"},"_npmOperationalInternal":{"tmp":"tmp/llm-throttle_1.0.0_1754074810057_0.9438707231902155","host":"s3://npm-registry-packages-npm-production"}},"1.0.1":{"name":"@aid-on/llm-throttle","version":"1.0.1","description":"高精度なLLMレート制限ライブラリ - Precise dual rate limiting for LLM APIs (RPM + TPM)","main":"dist/index.js","types":"dist/index.d.ts","module":"dist/index.mjs","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.js"}},"scripts":{"build":"tsup","dev":"tsup --watch","type-check":"tsc --noEmit","lint":"eslint src --ext .ts","test":"vitest run","test:coverage":"vitest --coverage","prepublishOnly":"npm run build && npm run type-check","demo:dev":"vite demo","demo:build":"vite build","demo:preview":"vite preview demo","deploy:manual":"npm run build && npm run demo:build && gh-pages -d demo-dist"},"keywords":["rate-limit","throttle","llm","api","rpm","tpm","token-bucket","rate-limiting","openai","anthropic","typescript"],"author":{"name":"aid-on"},"license":"MIT","devDependencies":{"@types/node":"^20.0.0","@types/react":"^19.1.9","@types/react-dom":"^19.1.7","@typescript-eslint/eslint-plugin":"^6.0.0","@typescript-eslint/parser":"^6.0.0","@vitejs/plugin-react":"^4.7.0","@vitest/coverage-v8":"^1.0.0","eslint":"^8.0.0","gh-pages":"^6.3.0","react":"^19.1.1","react-dom":"^19.1.1","tsup":"^8.0.0","typescript":"^5.0.0","vite":"^5.4.19","vitest":"^1.0.0"},"optionalDependencies":{"@aid-on/fuzztok":"^1.0.0"},"engines":{"node":">=16.0.0"},"publishConfig":{"access":"public"},"repository":{"type":"git","url":"git+https://github.com/Aid-On/llm-throttle.git"},"homepage":"https://aid-on.github.io/llm-throttle/","bugs":{"url":"https://github.com/Aid-On/llm-throttle/issues"},"_id":"@aid-on/llm-throttle@1.0.1","gitHead":"1c3514bbf799a21c95e9642683733c8a32f2a1d1","_nodeVersion":"20.10.0","_npmVersion":"10.2.3","dist":{"integrity":"sha512-y5W3+UUdbCRCGSL1khMpTv7RTC59i4f8NdHKHgw0jNbZ/8U8DnA4IsdddSH58dVI7aIfRRMy2VKVGAkr6MaLsA==","shasum":"0d5f76070a9dd664e90d04d857dbe98f27358b4b","tarball":"https://registry.npmjs.org/@aid-on/llm-throttle/-/llm-throttle-1.0.1.tgz","fileCount":9,"unpackedSize":312057,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQCsG5S42BBvEKF1J+z5Hhj0SP8RqxMnlWkmYndzX0GdOgIgNQ+T55BLg768C4KmN5e583TYieTcHX4ys662+68is2o="}]},"_npmUser":{"name":"aid-on","email":"hiromi.motodera@aid-on.org"},"directories":{},"maintainers":[{"name":"aid-on","email":"hiromi.motodera@aid-on.org"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/llm-throttle_1.0.1_1754216224618_0.004997561239898252"},"_hasShrinkwrap":false}},"time":{"created":"2025-08-01T19:00:09.990Z","modified":"2025-08-03T10:17:05.003Z","1.0.0":"2025-08-01T19:00:10.238Z","1.0.1":"2025-08-03T10:17:04.821Z"},"bugs":{"url":"https://github.com/Aid-On/llm-throttle/issues"},"author":{"name":"aid-on"},"license":"MIT","homepage":"https://aid-on.github.io/llm-throttle/","keywords":["rate-limit","throttle","llm","api","rpm","tpm","token-bucket","rate-limiting","openai","anthropic","typescript"],"repository":{"type":"git","url":"git+https://github.com/Aid-On/llm-throttle.git"},"description":"高精度なLLMレート制限ライブラリ - Precise dual rate limiting for LLM APIs (RPM + TPM)","maintainers":[{"name":"aid-on","email":"hiromi.motodera@aid-on.org"}],"readme":"# @aid-on/llm-throttle\n\nPrecise dual rate limiting for LLM APIs (RPM + TPM)\n\n## Overview\n\n`@aid-on/llm-throttle` is a high-precision rate limiting library specialized for LLM API calls. It simultaneously controls both RPM (Requests Per Minute) and TPM (Tokens Per Minute) to achieve efficient API usage.\n\n## Features\n\n- **Dual Rate Limiting**: Simultaneously manages both RPM and TPM\n- **Token Bucket Algorithm**: Smoothed rate limiting with burst handling\n- **Real-time Adjustment**: Post-adjustment based on actual token consumption\n- **Detailed Metrics**: Usage visualization and efficiency tracking\n- **Full TypeScript Support**: Type-safe development experience\n- **Zero Dependencies**: Lightweight design with no external library dependencies\n\n## Installation\n\n```bash\nnpm install @aid-on/llm-throttle\n```\n\n## Basic Usage\n\n```typescript\nimport { LLMThrottle } from '@aid-on/llm-throttle';\n\n// Configure rate limits\nconst limiter = new LLMThrottle({\n  rpm: 60,     // 60 requests per minute\n  tpm: 10000   // 10,000 tokens per minute\n});\n\n// Check before request\nconst requestId = 'unique-request-id';\nconst estimatedTokens = 1500;\n\nif (limiter.consume(requestId, estimatedTokens)) {\n  // Execute API call\n  const response = await callLLMAPI();\n  \n  // Adjust with actual token usage\n  const actualTokens = response.usage.total_tokens;\n  limiter.adjustConsumption(requestId, actualTokens);\n} else {\n  console.log('Rate limit reached');\n}\n```\n\n## Advanced Usage\n\n### Burst Limit Configuration\n\n```typescript\nconst limiter = new LLMThrottle({\n  rpm: 60,\n  tpm: 10000,\n  burstRPM: 120,    // Allow up to 120 requests in short bursts\n  burstTPM: 20000   // Allow up to 20,000 tokens in short bursts\n});\n```\n\n### Error Handling\n\n```typescript\nimport { RateLimitError } from '@aid-on/llm-throttle';\n\ntry {\n  limiter.consumeOrThrow(requestId, estimatedTokens);\n  // API call processing\n} catch (error) {\n  if (error instanceof RateLimitError) {\n    console.log(`Limit reason: ${error.reason}`);\n    console.log(`Available in: ${error.availableIn}ms`);\n  }\n}\n```\n\n### Getting Metrics\n\n```typescript\nconst metrics = limiter.getMetrics();\n\nconsole.log('RPM usage:', metrics.rpm.percentage + '%');\nconsole.log('TPM usage:', metrics.tpm.percentage + '%');\nconsole.log('Average tokens/request:', metrics.consumptionHistory.averageTokensPerRequest);\nconsole.log('Estimation accuracy:', metrics.efficiency);\n```\n\n### Pre-check\n\n```typescript\nconst check = limiter.canProcess(estimatedTokens);\n\nif (check.allowed) {\n  // Can process\n  limiter.consume(requestId, estimatedTokens);\n} else {\n  console.log(`Limit reason: ${check.reason}`);\n  console.log(`Available in: ${check.availableIn}ms`);\n}\n```\n\n## API Reference\n\n### LLMThrottle\n\n#### Constructor\n\n```typescript\nnew LLMThrottle(config: DualRateLimitConfig)\n```\n\n#### Methods\n\n- `canProcess(estimatedTokens: number): RateLimitCheckResult` - Check if processing is possible\n- `consume(requestId: string, estimatedTokens: number, metadata?: Record<string, unknown>): boolean` - Consume tokens\n- `consumeOrThrow(requestId: string, estimatedTokens: number, metadata?: Record<string, unknown>): void` - Throw error on consumption failure\n- `adjustConsumption(requestId: string, actualTokens: number): void` - Adjust with actual consumption\n- `getMetrics(): RateLimitMetrics` - Get usage metrics\n- `getConsumptionHistory(): ConsumptionRecord[]` - Get consumption history\n- `reset(): void` - Reset limit state\n- `setHistoryRetention(ms: number): void` - Set history retention period\n\n### Type Definitions\n\n```typescript\ninterface DualRateLimitConfig {\n  rpm: number;\n  tpm: number;\n  burstRPM?: number;\n  burstTPM?: number;\n  clock?: () => number;\n}\n\ninterface RateLimitCheckResult {\n  allowed: boolean;\n  reason?: 'rpm_limit' | 'tpm_limit';\n  availableIn?: number;\n  availableTokens?: {\n    rpm: number;\n    tpm: number;\n  };\n}\n\ninterface RateLimitMetrics {\n  rpm: {\n    used: number;\n    available: number;\n    limit: number;\n    percentage: number;\n  };\n  tpm: {\n    used: number;\n    available: number;\n    limit: number;\n    percentage: number;\n  };\n  efficiency: number;\n  consumptionHistory: {\n    count: number;\n    averageTokensPerRequest: number;\n    totalTokens: number;\n  };\n}\n```\n\n## Practical Examples\n\n### Integration with OpenAI API\n\n```typescript\nimport OpenAI from 'openai';\nimport { LLMThrottle } from '@aid-on/llm-throttle';\n\nconst openai = new OpenAI();\nconst limiter = new LLMThrottle({\n  rpm: 500,    // Example OpenAI Tier 1 limits\n  tpm: 10000\n});\n\nasync function chatCompletion(messages: any[], requestId: string) {\n  const estimatedTokens = estimateTokens(messages); // Custom estimation logic\n  \n  if (!limiter.consume(requestId, estimatedTokens)) {\n    throw new Error('Rate limit reached');\n  }\n  \n  try {\n    const response = await openai.chat.completions.create({\n      model: 'gpt-3.5-turbo',\n      messages\n    });\n    \n    // Adjust with actual usage\n    const actualTokens = response.usage?.total_tokens || estimatedTokens;\n    limiter.adjustConsumption(requestId, actualTokens);\n    \n    return response;\n  } catch (error) {\n    // Return estimated value on error\n    limiter.adjustConsumption(requestId, 0);\n    throw error;\n  }\n}\n```\n\n### Multi-service Integration\n\n```typescript\nclass APIManager {\n  private limiters = new Map<string, LLMThrottle>();\n  \n  constructor() {\n    // Service-specific limit configuration\n    this.limiters.set('openai', new LLMThrottle({\n      rpm: 500, tpm: 10000\n    }));\n    this.limiters.set('anthropic', new LLMThrottle({\n      rpm: 1000, tpm: 20000\n    }));\n  }\n  \n  async callAPI(service: string, requestId: string, estimatedTokens: number) {\n    const limiter = this.limiters.get(service);\n    if (!limiter) throw new Error(`Unknown service: ${service}`);\n    \n    const check = limiter.canProcess(estimatedTokens);\n    if (!check.allowed) {\n      throw new RateLimitError(\n        `Rate limit exceeded for ${service}: ${check.reason}`,\n        check.reason!,\n        check.availableIn!\n      );\n    }\n    \n    limiter.consume(requestId, estimatedTokens);\n    // API call processing...\n  }\n}\n```\n\n## Testing\n\n```bash\nnpm test\n```\n\n## License\n\nMIT License","readmeFilename":"README.md"}