{"_id":"@andy-toolforge/tts-generator","_rev":"6-d6d6ae5cb816eba2449dd504ff85ea2f","name":"@andy-toolforge/tts-generator","dist-tags":{"latest":"0.5.0"},"versions":{"0.1.0":{"name":"@andy-toolforge/tts-generator","version":"0.1.0","_id":"@andy-toolforge/tts-generator@0.1.0","maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"homepage":"https://github.com/andy-pham-it/toolforge#readme","bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"dist":{"shasum":"c358e5b86ea052e7653a2a5ce8b09276c99a5878","tarball":"https://registry.npmjs.org/@andy-toolforge/tts-generator/-/tts-generator-0.1.0.tgz","fileCount":14,"integrity":"sha512-AR7rS4LR7bfsxV2o8xok+LinTkySaIIO1XQJZITPZIK6Gi1/lpg1cBtYlBr8CitcC5qu7JFw7HLTWDW+mpIrKw==","signatures":[{"sig":"MEYCIQDL/1z4WOL7Ia+VRYge2S0MCv8OkOqDI55V8LYHkhr8iAIhAPp7z+6bVjNJuohpYENxLauclu0APINqBp5ZEGp03DA6","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":59937},"main":"lib/index.js","gitHead":"874522837d0b8e076122188cf70e0d79ec2928a3","scripts":{"test":"node --test lib/*.test.js","postinstall":"node skills/postinstall.js"},"_npmUser":{"name":"andy_pham","email":"phamlehoaian@gmail.com"},"repository":{"url":"git+https://github.com/andy-pham-it/toolforge.git","type":"git"},"_npmVersion":"10.8.2","description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","directories":{},"_nodeVersion":"20.20.2","dependencies":{"@andy-toolforge/core":"^1.0.0"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/tts-generator_0.1.0_1783412580921_0.8369187829978355","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@andy-toolforge/tts-generator","version":"0.1.1","_id":"@andy-toolforge/tts-generator@0.1.1","maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"homepage":"https://github.com/andy-pham-it/toolforge#readme","bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"dist":{"shasum":"fd99aa23936ef83809cb6281ec74631cf25e9ca3","tarball":"https://registry.npmjs.org/@andy-toolforge/tts-generator/-/tts-generator-0.1.1.tgz","fileCount":15,"integrity":"sha512-cmveHSwsBKyUpBNg40DbHGWDc8YWnXj8N9TUfiPjtLT8zPhvjpbbvK+TTGggdfXYPoTjCGv4NSZJLnGMC2ggaA==","signatures":[{"sig":"MEQCIBSRZAWHbVoY3fc/iz/DQgjuE8UBcoJdXcP8CMIFWtuEAiAubpXP4u30oM47O7leNgeH2T8x/k/HdxFw8ndG3B84+A==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":76042},"main":"lib/index.js","gitHead":"e09a6db0b21c3071864017dda9baab81ac5a4eb0","scripts":{"test":"node --test lib/*.test.js","postinstall":"node skills/postinstall.js"},"_npmUser":{"name":"andy_pham","email":"phamlehoaian@gmail.com"},"repository":{"url":"git+https://github.com/andy-pham-it/toolforge.git","type":"git"},"_npmVersion":"10.8.2","description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","directories":{},"_nodeVersion":"20.20.2","dependencies":{"@andy-toolforge/core":"^1.0.0"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/tts-generator_0.1.1_1783435906284_0.4190340766458882","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@andy-toolforge/tts-generator","version":"0.2.0","_id":"@andy-toolforge/tts-generator@0.2.0","maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"homepage":"https://github.com/andy-pham-it/toolforge#readme","bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"dist":{"shasum":"88e7c95d4bc201806b3be8905e90cd0d1de7ef73","tarball":"https://registry.npmjs.org/@andy-toolforge/tts-generator/-/tts-generator-0.2.0.tgz","fileCount":16,"integrity":"sha512-8IEG+jcV9aO0hEaeC1RsjWul971jT1FmgxP8FkORWBgGQEbWe8ltyN3dc/+8hqomvYw1FqDThSzGqDQ6yZoyjA==","signatures":[{"sig":"MEQCIDRtGRBxQSnThfeGelgPWy3OKuYtgvGKLlTQqV7JNeykAiAhCQnBEoykX5Mv5ZlihYAhtCdPhRaEcD+Yrr4fAj9VvQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":97878},"main":"lib/index.js","gitHead":"dae670a87cbaf9b208495b20e0ff8afb24bfd8b3","scripts":{"test":"node --test lib/*.test.js","postinstall":"node skills/postinstall.js"},"_npmUser":{"name":"andy_pham","email":"phamlehoaian@gmail.com"},"repository":{"url":"git+https://github.com/andy-pham-it/toolforge.git","type":"git"},"_npmVersion":"10.8.2","description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","directories":{},"_nodeVersion":"20.20.2","dependencies":{"ws":"^8.18.0","@andy-toolforge/core":"^1.0.0"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/tts-generator_0.2.0_1783439270795_0.5565943016813337","host":"s3://npm-registry-packages-npm-production"}},"0.3.0":{"name":"@andy-toolforge/tts-generator","version":"0.3.0","_id":"@andy-toolforge/tts-generator@0.3.0","maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"homepage":"https://github.com/andy-pham-it/toolforge#readme","bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"dist":{"shasum":"9e0e9ff86d8d8c80642faddfa338a095ad99b20a","tarball":"https://registry.npmjs.org/@andy-toolforge/tts-generator/-/tts-generator-0.3.0.tgz","fileCount":17,"integrity":"sha512-+gobnD5Q4NXYipp9Wlkg4WkC2OGeXiMfZiLfWkJwk3vaNLbwAYxMVd7Fz+9Smr3oPAA9g5HLmC1zLsFr0CpDSA==","signatures":[{"sig":"MEUCIHOaCL3Z4YKcw/INkXaQlS6QLW7I0+LsnSsFPx43j+8ZAiEA1wo5Tt9Yp0VLs5vTNwB0ddXad87mokogLAm901ZDAcw=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":124985},"main":"lib/index.js","gitHead":"4e6b9acf1cca46894d689bc324422832321b08e5","scripts":{"test":"node --test lib/*.test.js","postinstall":"node skills/postinstall.js"},"_npmUser":{"name":"andy_pham","email":"phamlehoaian@gmail.com"},"repository":{"url":"git+https://github.com/andy-pham-it/toolforge.git","type":"git"},"_npmVersion":"10.8.2","description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","directories":{},"_nodeVersion":"20.20.2","dependencies":{"@google/genai":"^2.10.0","@andy-toolforge/core":"^1.0.0"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/tts-generator_0.3.0_1783476340435_0.7177531808045843","host":"s3://npm-registry-packages-npm-production"}},"0.4.0":{"name":"@andy-toolforge/tts-generator","version":"0.4.0","_id":"@andy-toolforge/tts-generator@0.4.0","maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"homepage":"https://github.com/andy-pham-it/toolforge#readme","bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"dist":{"shasum":"feb5005dec05fdddcd43d848d2ec25718e663d60","tarball":"https://registry.npmjs.org/@andy-toolforge/tts-generator/-/tts-generator-0.4.0.tgz","fileCount":22,"integrity":"sha512-KKvmyrbzTpZu+C9lDaNvNE+4Tdq2hBFfdF7C9yxZSoQDFA2YrZ3X0IUhSEBHegKh+NU9W+7ZTxbq+faMCfiK0A==","signatures":[{"sig":"MEUCIBhucUPPPhuSNAheeDjL6upsWC8jmEQT8MZ/V44qsUN8AiEAqUwVPlT8EfyJ+10tkq+Y/XPF5LZjGbzZJqnynF+h74Y=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":181635},"main":"lib/index.js","gitHead":"bda537600aa40b3089f8eceb13c381c4f9e8415e","scripts":{"test":"node --test lib/*.test.js","postinstall":"node skills/postinstall.js"},"_npmUser":{"name":"andy_pham","email":"phamlehoaian@gmail.com"},"repository":{"url":"git+https://github.com/andy-pham-it/toolforge.git","type":"git"},"_npmVersion":"10.8.2","description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","directories":{},"_nodeVersion":"20.20.2","dependencies":{"express":"^4.21.0","@google/genai":"^2.10.0","@andy-toolforge/core":"^1.0.0"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/tts-generator_0.4.0_1783536502702_0.1432903357943982","host":"s3://npm-registry-packages-npm-production"}},"0.5.0":{"name":"@andy-toolforge/tts-generator","version":"0.5.0","description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","main":"lib/index.js","repository":{"type":"git","url":"git+https://github.com/andy-pham-it/toolforge.git"},"scripts":{"postinstall":"node skills/postinstall.js","test":"node --test lib/*.test.js"},"dependencies":{"@andy-toolforge/core":"^1.0.0","@google/genai":"^2.10.0","express":"^4.21.0"},"_id":"@andy-toolforge/tts-generator@0.5.0","gitHead":"ef7dc9fd191ca62c2304665520f951f1433c7bd2","bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"homepage":"https://github.com/andy-pham-it/toolforge#readme","_nodeVersion":"20.20.2","_npmVersion":"10.8.2","dist":{"integrity":"sha512-kCVVrn4SahAc5ggJ2F5ZgVVRtEg4aypxE6Qp2b1HAburKGl2hHy6MmCB+1GrX46Ww/iO/bSfYcj1aDTs9HFYVw==","shasum":"254a8a6a1ec615728ddddfc5e2bb491b46ebd22c","tarball":"https://registry.npmjs.org/@andy-toolforge/tts-generator/-/tts-generator-0.5.0.tgz","fileCount":23,"unpackedSize":193219,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIGlfiX/95nGK+jlz2bwLIbCnNOc3ssUnmydlyl8+JPOUAiAV2B263L17GW3PVVYaNhf95q/AG52f63A6QIkmchBSbg=="}]},"_npmUser":{"name":"andy_pham","email":"phamlehoaian@gmail.com"},"directories":{},"maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/tts-generator_0.5.0_1783869908623_0.5313698380712388"},"_hasShrinkwrap":false}},"time":{"created":"2026-07-07T08:23:00.813Z","modified":"2026-07-12T15:25:08.903Z","0.1.0":"2026-07-07T08:23:01.061Z","0.1.1":"2026-07-07T14:51:46.433Z","0.2.0":"2026-07-07T15:47:50.925Z","0.3.0":"2026-07-08T02:05:40.585Z","0.4.0":"2026-07-08T18:48:22.915Z","0.5.0":"2026-07-12T15:25:08.775Z"},"bugs":{"url":"https://github.com/andy-pham-it/toolforge/issues"},"homepage":"https://github.com/andy-pham-it/toolforge#readme","repository":{"type":"git","url":"git+https://github.com/andy-pham-it/toolforge.git"},"description":"Toolforge domain: text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output","maintainers":[{"name":"andy_pham","email":"phamlehoaian@gmail.com"}],"readme":"# @andy-toolforge/tts-generator\n\n[![npm](https://img.shields.io/npm/v/@andy-toolforge/tts-generator)](https://npmjs.com/package/@andy-toolforge/tts-generator)\n[![License](https://img.shields.io/npm/l/@andy-toolforge/tts-generator)](https://github.com/andy-pham-it/toolforge)\n\n**Text-to-speech generation using Gemini TTS models — script segmentation, multi-voice, batch/stream/single output.** Thuộc hệ sinh thái [toolforge](https://github.com/andy-pham-it/toolforge).\n\nPackage này hỗ trợ **hai API mode** song song:\n\n| Mode | API | Models | Use case |\n|------|-----|--------|----------|\n| `interactions` (REST) | Gemini Interactions API | `gemini-*-tts-preview` | Batch TTS đơn giản, gọi REST một lần → WAV |\n| `live` (WebSocket) | Gemini Live API (BidiGenerateContent) | `gemini-live-*-native-audio`, `gemini-*-live-*` | Real-time streaming, bidirectional audio dialog |\n\nPackage này thay thế hoàn toàn workflow thủ công (chia script trong AI Studio → generate từng đoạn → ghép trong CapCut) bằng một API tự động, hỗ trợ:\n\n- **Smart segmentation:** LLM-based chia script thành các đoạn logical (tự động fallback regex nếu LLM không available)\n- **30 Gemini TTS voices:** Từ Zephyr (Bright) đến Sulafat (Warm), mỗi giọng có style riêng\n- **Audio tags:** 200+ expressive tags ([whispers], [laughs], [determination]...) để điều khiển giọng đọc\n- **3 output modes:** batch (mảng segment-audio pairs), single (concatenated audio), stream (ordered với position metadata)\n- **Multi-speaker:** Hỗ trợ đến 2 speakers trong cùng một interaction\n- **403 quota fallback:** Tự động fallback cross-model (trong cùng mode) khi hết quota\n- **Batching:** Configurable concurrency để tránh rate limit\n\n## Installation\n\n```bash\nnpm install @andy-toolforge/tts-generator\n```\n\nYêu cầu: `@andy-toolforge/core` (tự động cài kèm) và Gemini API key.\n\n## Quick Start\n\n```javascript\nconst { TTSPlanner, TTSGenerator, OutputFormatter } = require('@andy-toolforge/tts-generator');\n\n// 1. Segmentation với LLM\nconst planner = new TTSPlanner({ llm });\nconst plan = await planner.plan(script, title);\n\n// 2. Generate audio với Gemini TTS\nconst gen = new TTSGenerator({\n    apiKey: process.env.GEMINI_API_KEY,\n});\nconst results = await gen.generateBatch(plan.segments);\n\n// 3. Format output\nconst formatter = new OutputFormatter();\nconst batch = formatter.formatBatch(plan.segments, results.map(r => r.audio));\n```\n\n## API Reference\n\n```javascript\nconst {\n    VOICES,            // { voiceName: { style, description } }\n    VOICE_NAMES,       // ['Zephyr', 'Puck', ...]\n    getVoice,          // (name) → { style, description } | null\n    pickVoiceForTone,  // (tone) → voice name\n\n    TTSPlanner,        // Script segmentation\n    TTSGenerator,      // Gemini TTS API client (Interactions REST)\n    LiveTTSGenerator,  // Gemini TTS API client (Live WebSocket)\n    OutputFormatter,   // Output formatting (batch/single/stream)\n\n    LIVE_MODELS,       // { modelName: { description } }\n    LIVE_MODEL_NAMES,  // ['gemini-live-2.5-flash-native-audio', ...]\n} = require('@andy-toolforge/tts-generator');\n```\n\n---\n\n### VOICES / VOICE_NAMES / getVoice / pickVoiceForTone\n\nXem danh sách và chọn giọng đọc từ 30 Gemini TTS voices.\n\n```javascript\nconst { VOICES, VOICE_NAMES, getVoice, pickVoiceForTone } = require('@andy-toolforge/tts-generator');\n\n// Danh sách đầy đủ\nconsole.log(VOICE_NAMES);         // ['Zephyr', 'Puck', 'Charon', ...]\nconsole.log(VOICES.Kore);         // { style: 'Firm', description: 'Assertive...' }\n\n// Tra cứu (case-insensitive)\nconst v = getVoice('kore');       // → { style: 'Firm', description: '...' }\n\n// Chọn giọng theo tone nội dung\npickVoiceForTone('informative');  // → 'Charon' | 'Iapetus' | 'Sadaltager'\npickVoiceForTone('upbeat');       // → 'Zephyr' | 'Puck' | 'Laomedeia'\npickVoiceForTone('calm');         // → 'Callirrhoe' | 'Umbriel' | 'Vindemiatrix'\npickVoiceForTone('authoritative');// → 'Kore' | 'Orus' | 'Alnilam'\npickVoiceForTone('friendly');     // → 'Achird' | 'Sulafat' | 'Despina'\n```\n\n| Tone | Voices |\n|------|--------|\n| `informative` | Charon, Iapetus, Sadaltager |\n| `upbeat` | Zephyr, Puck, Laomedeia |\n| `calm` | Callirrhoe, Umbriel, Vindemiatrix |\n| `authoritative` | Kore, Orus, Alnilam |\n| `friendly` | Achird, Sulafat, Despina |\n\n---\n\n### TTSPlanner\n\nChia script thành các logical segments cho TTS. Sử dụng LLM (nếu được cung cấp) hoặc fallback regex paragraph splitting.\n\n**Constructor:** `new TTSPlanner({ llm, maxRetries })`\n\n| Param | Type | Default | Mô tả |\n|-------|------|---------|-------|\n| `llm` | Object | required | LLM-compatible instance với `chat()` method. Nếu không có, dùng regex fallback. |\n| `maxRetries` | number | `1` | Số lần retry khi LLM trả về invalid JSON |\n\n**Method: `plan(script, title, options)`**\n\n| Param | Type | Default | Mô tả |\n|-------|------|---------|-------|\n| `script` | string | required | Full script text |\n| `title` | string | required | Episode title |\n| `options.voice` | string | `'auto'` | Override voice cho tất cả segments |\n| `options.language` | string | `'auto'` | Language code (vi, en, auto) |\n| `options.pace` | string | `'normal'` | Speed (slow, normal, fast) |\n\n**Return: `{ segments: Segment[], metadata: Metadata }`**\n\n```javascript\nconst { TTSPlanner } = require('@andy-toolforge/tts-generator');\nconst planner = new TTSPlanner({ llm });\n\nconst plan = await planner.plan(script, 'Tập 1: AI và Tương Lai', {\n    voice: 'Charon',\n    pace: 'normal',\n});\n\n// plan.segments → [\n//   { id: 1, text: '...', title: 'Giới thiệu', voice: 'Charon', pace: 'normal',\n//     audioTags: ['neutral'], language: 'vi', estimatedDuration: 30 },\n//   ...\n// ]\n// plan.metadata → { totalEstimatedDuration: 300, voiceCount: 1, languages: ['vi'] }\n```\n\n**Mỗi segment có cấu trúc:**\n\n| Field | Type | Mô tả |\n|-------|------|-------|\n| `id` | number | Segment index (1-based) |\n| `text` | string | Nội dung text cần TTS |\n| `title` | string | Short descriptive title |\n| `voice` | string | Voice name hoặc `\"auto\"` |\n| `pace` | string | `\"slow\"` \\| `\"normal\"` \\| `\"fast\"` |\n| `audioTags` | string[] | Expressive tags: `[\"determination\", \"whispers\"]` |\n| `language` | string | `\"vi\"` \\| `\"en\"` \\| `\"auto\"` |\n| `estimatedDuration` | number | Estimated seconds |\n\n---\n\n### TTSGenerator\n\nGọi Gemini Interactions API để sinh audio từ text segments.\n\n**Constructor:** `new TTSGenerator(config)`\n\n```javascript\nconst { TTSGenerator } = require('@andy-toolforge/tts-generator');\n\nconst gen = new TTSGenerator({\n    apiKey: process.env.GEMINI_API_KEY,  // Hoặc set env GEMINI_API_KEY / GOOGLE_API_KEY\n    tts: {\n        model: 'gemini-3.1-flash-tts-preview',   // Default\n        fallback: 'gemini-2.5-flash-preview-tts', // Fallback khi 403\n    },\n    maxRetries: 2,  // Retries per model\n    baseDelay: 1000, // Exponential backoff base (ms)\n});\n```\n\n| Config | Type | Default | Mô tả |\n|--------|------|---------|-------|\n| `apiKey` | string | `GEMINI_API_KEY` env | Gemini API key (required) |\n| `tts.model` | string | `gemini-3.1-flash-tts-preview` | Primary model |\n| `tts.fallback` | string | `gemini-2.5-flash-preview-tts` | Fallback on 403 |\n| `maxRetries` | number | `2` | Retries per model |\n| `baseDelay` | number | `1000` | Backoff base (ms) |\n\n**Method: `generate(segment)` → `{ id, text, audio: Buffer, voice, format }`**\n\nSinh audio cho một segment.\n\n```javascript\nconst result = await gen.generate({\n    id: 1,\n    text: 'Xin chào các bạn, hôm nay chúng ta sẽ nói về AI.',\n    voice: 'Charon',\n    audioTags: ['determination'],\n});\n\nconsole.log(result.audio);  // Buffer (WAV)\nconsole.log(result.format); // 'wav'\n```\n\n**Method: `generateBatch(segments, options)` → `Array<Result>`**\n\nSinh audio cho nhiều segments với concurrency control. Failed segments được trả về với `error` property thay vì throw.\n\n```javascript\nconst results = await gen.generateBatch(plan.segments, { concurrency: 3 });\n\nresults.forEach(r => {\n    if (r.error) {\n        console.error(`Segment ${r.id} failed: ${r.error}`);\n    } else {\n        console.log(`Segment ${r.id}: ${r.audio.length} bytes`);\n    }\n});\n```\n\n| Option | Type | Default | Mô tả |\n|--------|------|---------|-------|\n| `concurrency` | number | `3` | Max concurrent API calls |\n\n**Request body format (Interactions API):**\n\n```json\n{\n  \"model\": \"gemini-3.1-flash-tts-preview\",\n  \"input\": \"[neutral] Xin chào các bạn...\",\n  \"response_format\": { \"type\": \"audio\" },\n  \"generation_config\": {\n    \"speech_config\": [{ \"voice\": \"Charon\" }]\n  }\n}\n```\n\nAudio tags được nhúng inline dạng `[tag]` markers trong input text. Voice config là array để hỗ trợ multi-speaker.\n\n---\n\n### LiveTTSGenerator\n\nGọi **Gemini Live API (WebSocket)** để sinh audio từ text segments. Hỗ trợ 3 models Live API với auto-fallback chain.\n\n```javascript\nconst { LiveTTSGenerator } = require('@andy-toolforge/tts-generator');\n\nconst gen = new LiveTTSGenerator({\n    apiKey: process.env.GEMINI_API_KEY,\n    live: {\n        models: [\n            'gemini-live-2.5-flash-native-audio',  // Primary\n            'gemini-3.1-flash-live-preview',        // Fallback 1\n            'gemini-3.5-live-translate-preview',     // Fallback 2\n        ],\n    },\n    maxRetries: 2,\n    baseDelay: 1000,\n});\n```\n\n**Constructor: `new LiveTTSGenerator(config)`**\n\n| Config | Type | Default | Mô tả |\n|--------|------|---------|-------|\n| `apiKey` | string | `GEMINI_API_KEY` env | Gemini API key (required) |\n| `live.models` | string[] | `[gemini-live-2.5-flash-native-audio, gemini-3.1-flash-live-preview, gemini-3.5-live-translate-preview]` | Model chain (auto-fallback) |\n| `maxRetries` | number | `2` | Retries per model trước khi fallback |\n| `baseDelay` | number | `1000` | Exponential backoff base (ms) |\n| `WebSocket` | Function | Native `WebSocket` | Custom WS constructor (cho test) |\n\n**Protocol flow:**\n\n```\nClient → Server (WebSocket):\n  1. setup:  { model, generationConfig: { responseModalities, speechConfig } }\n  2. clientContent: { turns: [{ role, parts }], turnComplete: true }\n\nServer → Client:\n  1. setupComplete: {}\n  2. serverContent: { modelTurn: { parts: [{ inlineData: { mimeType, data: base64 } }] } }\n```\n\n**Method: `generate(segment)` → `{ id, text, audio: Buffer, voice, format }`**\n\nMở WebSocket → send setup → gửi text → nhận audio chunks → close.\n\n```javascript\nconst result = await gen.generate({\n    id: 1,\n    text: 'Xin chào các bạn, hôm nay chúng ta sẽ nói về AI.',\n    voice: 'Charon',\n});\n\nconsole.log(result.audio);   // Buffer (WAV/L16)\nconsole.log(result.format);  // 'wav' or 'l16'\n```\n\nAudio response có thể là WAV (`audio/wav`) hoặc PCM16 (`audio/L16`) tùy model và config.\n\n**Method: `generateBatch(segments, options)` → `Array<Result>`**\n\nMở WebSocket riêng cho mỗi segment. Failed segments trả về với `error` property.\n\n```javascript\nconst results = await gen.generateBatch(plan.segments, { concurrency: 2 });\n\nresults.forEach(r => {\n    if (r.error) {\n        console.error(`Segment ${r.id} failed: ${r.error}`);\n    }\n});\n```\n\n| Option | Type | Default | Mô tả |\n|--------|------|---------|-------|\n| `concurrency` | number | `2` | Max concurrent WebSocket connections |\n\n**Model fallback chain:**\n\nKhi một model trả về lỗi (quota, rate limit, server error), `LiveTTSGenerator` tự động fallback sang model tiếp theo trong danh sách. Nếu hết model, trả về error.\n\n```javascript\n// Auto-fallback: nếu gemini-live-2.5-flash-native-audio bị 429\n// → gemini-3.1-flash-live-preview → gemini-3.5-live-translate-preview\n```\n\n**Voice config trong Live API:**\n\n```javascript\n// Voice được cấu hình qua speechConfig trong setup message:\n{\n    setup: {\n        model: 'models/gemini-live-2.5-flash-native-audio',\n        generationConfig: {\n            responseModalities: ['AUDIO'],\n            speechConfig: {\n                voiceConfig: {\n                    prebuiltVoiceConfig: {\n                        voiceName: 'Charon'\n                    }\n                }\n            }\n        }\n    }\n}\n```\n\nCác voice tương thích với Live API: Zephyr, Puck, Charon, Kore, Fenrir, Leda, Orus, Aoede, Callirrhoe, Autonoe, Enceladus, Iapetus, Umbriel, Algieba, Despina, Erinome, Algenib, Rasalgethi, Laomedeia, Achernar, Alnilam, Schedar, Gacrux, Pulcherrima, Achird, Zubenelgenubi, Vindemiatrix, Sadachbia, Sadaltager, Sulafat.\n\n---\n\n### OutputFormatter\n\nĐịnh dạng output từ segments + audio buffers.\n\n```javascript\nconst { OutputFormatter } = require('@andy-toolforge/tts-generator');\nconst formatter = new OutputFormatter();\n```\n\n**`formatBatch(segments, audioBuffers)`** → `{ segments: Array }`\n\nTrả về mảng segment-audio pairs.\n\n```javascript\nconst batch = formatter.formatBatch(plan.segments, audioBuffers);\n// batch.segments[0] → { id: 1, text: '...', audio: Buffer, voice: 'Charon', duration: 30 }\n```\n\n**`formatSingle(audioBuffers)`** → `Buffer`\n\nNối tất cả audio thành một file duy nhất. WAV-aware: tự động strip RIFF headers từ các file phụ và update size fields.\n\n```javascript\nconst combined = formatter.formatSingle(audioBuffers);\n// combined → Buffer of concatenated WAV audio\n```\n\n**`formatStream(segments, audioBuffers)`** → `AsyncGenerator`\n\nYields từng segment-audio pair dạng async generator.\n\n```javascript\nfor await (const item of formatter.formatStream(plan.segments, audioBuffers)) {\n    console.log(`Segment ${item.id}: ${item.audio.length} bytes`);\n}\n```\n\n---\n\n## Tutorial: Tạo Podcast Tự Động (A→Z)\n\n```javascript\nconst { TTSPlanner, TTSGenerator, OutputFormatter } = require('@andy-toolforge/tts-generator');\n\nasync function producePodcastAudio(script, title, options = {}) {\n    // 1. Segment script\n    const planner = new TTSPlanner({ llm: options.llm });\n    const plan = await planner.plan(script, title, {\n        voice: options.voice || 'auto',\n        pace: options.pace || 'normal',\n    });\n    console.log(`→ ${plan.segments.length} segments, ~${plan.metadata.totalEstimatedDuration}s`);\n\n    // 2. Generate audio\n    const gen = new TTSGenerator({\n        apiKey: process.env.GEMINI_API_KEY,\n        tts: { model: options.model },\n    });\n    const results = await gen.generateBatch(plan.segments, { concurrency: 3 });\n\n    // Check errors\n    const errors = results.filter(r => r.error);\n    if (errors.length > 0) {\n        console.warn(`⚠ ${errors.length} segments failed:`, errors.map(e => `${e.id}: ${e.error}`));\n    }\n\n    // 3. Output\n    const formatter = new OutputFormatter();\n    const successful = results.filter(r => !r.error);\n\n    const batch = formatter.formatBatch(\n        plan.segments.filter(s => !errors.find(e => e.id === s.id)),\n        successful.map(r => r.audio),\n    );\n\n    const single = formatter.formatSingle(successful.map(r => r.audio));\n\n    return {\n        segments: batch.segments,\n        combinedAudio: single,\n        metadata: plan.metadata,\n        errors,\n    };\n}\n\n// Usage\nconst result = await producePodcastAudio(\n    'Xin chào các bạn...\\n\\nHôm nay chúng ta sẽ nói về AI...',\n    'Tập 1: AI và Cuộc Sống',\n    { voice: 'Charon', pace: 'normal' }\n);\n\n// Lưu audio\nconst fs = require('fs');\nfs.writeFileSync('podcast.wav', result.combinedAudio);\n```\n\n## MCP Tools\n\nKhi dùng với [@andy-toolforge/mcp](https://npmjs.com/package/@andy-toolforge/mcp), package này auto-discover 2 tools:\n\n| Tool | Description |\n|------|-------------|\n| `generate_tts` | Full TTS pipeline: script → planner → TTS → output (batch/single/stream modes) |\n| `list_tts_voices` | List all 30 Gemini TTS voices with descriptions and style guides |\n\nKhông cần config thêm — tools tự xuất hiện khi MCP server start.\n\n### generate_tts parameters\n\n| Param | Type | Default | Mô tả |\n|-------|------|---------|-------|\n| `script` | string | required | Full podcast script |\n| `title` | string | required | Episode title |\n| `voice` | string | `\"auto\"` | Voice override (30 options) |\n| `mode` | string | `\"batch\"` | `batch` \\| `single` \\| `stream` |\n| `language` | string | `\"auto\"` | `vi` \\| `en` \\| `auto` |\n| `pace` | string | `\"normal\"` | `slow` \\| `normal` \\| `fast` |\n| `tags` | string | `\"\"` | Comma-separated audio tags |\n| `api_mode` | string | `\"interactions\"` | `interactions` (REST TTS API) \\| `live` (WebSocket Live API) |\n\n## Gemini TTS Voice List (30 voices)\n\n| Voice | Style | Mô tả |\n|-------|-------|-------|\n| Zephyr | Bright | Energetic, positive — great for introductions |\n| Puck | Upbeat | Playful, lively — good for light-hearted segments |\n| Charon | Informative | Educational, calm — ideal for explanatory narration |\n| Kore | Firm | Assertive, authoritative — strong for persuasive content |\n| Fenrir | Excitable | Passionate, enthusiastic — high-energy delivery |\n| Leda | Youthful | Fresh, young — good for casual/younger audience |\n| Orus | Firm | Steady, grounded — warmer alternative to Kore |\n| Aoede | Breezy | Light, airy — effortless narration style |\n| Callirrhoe | Easy-going | Relaxed, conversational — natural dialogue feel |\n| Autonoe | Bright | Radiant, clear — similar to Zephyr with softer edge |\n| Enceladus | Breathy | Intimate — good for emotional/personal segments |\n| Iapetus | Clear | Crisp, precise — excellent for technical content |\n| Umbriel | Easy-going | Laid-back, unhurried — slow-paced narration |\n| Algieba | Smooth | Velvety, polished — luxurious listening experience |\n| Despina | Smooth | Silky, flowing — seamless narration flow |\n| Erinome | Clear | Bright-clear hybrid — articulate with warmth |\n| Algenib | Gravelly | Raspy, textured — distinctive character voice |\n| Rasalgethi | Informative | Deep, knowledgeable — authoritative explainer |\n| Laomedeia | Upbeat | Bouncy, cheerful — energetic short segments |\n| Achernar | Soft | Gentle, whispery — quiet introspective moments |\n| Alnilam | Firm | Bold, commanding — strong narrative presence |\n| Schedar | Even | Balanced, neutral — consistent all-purpose voice |\n| Gacrux | Mature | Seasoned, wise — older authoritative tone |\n| Pulcherrima | Forward | Direct, engaging — keeps listener attention |\n| Achird | Friendly | Warm, approachable — like a trusted friend |\n| Zubenelgenubi | Casual | Informal, everyday — relaxed conversation |\n| Vindemiatrix | Gentle | Tender, soothing — calming narration |\n| Sadachbia | Lively | Spirited, animated — lively storytelling |\n| Sadaltager | Knowledgeable | Well-informed, measured — expert narrator |\n| Sulafat | Warm | Rich, inviting — classic storytelling warmth |\n\n## Supported Languages\n\nGemini TTS hỗ trợ 70+ languages, tự động detect từ input text. Bao gồm:\n\n- **Vietnamese** (vi) — hỗ trợ đầy đủ\n- **English** (en) — hỗ trợ đầy đủ\n- **Chinese (Mandarin)** (cmn)\n- **Japanese** (ja), **Korean** (ko)\n- **French** (fr), **German** (de), **Spanish** (es)\n- Và nhiều hơn nữa — xem [Google AI TTS docs](https://ai.google.dev/gemini-api/docs/interactions/speech-generation)\n\n## Integration với các packages khác\n\n- **+ [@andy-toolforge/core](https://npmjs.com/package/@andy-toolforge/core):** Cung cấp LLMClient cho TTSPlanner\n- **+ [@andy-toolforge/mcp](https://npmjs.com/package/@andy-toolforge/mcp):** Expose tools qua MCP protocol\n- **+ [@andy-toolforge/footage-generation](https://npmjs.com/package/@andy-toolforge/footage-generation):** Kết hợp TTS với image generation cho podcast video\n\n## Architecture\n\n```\nScript Text\n    │\n    ▼\nTTSPlanner (LLM + regex fallback)\n    │  Chia script → logical segments\n    │\n    ├─ api_mode: \"interactions\" ────────────────────────── api_mode: \"live\" ───\n    │                                                                         │\n    ▼                                                                         ▼\nTTSGenerator (Gemini Interactions REST)          LiveTTSGenerator (Gemini Live WebSocket)\n    │  wss://generativelanguage.googleapis.com/       │  wss://...BidiGenerateContent?key=\n    │  ws/google.ai.generativelanguage.v1beta.        │\n    │  GenerativeService.BidiGenerateContent          │\n    │                                                │\n    ▼                                                ▼\nAudio Buffers (WAV / PCM16)                    Audio Buffers (WAV / L16)\n    │                                                │\n    └──────────────────┬─────────────────────────────┘\n                       ▼\n        OutputFormatter.formatBatch()   → { segments: [{ id, text, audio }] }\n        OutputFormatter.formatSingle()  → Buffer (concatenated WAV)\n        OutputFormatter.formatStream()  → AsyncGenerator<Segment>\n```\n\n## API Mode Comparison\n\n| Aspect | Interactions (REST) | Live (WebSocket) |\n|--------|--------------------|--------------------|\n| Transport | HTTP POST (REST) | WebSocket bidirectional |\n| Models | `gemini-*-tts-preview` | `gemini-live-*-native-audio`, `gemini-*-live-*` |\n| Audio format | WAV (audio/wav) | WAV or L16 PCM (audio/L16) |\n| Voice config | `speech_config[]` array | `speechConfig.voiceConfig.prebuiltVoiceConfig` |\n| Audio tags | `[tag]` inline in text | N/A (conversation-based) |\n| Multi-speaker | `speech_config[]` objects | Per-turn config |\n| Concurrency | HTTP connection pool | Per-segment WebSocket connection |\n| Best for | Batch TTS, pre-recorded content | Real-time dialog, interactive voice |\n\n## Live API Models\n\n| Model ID | Description |\n|----------|-------------|\n| `gemini-live-2.5-flash-native-audio` | Flash model with native audio I/O — best quality TTS |\n| `gemini-3.1-flash-live-preview` | Live API variant of Gemini 3.1 Flash — fast, general-purpose |\n| `gemini-3.5-live-translate-preview` | Live translation model — supports real-time translation + TTS |\n\n## Related\n\n- [@andy-toolforge/core](https://npmjs.com/package/@andy-toolforge/core) — Nền tảng LLM client\n- [@andy-toolforge/mcp](https://npmjs.com/package/@andy-toolforge/mcp) — MCP server với plugin discovery\n- [Google Gemini TTS API](https://ai.google.dev/gemini-api/docs/interactions/speech-generation) — Official Interactions API docs\n- [Google Gemini Live API](https://ai.google.dev/gemini-api/docs/live-api) — Official Live API docs\n","readmeFilename":"README.md"}