{"_id":"@cogitator-ai/voice","_rev":"7-f81637ccc002012ed445bb80c4e1faad","name":"@cogitator-ai/voice","dist-tags":{"latest":"0.1.13"},"versions":{"0.1.0":{"name":"@cogitator-ai/voice","version":"0.1.0","license":"MIT","_id":"@cogitator-ai/voice@0.1.0","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"c93ccc17385b7d96f68cbe810cf0d40a53b0a477","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.0.tgz","fileCount":2,"integrity":"sha512-5iztkohTIlDLD0cGolatA3C71HtoZuP5iLQGZ4aXGyzTZloPG141QYmP+NIlRXrDQlQtZ/w1zMGYPblUi97Osw==","signatures":[{"sig":"MEQCIDk+0+2k2IvHyrv/1vccxDBazJMOjijDYpP6bwyjwyYYAiBnrYdYrqJZukzAzRFpPjb/phmqlqkxgt/kZ/jT+XDqoA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":20210},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"deb50c53ca1cf197a48175fbfc89b39b7c8e9069","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/voice"},"_npmVersion":"10.9.0","description":"Voice/Realtime agent capabilities for Cogitator","directories":{},"_nodeVersion":"22.11.0","dependencies":{"ws":"^8.18.0","zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"workspace:*"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","@types/ws":"^8.5.0","typescript":"^5.7.2"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"workspace:*"},"peerDependenciesMeta":{"openai":{"optional":true},"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.1.0_1771680216820_0.03931929593800465","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@cogitator-ai/voice","version":"0.1.1","license":"MIT","_id":"@cogitator-ai/voice@0.1.1","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"ee0a8ee30dcb0dab498d2c2e1d5ac1397f38ab2d","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.1.tgz","fileCount":94,"integrity":"sha512-I4zxVBkZLcptnpY7Ag6cslMjIi2FE41reHUBMiiPg2c3Zf6ygmPETx69z92rpkOnzMVaTG0LTlwaB2RPukW7JA==","signatures":[{"sig":"MEUCIQDNIIUAJf65/4vQ5C+hcNaoi8g1uQyyrUzM6ssBEfib/gIgPqDByxNT13aDlMlBKwnhF1MNc5WOJ61KOFfNcF3NMO4=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":148563},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"0b719284288ff40d38c03b3d5afadb96d3749b1d","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/voice"},"_npmVersion":"10.9.0","description":"Voice/Realtime agent capabilities for Cogitator","directories":{},"_nodeVersion":"22.11.0","dependencies":{"ws":"^8.18.0","zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"workspace:*"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","@types/ws":"^8.5.0","typescript":"^5.7.2"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"workspace:*"},"peerDependenciesMeta":{"openai":{"optional":true},"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.1.1_1771680414625_0.3701472530149996","host":"s3://npm-registry-packages-npm-production"}},"0.1.4":{"name":"@cogitator-ai/voice","version":"0.1.4","license":"MIT","_id":"@cogitator-ai/voice@0.1.4","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"38fcc5b768a3271d6960da43a4a2978f8703bb44","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.4.tgz","fileCount":94,"integrity":"sha512-MzSYxgIx1ErirrOFOiucjfVY/Oqk5U6/JDraDMX8f0MN8r7yL+Yon/BZbT1kTznCTEcoHBWacl2Jt2i5qW9x5Q==","signatures":[{"sig":"MEUCIDWxcxyISjMqouV9WcuLEaZ7wpUMHu5BxxBAcpfsCVToAiEA88mRvHU94vrxXPM/fTG0J9YV683exAqzPsXJvRR6yLQ=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":148563},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"f508be4a1a30c7bd136f0fa8b2e6cee5521416ab","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/voice"},"_npmVersion":"10.9.0","description":"Voice/Realtime agent capabilities for Cogitator","directories":{},"_nodeVersion":"22.11.0","dependencies":{"ws":"^8.18.0","zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"workspace:*"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","@types/ws":"^8.5.0","typescript":"^5.7.2"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"workspace:*"},"peerDependenciesMeta":{"openai":{"optional":true},"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.1.4_1772035830892_0.2779300113235137","host":"s3://npm-registry-packages-npm-production"}},"0.1.7":{"name":"@cogitator-ai/voice","version":"0.1.7","license":"MIT","_id":"@cogitator-ai/voice@0.1.7","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"c88053efa4b65bbfc7ccea2ebc45b5eda44c989c","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.7.tgz","fileCount":94,"integrity":"sha512-+SE8XKwDWo56PjGMJje8sd0FZAnLhvJHFHvAtbssrNN7Pt+JpsfBEhX99QI36ateVnmRwt4Y4ofZjTGBaEBYWQ==","signatures":[{"sig":"MEYCIQCcjVyUvlotN9DGIncRzqKezT/tV8JV2gt9JiyeovoaKAIhAIFbnAZI2lbCTvmSv4gDSiVWnPK9tMXO15ob5rCWao1y","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":155982},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"467575702503c3581cb717f8aef8a72e1d8ae492","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/voice"},"_npmVersion":"10.9.0","description":"Voice/Realtime agent capabilities for Cogitator","directories":{},"_nodeVersion":"22.11.0","dependencies":{"ws":"^8.18.0","zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"workspace:*"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","@types/ws":"^8.5.0","typescript":"^5.7.2"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"workspace:*"},"peerDependenciesMeta":{"openai":{"optional":true},"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.1.7_1772067997422_0.5053291260592423","host":"s3://npm-registry-packages-npm-production"}},"0.1.11":{"name":"@cogitator-ai/voice","version":"0.1.11","license":"MIT","_id":"@cogitator-ai/voice@0.1.11","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"45c0af907534ba6a1c1bc36beb2975c1ebf1dec8","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.11.tgz","fileCount":94,"integrity":"sha512-S6vr7A1Cg39bGJ2OWi8Z7B/bzH7QCULZybRSh7o5lWh6I/QIuJ1uCjwJkbYm7yQLdfuP0HkQElSo8RYaKwHVdw==","signatures":[{"sig":"MEUCIAdS51ywIx8/UqBkCNFj13KFhw/nSRFXjYJ0ER4fG1CSAiEA/vBYtYtJtumtsYZlxFcDIC6zqWbtYmRfT96xc0AX7yo=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":156000},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"33b6a8cb29bfdd7d2255acfa240b2df5d558354e","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/voice"},"_npmVersion":"10.9.0","description":"Voice/Realtime agent capabilities for Cogitator","directories":{},"_nodeVersion":"22.11.0","dependencies":{"ws":"^8.18.0","zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"workspace:*"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","@types/ws":"^8.5.0","typescript":"^5.7.2"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"workspace:*"},"peerDependenciesMeta":{"openai":{"optional":true},"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.1.11_1778881429554_0.5017458732419826","host":"s3://npm-registry-packages-npm-production"}},"0.1.12":{"name":"@cogitator-ai/voice","version":"0.1.12","license":"MIT","_id":"@cogitator-ai/voice@0.1.12","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"0f67cdb08f88a8cfeabd8519d5997d271fab5992","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.12.tgz","fileCount":95,"integrity":"sha512-C9bPcRCgjAJI+EMIA9Fdoc4aTBYaTzwtF3XqcoUzdFAxTVFStoMEZU1gb3Rh7radFV7szktZL/RiXuY+AMhkcQ==","signatures":[{"sig":"MEUCIA3HssuCmCZ6FLVvaVLw0mxPzHjpfe6vIvn0eIKTTon8AiEAxtMG9ZJOrvTuWWJaUiIyZtrvsEGnEkSPQzpUimebnoU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":157068},"main":"./dist/index.js","type":"module","_from":"file:cogitator-ai-voice-0.1.12.tgz","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"_resolved":"/private/var/folders/20/4d95hdvj45j337gtqpbx3v1m0000gn/T/c3fc8c1d1fdc3ec45daf129e8f4927e4/cogitator-ai-voice-0.1.12.tgz","_integrity":"sha512-C9bPcRCgjAJI+EMIA9Fdoc4aTBYaTzwtF3XqcoUzdFAxTVFStoMEZU1gb3Rh7radFV7szktZL/RiXuY+AMhkcQ==","repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/voice"},"_npmVersion":"10.9.0","description":"Voice/Realtime agent capabilities for Cogitator","directories":{},"_nodeVersion":"22.11.0","dependencies":{"ws":"^8.18.0","zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"0.22.2"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","@types/ws":"^8.5.0","typescript":"^5.7.2"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"0.19.3"},"peerDependenciesMeta":{"openai":{"optional":true},"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.1.12_1778882317120_0.47914976579876667","host":"s3://npm-registry-packages-npm-production"}},"0.1.13":{"name":"@cogitator-ai/voice","version":"0.1.13","description":"Voice/Realtime agent capabilities for Cogitator","type":"module","main":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"dependencies":{"nanoid":"^5.0.4","ws":"^8.18.0","zod":"^4.3.6","@cogitator-ai/types":"0.22.3"},"peerDependencies":{"openai":"^6.0.0","@cogitator-ai/core":"0.19.4"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true},"openai":{"optional":true}},"devDependencies":{"@types/ws":"^8.5.0","typescript":"^5.7.2","vitest":"^4.0.18"},"repository":{"type":"git","url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","directory":"packages/voice"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"license":"MIT","scripts":{"build":"tsc","dev":"tsc --watch","clean":"rm -rf dist","typecheck":"tsc --noEmit","test":"vitest run","test:watch":"vitest"},"_id":"@cogitator-ai/voice@0.1.13","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","_integrity":"sha512-/rLLlQbyIesHBKwL2rah+A4rp5tVP2ks98RQaNROG0expArI6K7xdjzFD5GBPUBwkb0ZUUH4SqSYTiicEXFQEQ==","_resolved":"/private/var/folders/20/4d95hdvj45j337gtqpbx3v1m0000gn/T/3ccd9dd0e4a5e1c32ab9f305a65c0cb6/cogitator-ai-voice-0.1.13.tgz","_from":"file:cogitator-ai-voice-0.1.13.tgz","_nodeVersion":"22.23.1","_npmVersion":"10.9.8","dist":{"integrity":"sha512-/rLLlQbyIesHBKwL2rah+A4rp5tVP2ks98RQaNROG0expArI6K7xdjzFD5GBPUBwkb0ZUUH4SqSYTiicEXFQEQ==","shasum":"8ee519512ff5c3574773b47e4f191b0451cfb83f","tarball":"https://registry.npmjs.org/@cogitator-ai/voice/-/voice-0.1.13.tgz","fileCount":95,"unpackedSize":180277,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQCbQQf+D3iIIFZfxlr5SAEM/P7helFga2OgiqsUU6nmKAIhAKLYS7aiLym/SHQ/bjEfHcdxv7H04u0aTyiXjr4b9noC"}]},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"directories":{},"maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/voice_0.1.13_1785059729664_0.7873960102985207"},"_hasShrinkwrap":false}},"time":{"created":"2026-02-21T13:23:36.725Z","modified":"2026-07-26T09:55:30.005Z","0.1.0":"2026-02-21T13:23:36.952Z","0.1.1":"2026-02-21T13:26:54.758Z","0.1.4":"2026-02-25T16:10:31.041Z","0.1.7":"2026-02-26T01:06:37.573Z","0.1.11":"2026-05-15T21:43:49.693Z","0.1.12":"2026-05-15T21:58:37.350Z","0.1.13":"2026-07-26T09:55:29.802Z"},"bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"license":"MIT","homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","repository":{"type":"git","url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","directory":"packages/voice"},"description":"Voice/Realtime agent capabilities for Cogitator","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"readme":"# @cogitator-ai/voice\n\nVoice and Realtime agent capabilities for [Cogitator](https://github.com/cogitator-ai/cogitator).\n\nTwo modes: **Pipeline** (STT -> Agent -> TTS) for any LLM, and **Realtime** (native speech-to-speech) for OpenAI/Gemini.\n\n## Installation\n\n```bash\npnpm add @cogitator-ai/voice\n\n# Required for OpenAI STT/TTS\npnpm add openai\n\n# Optional dependencies\npnpm add onnxruntime-node  # Silero VAD (neural network-based)\n```\n\n## Features\n\n- **Pipeline Mode** — STT -> Agent -> TTS, works with any Cogitator agent and LLM backend\n- **Realtime Mode** — Native speech-to-speech via OpenAI Realtime API or Gemini Live API\n- **2 STT Providers** — OpenAI (gpt-4o-mini-transcribe) and Deepgram (nova-3, real-time streaming)\n- **2 TTS Providers** — OpenAI (gpt-4o-mini-tts) and ElevenLabs (eleven_flash_v2_5, ~75ms latency)\n- **2 VAD Providers** — Energy-based (zero deps) and Silero (ONNX neural network)\n- **WebSocket Transport** — Built-in server for browser/mobile clients\n- **Agent Tools** — Drop-in `transcribe_audio` and `speak_text` tools for any Cogitator agent\n- **Interruption Handling** — Barge-in support in both pipeline and realtime modes\n- **Audio Utilities** — PCM/WAV conversion, resampling, RMS calculation\n\n---\n\n## Quick Start\n\n```typescript\nimport { VoiceAgent, OpenAISTT, OpenAITTS } from '@cogitator-ai/voice';\nimport { Agent } from '@cogitator-ai/core';\n\nconst agent = new Agent({ instructions: 'You are a helpful assistant' });\n\nconst voiceAgent = new VoiceAgent({\n  agent,\n  mode: 'pipeline',\n  stt: new OpenAISTT({ apiKey: process.env.OPENAI_API_KEY! }),\n  tts: new OpenAITTS({ apiKey: process.env.OPENAI_API_KEY! }),\n});\n\nawait voiceAgent.listen(8080);\n```\n\nConnect from any WebSocket client at `ws://localhost:8080/voice` — send binary audio frames, receive binary audio + JSON control messages.\n\n---\n\n## STT Providers\n\n| Provider      | Default Model            | Streaming           | Word Timestamps | Notes                                          |\n| ------------- | ------------------------ | ------------------- | --------------- | ---------------------------------------------- |\n| `OpenAISTT`   | `gpt-4o-mini-transcribe` | Buffered            | Yes             | Also supports `gpt-4o-transcribe`, `whisper-1` |\n| `DeepgramSTT` | `nova-3`                 | Real-time WebSocket | Yes             | Interim results, endpointing, auto-punctuation |\n\n### OpenAI STT\n\n```typescript\nimport { OpenAISTT } from '@cogitator-ai/voice';\n\nconst stt = new OpenAISTT({\n  apiKey: process.env.OPENAI_API_KEY!,\n  model: 'gpt-4o-mini-transcribe',\n});\n\nconst result = await stt.transcribe(audioBuffer, { language: 'en' });\nconsole.log(result.text);\nconsole.log(result.words);\nconsole.log(result.duration);\n```\n\n### Deepgram STT\n\nReal-time streaming with interim results:\n\n```typescript\nimport { DeepgramSTT } from '@cogitator-ai/voice';\n\nconst stt = new DeepgramSTT({\n  apiKey: process.env.DEEPGRAM_API_KEY!,\n  model: 'nova-3',\n  language: 'en',\n});\n\nconst stream = stt.createStream({ interimResults: true, endpointing: 500 });\n\nstream.on('partial', (text) => {\n  console.log('partial:', text);\n});\n\nstream.on('final', (result) => {\n  console.log('final:', result.text);\n});\n\nstream.write(audioChunk1);\nstream.write(audioChunk2);\nawait stream.close();\n```\n\n---\n\n## TTS Providers\n\n| Provider        | Default Model       | Streaming | Voices                                                                                 | Notes                                                                                   |\n| --------------- | ------------------- | --------- | -------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |\n| `OpenAITTS`     | `gpt-4o-mini-tts`   | Yes       | alloy, ash, ballad, coral, echo, fable, onyx, nova, sage, shimmer, verse, marin, cedar | Also supports `tts-1`, `tts-1-hd`. Supports `instructions` for voice style control      |\n| `ElevenLabsTTS` | `eleven_flash_v2_5` | Yes       | By voice ID                                                                            | ~75ms latency. Also supports `eleven_turbo_v2_5`, `eleven_multilingual_v2`, `eleven_v3` |\n\n### OpenAI TTS\n\n```typescript\nimport { OpenAITTS } from '@cogitator-ai/voice';\n\nconst tts = new OpenAITTS({\n  apiKey: process.env.OPENAI_API_KEY!,\n  model: 'gpt-4o-mini-tts',\n  voice: 'coral',\n});\n\nconst audio = await tts.synthesize('Hello, world!', {\n  speed: 1.0,\n  format: 'mp3',\n  instructions: 'Speak in a warm, friendly tone',\n});\n\nfor await (const chunk of tts.streamSynthesize('Streaming response...')) {\n  process.stdout.write('.');\n}\n```\n\n### ElevenLabs TTS\n\n```typescript\nimport { ElevenLabsTTS } from '@cogitator-ai/voice';\n\nconst tts = new ElevenLabsTTS({\n  apiKey: process.env.ELEVENLABS_API_KEY!,\n  model: 'eleven_flash_v2_5',\n  voiceId: '21m00Tcm4TlvDq8ikWAM',\n});\n\nconst audio = await tts.synthesize('Hello!', { format: 'mp3' });\n\nfor await (const chunk of tts.streamSynthesize('Streaming...')) {\n  process.stdout.write('.');\n}\n```\n\n---\n\n## VAD Providers\n\n| Provider    | Accuracy | Dependencies       | Speed      | Notes                                             |\n| ----------- | -------- | ------------------ | ---------- | ------------------------------------------------- |\n| `EnergyVAD` | Basic    | None               | Fast       | RMS energy threshold, good for quiet environments |\n| `SileroVAD` | High     | `onnxruntime-node` | ~3ms/frame | Neural network, works in noisy environments       |\n\n### Energy VAD\n\nZero-dependency voice activity detection based on audio energy levels:\n\n```typescript\nimport { EnergyVAD } from '@cogitator-ai/voice';\n\nconst vad = new EnergyVAD({\n  threshold: 0.01,\n  silenceDuration: 500,\n  sampleRate: 16000,\n});\n\nconst event = vad.process(float32Samples);\n\nswitch (event.type) {\n  case 'speech_start':\n    console.log('User started speaking');\n    break;\n  case 'speech_end':\n    console.log(`Speech ended after ${event.duration}ms`);\n    break;\n  case 'speech':\n    console.log(`Speech probability: ${event.probability}`);\n    break;\n  case 'silence':\n    break;\n}\n```\n\n### Silero VAD\n\nNeural network-based VAD using the Silero ONNX model:\n\n```typescript\nimport { SileroVAD } from '@cogitator-ai/voice';\n\nconst vad = new SileroVAD({\n  modelPath: './silero_vad.onnx',\n  threshold: 0.5,\n  silenceDuration: 500,\n  sampleRate: 16000,\n});\n\nawait vad.init();\n\nconst event = await vad.process(float32Samples);\n```\n\n---\n\n## Pipeline Mode\n\nThe pipeline processes audio through a three-stage loop: STT -> Agent -> TTS. Works with any Cogitator agent regardless of the underlying LLM.\n\n### One-shot processing\n\n```typescript\nimport { VoicePipeline, OpenAISTT, OpenAITTS } from '@cogitator-ai/voice';\n\nconst pipeline = new VoicePipeline({\n  stt: new OpenAISTT({ apiKey: process.env.OPENAI_API_KEY! }),\n  tts: new OpenAITTS({ apiKey: process.env.OPENAI_API_KEY! }),\n  agent: myAgent,\n});\n\nconst result = await pipeline.process(audioBuffer);\nconsole.log(result.transcript);\nconsole.log(result.response);\n// result.audio — synthesized response audio\n```\n\n### Streaming sessions\n\nFor continuous conversations with VAD-driven turn detection:\n\n```typescript\nimport { VoicePipeline, OpenAISTT, OpenAITTS, EnergyVAD } from '@cogitator-ai/voice';\n\nconst pipeline = new VoicePipeline({\n  stt: new OpenAISTT({ apiKey: process.env.OPENAI_API_KEY! }),\n  tts: new OpenAITTS({ apiKey: process.env.OPENAI_API_KEY! }),\n  vad: new EnergyVAD({ threshold: 0.01, silenceDuration: 500 }),\n  agent: myAgent,\n});\n\nconst session = pipeline.createSession();\n\nsession.on('speech_start', () => {\n  console.log('User started speaking');\n});\n\nsession.on('transcript', (text, isFinal) => {\n  console.log(isFinal ? `Final: ${text}` : `Interim: ${text}`);\n});\n\nsession.on('agent_response', (text) => {\n  console.log('Agent:', text);\n});\n\nsession.on('audio', (chunk) => {\n  playAudio(chunk);\n});\n\nsession.pushAudio(pcm16Chunk);\n\nsession.interrupt();\n\nawait session.close();\n```\n\n---\n\n## Realtime Mode\n\nNative speech-to-speech without the STT/TTS pipeline. The LLM directly processes and generates audio. Lower latency, more natural conversation flow.\n\n### OpenAI Realtime\n\n```typescript\nimport { RealtimeSession } from '@cogitator-ai/voice';\n\nconst session = new RealtimeSession({\n  provider: 'openai',\n  apiKey: process.env.OPENAI_API_KEY!,\n  model: 'gpt-4o-mini-realtime-preview',\n  voice: 'coral',\n  instructions: 'You are a helpful assistant.',\n  tools: [\n    {\n      name: 'get_weather',\n      description: 'Get current weather',\n      parameters: { type: 'object', properties: { city: { type: 'string' } } },\n      execute: async (args) => ({ temp: 72, unit: 'F' }),\n    },\n  ],\n});\n\nsession.on('connected', () => console.log('Connected'));\nsession.on('audio', (chunk) => playAudio(chunk));\nsession.on('transcript', (text, role) => console.log(`${role}: ${text}`));\nsession.on('speech_start', () => console.log('VAD: speech detected'));\nsession.on('tool_call', (name, args) => console.log(`Tool: ${name}`, args));\n\nawait session.connect();\n\nsession.pushAudio(pcm16Chunk);\nsession.sendText('Hello!');\n\nsession.interrupt();\nsession.close();\n```\n\n### Gemini Live\n\n```typescript\nimport { RealtimeSession } from '@cogitator-ai/voice';\n\nconst session = new RealtimeSession({\n  provider: 'gemini',\n  apiKey: process.env.GOOGLE_API_KEY!,\n  model: 'gemini-live-2.5-flash-native-audio',\n  voice: 'Puck',\n  instructions: 'You are a helpful assistant.',\n});\n\nsession.on('audio', (chunk) => playAudio(chunk));\nsession.on('transcript', (text, role) => console.log(`${role}: ${text}`));\n\nawait session.connect();\nsession.pushAudio(pcm16Chunk);\n```\n\n---\n\n## WebSocket Transport\n\nBuilt-in WebSocket server for connecting browser/mobile clients. Binary frames carry audio, text frames carry JSON control messages.\n\n```typescript\nimport { WebSocketTransport, VoiceClient } from '@cogitator-ai/voice';\n\nconst transport = new WebSocketTransport({\n  path: '/voice',\n  maxConnections: 100,\n});\n\ntransport.on('connection', (client: VoiceClient) => {\n  console.log(`Client connected: ${client.id}`);\n\n  client.on('audio', (chunk) => {\n    // PCM16 audio from client\n  });\n\n  client.on('message', (msg) => {\n    // JSON control messages\n  });\n\n  client.on('close', () => {\n    console.log(`Client disconnected: ${client.id}`);\n  });\n\n  client.sendAudio(responseChunk);\n  client.sendMessage({ type: 'transcript', text: 'Hello' });\n});\n\nawait transport.listen(8080);\n\n// or attach to an existing HTTP server\ntransport.attachToServer(httpServer);\n\nawait transport.close();\n```\n\n### WebSocket Protocol\n\n| Direction        | Format      | Content                                                                             |\n| ---------------- | ----------- | ----------------------------------------------------------------------------------- |\n| Client -> Server | Binary      | PCM16 audio frames (16kHz, mono, 16-bit LE)                                         |\n| Client -> Server | Text (JSON) | Control messages                                                                    |\n| Server -> Client | Binary      | PCM16 or encoded audio response                                                     |\n| Server -> Client | Text (JSON) | `{ type: 'transcript' \\| 'agent_response' \\| 'speech_start' \\| 'speech_end', ... }` |\n\n---\n\n## VoiceAgent\n\nHigh-level class that wires together transport, pipeline/realtime, and session management:\n\n```typescript\nimport { VoiceAgent, OpenAISTT, OpenAITTS, EnergyVAD } from '@cogitator-ai/voice';\n\nconst voiceAgent = new VoiceAgent({\n  agent: myAgent,\n  mode: 'pipeline',\n  stt: new OpenAISTT({ apiKey: process.env.OPENAI_API_KEY! }),\n  tts: new OpenAITTS({ apiKey: process.env.OPENAI_API_KEY! }),\n  vad: new EnergyVAD(),\n  transport: { path: '/voice', maxConnections: 50 },\n});\n\nvoiceAgent.on('session_start', (id) => console.log(`Session started: ${id}`));\nvoiceAgent.on('session_end', (id) => console.log(`Session ended: ${id}`));\nvoiceAgent.on('error', (err) => console.error(err));\n\nconsole.log(`Active sessions: ${voiceAgent.activeSessions}`);\n\nawait voiceAgent.listen(8080);\nawait voiceAgent.close();\n```\n\n### Realtime mode with VoiceAgent\n\n```typescript\nconst voiceAgent = new VoiceAgent({\n  agent: myAgent,\n  mode: 'realtime',\n  realtimeProvider: 'openai',\n  realtimeApiKey: process.env.OPENAI_API_KEY!,\n  realtimeModel: 'gpt-4o-mini-realtime-preview',\n  voice: 'coral',\n});\n\nawait voiceAgent.listen(8080);\n```\n\n---\n\n## Agent Integration\n\nGive any Cogitator agent the ability to transcribe audio or synthesize speech:\n\n```typescript\nimport { Agent, tool } from '@cogitator-ai/core';\nimport { voiceTools, OpenAISTT, OpenAITTS } from '@cogitator-ai/voice';\nimport { z } from 'zod';\n\nconst [transcribe, speak] = voiceTools({\n  stt: new OpenAISTT({ apiKey: process.env.OPENAI_API_KEY! }),\n  tts: new OpenAITTS({ apiKey: process.env.OPENAI_API_KEY! }),\n});\n\nconst transcribeTool = tool({\n  name: transcribe.name,\n  description: transcribe.description,\n  parameters: z.object({\n    audioBase64: z.string().describe('Base64-encoded audio data'),\n    language: z.string().optional().describe('Language code'),\n  }),\n  execute: async (params) => transcribe.execute(params),\n});\n\nconst speakTool = tool({\n  name: speak.name,\n  description: speak.description,\n  parameters: z.object({\n    text: z.string().describe('Text to convert to speech'),\n    voice: z.string().optional().describe('Voice to use'),\n  }),\n  execute: async (params) => speak.execute(params),\n});\n\nconst agent = new Agent({\n  name: 'voice-assistant',\n  model: 'gpt-4o',\n  instructions: 'You can transcribe audio and generate speech.',\n  tools: [transcribeTool, speakTool],\n});\n```\n\n---\n\n## Audio Utilities\n\nLow-level audio conversion functions for working with PCM and WAV data:\n\n```typescript\nimport {\n  float32ToPcm16,\n  pcm16ToFloat32,\n  pcmToWav,\n  wavToPcm,\n  resample,\n  calculateRMS,\n} from '@cogitator-ai/voice';\n\nconst pcm = float32ToPcm16(float32Samples);\nconst floats = pcm16ToFloat32(pcmBuffer);\n\nconst wav = pcmToWav(pcmBuffer, 16000);\nconst { samples, sampleRate } = wavToPcm(wavBuffer);\n\nconst resampled = resample(float32Samples, 44100, 16000);\n\nconst rms = calculateRMS(float32Samples);\n```\n\n---\n\n## Configuration Reference\n\n### `VoiceAgentConfig`\n\n| Field              | Type                                     | Required      | Description                          |\n| ------------------ | ---------------------------------------- | ------------- | ------------------------------------ |\n| `agent`            | `{ run(input) => Promise<{ content }> }` | Yes           | Cogitator agent or compatible object |\n| `mode`             | `'pipeline' \\| 'realtime'`               | Yes           | Processing mode                      |\n| `stt`              | `STTProvider`                            | Pipeline only | Speech-to-text provider              |\n| `tts`              | `TTSProvider`                            | Pipeline only | Text-to-speech provider              |\n| `vad`              | `VADProvider`                            | No            | Voice activity detection             |\n| `realtimeProvider` | `'openai' \\| 'gemini'`                   | Realtime only | Realtime API provider                |\n| `realtimeApiKey`   | `string`                                 | Realtime only | API key for realtime provider        |\n| `realtimeModel`    | `string`                                 | No            | Model override                       |\n| `voice`            | `string`                                 | No            | Voice for TTS or realtime            |\n| `transport`        | `WebSocketTransportConfig`               | No            | Transport options                    |\n\n### `OpenAISTTConfig`\n\n| Field     | Type     | Default                  | Description         |\n| --------- | -------- | ------------------------ | ------------------- |\n| `apiKey`  | `string` | —                        | OpenAI API key      |\n| `model`   | `string` | `gpt-4o-mini-transcribe` | Model ID            |\n| `baseURL` | `string` | —                        | Custom API base URL |\n\n### `DeepgramSTTConfig`\n\n| Field      | Type     | Default  | Description           |\n| ---------- | -------- | -------- | --------------------- |\n| `apiKey`   | `string` | —        | Deepgram API key      |\n| `model`    | `string` | `nova-3` | Model ID              |\n| `language` | `string` | —        | Default language code |\n\n### `OpenAITTSConfig`\n\n| Field     | Type     | Default           | Description         |\n| --------- | -------- | ----------------- | ------------------- |\n| `apiKey`  | `string` | —                 | OpenAI API key      |\n| `model`   | `string` | `gpt-4o-mini-tts` | Model ID            |\n| `voice`   | `string` | `alloy`           | Default voice       |\n| `baseURL` | `string` | —                 | Custom API base URL |\n\n### `ElevenLabsTTSConfig`\n\n| Field     | Type     | Default                | Description        |\n| --------- | -------- | ---------------------- | ------------------ |\n| `apiKey`  | `string` | —                      | ElevenLabs API key |\n| `voiceId` | `string` | `21m00Tcm4TlvDq8ikWAM` | Default voice ID   |\n| `model`   | `string` | `eleven_flash_v2_5`    | Model ID           |\n\n### `EnergyVADConfig`\n\n| Field             | Type     | Default | Description                               |\n| ----------------- | -------- | ------- | ----------------------------------------- |\n| `threshold`       | `number` | `0.01`  | RMS energy threshold for speech detection |\n| `silenceDuration` | `number` | `500`   | Silence duration (ms) before `speech_end` |\n| `sampleRate`      | `number` | `16000` | Audio sample rate in Hz                   |\n\n### `SileroVADConfig`\n\n| Field             | Type     | Default | Description                               |\n| ----------------- | -------- | ------- | ----------------------------------------- |\n| `modelPath`       | `string` | —       | Path to `silero_vad.onnx` model file      |\n| `threshold`       | `number` | `0.5`   | Speech probability threshold (0-1)        |\n| `silenceDuration` | `number` | `500`   | Silence duration (ms) before `speech_end` |\n| `sampleRate`      | `number` | `16000` | Audio sample rate in Hz                   |\n\n### `WebSocketTransportConfig`\n\n| Field            | Type     | Default  | Description                    |\n| ---------------- | -------- | -------- | ------------------------------ |\n| `path`           | `string` | `/voice` | WebSocket endpoint path        |\n| `maxConnections` | `number` | `100`    | Maximum concurrent connections |\n\n---\n\n## Examples\n\nSee [`examples/voice/`](../../examples/voice/) for runnable examples:\n\n- **01-pipeline-basic.ts** — Basic pipeline with OpenAI STT + TTS\n- **02-realtime-openai.ts** — OpenAI Realtime API voice agent\n- **03-realtime-gemini.ts** — Gemini Live voice agent\n\n---\n\n## License\n\nMIT\n","readmeFilename":"README.md"}