{"_id":"@auditionhub/voice-detector","_rev":"5-ff14548a7b10a02dcd4f57e0cbcbc20c","name":"@auditionhub/voice-detector","dist-tags":{"latest":"0.1.4"},"versions":{"0.1.0":{"name":"@auditionhub/voice-detector","version":"0.1.0","keywords":["speech","vad","stt","web-speech","vosk","browser","react"],"author":{"name":"Craig Baker","email":"craig.d.baker@gmail.com"},"license":"MIT","_id":"@auditionhub/voice-detector@0.1.0","maintainers":[{"name":"craigdbaker","email":"craig.d.baker@gmail.com"}],"dist":{"shasum":"289d233a6c5a1ddb0b9ffdbd978d49bc7780a9fa","tarball":"https://registry.npmjs.org/@auditionhub/voice-detector/-/voice-detector-0.1.0.tgz","fileCount":8,"integrity":"sha512-IJ+Ocm3CSkippeHTAUQm6Gow/DSHCQuuKJe183y2E8KeO/ZdYp88EbAv84LjLno+tbiq3oetQGUvnoI0smsMOw==","signatures":[{"sig":"MEUCIG9JZ+I5CzR9ux9fRpfKqEl+fz8DD4WIRjhVms2aSZVbAiEAiJwjMfdqlyaJJbapACWudiHqnMit3ZVeuViWKYHJyxw=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":239607},"main":"dist/index.cjs","type":"module","types":"dist/index.d.ts","module":"dist/index.js","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"scripts":{"dev":"tsup --watch","lint":"eslint \"src/**/*.{ts,tsx}\"","test":"vitest","build":"tsup","clean":"rimraf dist","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"craigdbaker","email":"craig.d.baker@gmail.com"},"_npmVersion":"11.6.0","description":"Real-time line completion detection for browser apps (Web Speech default, Vosk-WASM fallback)","directories":{},"sideEffects":false,"_nodeVersion":"23.6.1","dependencies":{"assemblyai":"^4.18.5","@google/generative-ai":"^0.21.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.1","eslint":"^8.57.1","rimraf":"^6.0.1","vitest":"^2.1.3","typescript":"^5.6.3","@types/node":"^20.11.30","@typescript-eslint/parser":"^7.0.0","@typescript-eslint/eslint-plugin":"^7.0.0"},"_npmOperationalInternal":{"tmp":"tmp/voice-detector_0.1.0_1762007485489_0.9227744963778832","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@auditionhub/voice-detector","version":"0.1.1","keywords":["speech","vad","stt","web-speech","vosk","browser","react"],"author":{"name":"Craig Baker","email":"craig.d.baker@gmail.com"},"license":"MIT","_id":"@auditionhub/voice-detector@0.1.1","maintainers":[{"name":"craigdbaker","email":"craig.d.baker@gmail.com"}],"dist":{"shasum":"720689e0ab839b5c6bff67a6cee7108dd3b6c052","tarball":"https://registry.npmjs.org/@auditionhub/voice-detector/-/voice-detector-0.1.1.tgz","fileCount":8,"integrity":"sha512-kc6zG0Vu0djsL+OSXHWRqjKZUqTjllGwKdv6AO0DQNTNK/+xvu9JAzgyQdyCpDxLodhQaA8BoHQgzzs0wAewoA==","signatures":[{"sig":"MEYCIQCs7X4KeB2xblRpTfC7mfbIiyUwX7Wnv4W2BL+jjDTWZwIhANYMy8O8Q8no8DwyIwyjpFWTZ1muDOgXXlh0Mw6ymIZ0","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":239619},"main":"dist/index.cjs","type":"module","types":"dist/index.d.ts","module":"dist/index.js","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"gitHead":"add26fe96d375e52401ba82e63cb65d4f443553a","scripts":{"dev":"tsup --watch","lint":"eslint \"src/**/*.{ts,tsx}\"","test":"vitest","build":"tsup","clean":"rimraf dist","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"craigdbaker","email":"craig.d.baker@gmail.com"},"_npmVersion":"11.6.0","description":"Real-time line completion detection for browser apps (Web Speech default, Vosk-WASM fallback)","directories":{},"sideEffects":false,"_nodeVersion":"23.6.1","dependencies":{"assemblyai":"^4.18.5","@google/generative-ai":"^0.21.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.1","eslint":"^8.57.1","rimraf":"^6.0.1","vitest":"^2.1.3","typescript":"^5.6.3","@types/node":"^20.11.30","@typescript-eslint/parser":"^7.0.0","@typescript-eslint/eslint-plugin":"^7.0.0"},"_npmOperationalInternal":{"tmp":"tmp/voice-detector_0.1.1_1762008051199_0.5608250773456187","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@auditionhub/voice-detector","version":"0.1.2","keywords":["speech","vad","stt","web-speech","vosk","browser","react"],"author":{"name":"Craig Baker","email":"craig.d.baker@gmail.com"},"license":"MIT","_id":"@auditionhub/voice-detector@0.1.2","maintainers":[{"name":"craigdbaker","email":"craig.d.baker@gmail.com"}],"dist":{"shasum":"cf8d4a4b864c9ccd5c1316c8db10e3b11ad31b84","tarball":"https://registry.npmjs.org/@auditionhub/voice-detector/-/voice-detector-0.1.2.tgz","fileCount":8,"integrity":"sha512-5pVoBLl4XEm6xn3tsyrdtN3KNQNxfFOyWozKpDWrcnyy/UwVR90EO8rX2eZ13WzKu4F6Hwp6Q8wFqYQebPS7PA==","signatures":[{"sig":"MEUCIQCvpo4hJsrNz2KAtrXTz6Lm8CYSumxFHluivb7TOS+bagIgAO1BzuaaIPjl71AUv0dWJioXnnzqILtD8i2DQ5urXMU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":239619},"main":"dist/index.cjs","type":"module","types":"dist/index.d.ts","module":"dist/index.js","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"gitHead":"add26fe96d375e52401ba82e63cb65d4f443553a","scripts":{"dev":"tsup --watch","lint":"eslint \"src/**/*.{ts,tsx}\"","test":"vitest","build":"tsup","clean":"rimraf dist","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"craigdbaker","email":"craig.d.baker@gmail.com"},"_npmVersion":"11.6.0","description":"Real-time line completion detection for browser apps (Web Speech default, Vosk-WASM fallback)","directories":{},"sideEffects":false,"_nodeVersion":"23.6.1","dependencies":{"assemblyai":"^4.18.5","@google/generative-ai":"^0.21.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.1","eslint":"^8.57.1","rimraf":"^6.0.1","vitest":"^2.1.3","typescript":"^5.6.3","@types/node":"^20.11.30","@typescript-eslint/parser":"^7.0.0","@typescript-eslint/eslint-plugin":"^7.0.0"},"_npmOperationalInternal":{"tmp":"tmp/voice-detector_0.1.2_1762008091675_0.535576745873265","host":"s3://npm-registry-packages-npm-production"}},"0.1.3":{"name":"@auditionhub/voice-detector","version":"0.1.3","keywords":["speech","vad","stt","web-speech","vosk","browser","react"],"author":{"name":"Craig Baker","email":"craig.d.baker@gmail.com"},"license":"MIT","_id":"@auditionhub/voice-detector@0.1.3","maintainers":[{"name":"craigdbaker","email":"craig.d.baker@gmail.com"}],"dist":{"shasum":"9c7fde1f552509b4ddaa181b64bde3e4d8971fd5","tarball":"https://registry.npmjs.org/@auditionhub/voice-detector/-/voice-detector-0.1.3.tgz","fileCount":8,"integrity":"sha512-Rclmeluth8uUtVEVyHG2riX2p45PwuhqAkrQvTbgxs02s52/PjwDw7NpSO9VTNm+nzdhr7LZA3mioJ79U1OJ0A==","signatures":[{"sig":"MEYCIQCUTvuFW1uj60VsxcsHHBo6fKpbaggLD86acLJdMrUChgIhAM5srRxKgoExTrpelK2YdK90mCfr97mgf/6at1EBoixq","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":244325},"main":"dist/index.cjs","type":"module","types":"dist/index.d.ts","module":"dist/index.js","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"gitHead":"add26fe96d375e52401ba82e63cb65d4f443553a","scripts":{"dev":"tsup --watch","lint":"eslint \"src/**/*.{ts,tsx}\"","test":"vitest","build":"tsup","clean":"rimraf dist","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"craigdbaker","email":"craig.d.baker@gmail.com"},"_npmVersion":"11.6.0","description":"Real-time line completion detection for browser apps (Web Speech default, Vosk-WASM fallback)","directories":{},"sideEffects":false,"_nodeVersion":"23.6.1","dependencies":{"assemblyai":"^4.18.5","@google/generative-ai":"^0.21.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.1","eslint":"^8.57.1","rimraf":"^6.0.1","vitest":"^2.1.3","typescript":"^5.6.3","@types/node":"^20.11.30","@typescript-eslint/parser":"^7.0.0","@typescript-eslint/eslint-plugin":"^7.0.0"},"_npmOperationalInternal":{"tmp":"tmp/voice-detector_0.1.3_1762015821013_0.09186860301193933","host":"s3://npm-registry-packages-npm-production"}},"0.1.4":{"name":"@auditionhub/voice-detector","version":"0.1.4","description":"Real-time line completion detection for browser apps (Web Speech default, Vosk-WASM fallback)","license":"MIT","author":{"name":"Craig Baker","email":"craig.d.baker@gmail.com"},"type":"module","main":"dist/index.cjs","module":"dist/index.js","types":"dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"sideEffects":false,"scripts":{"build":"tsup","clean":"rimraf dist","dev":"tsup --watch","lint":"eslint \"src/**/*.{ts,tsx}\"","test":"vitest","typecheck":"tsc -p tsconfig.json --noEmit"},"engines":{"node":">=18"},"keywords":["speech","vad","stt","web-speech","vosk","browser","react"],"dependencies":{"ai":"^4.0.0","@ai-sdk/openai":"^1.0.0","@google/generative-ai":"^0.24.1","assemblyai":"^4.18.5"},"devDependencies":{"@types/node":"^20.11.30","@typescript-eslint/eslint-plugin":"^7.0.0","@typescript-eslint/parser":"^7.0.0","eslint":"^8.57.1","rimraf":"^6.0.1","tsup":"^8.0.1","typescript":"^5.6.3","vitest":"^2.1.3"},"_id":"@auditionhub/voice-detector@0.1.4","gitHead":"ea6819d4a725a4bec7e9f13b104b3362745d187e","_nodeVersion":"23.6.1","_npmVersion":"11.6.0","dist":{"integrity":"sha512-HrL1sVmkClMLwHS4bwJugemiJCXzhkZwN+nGkhugYC/qwiMRCwASgI52obkLTPFkLRPT5a0iha/derQ4VS14Hw==","shasum":"a4f8e02aa149ae78bfea6b65b04bdb1ebbbfdec2","tarball":"https://registry.npmjs.org/@auditionhub/voice-detector/-/voice-detector-0.1.4.tgz","fileCount":8,"unpackedSize":323026,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQCs59nDjTl9YznKx0GmQ3KM+0IpYoyKqFFqHLcILeoxuQIgHLAf9giX0jcMT8pOFZGObfe09ef8wRNvJ7glD4PccUQ="}]},"_npmUser":{"name":"craigdbaker","email":"craig.d.baker@gmail.com"},"directories":{},"maintainers":[{"name":"craigdbaker","email":"craig.d.baker@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/voice-detector_0.1.4_1762309411970_0.2934104318697939"},"_hasShrinkwrap":false}},"time":{"created":"2025-11-01T14:31:25.400Z","modified":"2025-11-05T02:23:32.440Z","0.1.0":"2025-11-01T14:31:25.738Z","0.1.1":"2025-11-01T14:40:51.427Z","0.1.2":"2025-11-01T14:41:31.922Z","0.1.3":"2025-11-01T16:50:21.189Z","0.1.4":"2025-11-05T02:23:32.190Z"},"author":{"name":"Craig Baker","email":"craig.d.baker@gmail.com"},"license":"MIT","keywords":["speech","vad","stt","web-speech","vosk","browser","react"],"description":"Real-time line completion detection for browser apps (Web Speech default, Vosk-WASM fallback)","maintainers":[{"name":"craigdbaker","email":"craig.d.baker@gmail.com"}],"readme":"# @auditionhub/voice-detector\n\nReal-time line completion detection for spoken lines in the browser. Detect when\na user has finished a line using pause duration, fuzzy suffix matching, or\nAI-powered semantic verification. Optimized for low latency with support for\nAssemblyAI streaming transcription and Gemini semantic analysis.\n\n## Install\n\n```bash\nnpm install @auditionhub/voice-detector\n```\n\n## 🔒 Security Best Practices\n\n**Important**: API keys should never be exposed to the browser! For production\napplications:\n\n- **AssemblyAI**: Generate temporary tokens server-side (see example below)\n- **Gemini**: Use `geminiProxyUrl` to proxy API calls through your server\n- See [Remix Example](./examples/remix/) for a complete secure implementation\n\n```ts\n// ✅ GOOD: Server generates tokens, client uses proxy\nconst detector = new LineDetector({\n\tassemblyAiToken: await fetchTokenFromServer(),\n\tgeminiProxyUrl: '/api/gemini-proxy',\n})\n\n// ❌ BAD: Direct API keys in browser\nconst detector = new LineDetector({\n\tassemblyAiApiKey: 'secret-key', // ⚠️ Exposed to browser!\n\tgeminiApiKey: 'secret-key', // ⚠️ Exposed to browser!\n})\n```\n\n## Basic usage\n\n### Simple Web Speech (Default)\n\n```ts\nimport { LineDetector } from '@auditionhub/voice-detector'\n\nconst detector = new LineDetector({\n\tpauseMs: 800,\n\tsuffixWords: 3,\n\tuseWebSpeech: true,\n})\n\nawait detector.init()\n\ndetector.on('lineComplete', (ev) => {\n\tconsole.log('Line complete:', ev.reason, ev.elapsedMs)\n})\n\nawait detector.startLine({ text: 'I think we should head back now.' })\n```\n\n### With AssemblyAI + Gemini (Recommended)\n\n```ts\nimport { LineDetector } from '@auditionhub/voice-detector'\n\nconst detector = new LineDetector({\n\tassemblyAiToken: 'your-temporary-token', // Generate server-side for browser\n\tgeminiApiKey: 'your-gemini-key',\n\tsemanticThreshold: 0.75,\n\tenableSemanticMatching: true,\n\tpauseMs: 800,\n})\n\nawait detector.init()\n\ndetector.on('semanticMatch', (ev) => {\n\tconsole.log('Match probability:', ev.probability)\n\tconsole.log('Reason:', ev.reason)\n})\n\ndetector.on('lineComplete', (ev) => {\n\tconsole.log('Line complete:', ev.reason, ev.elapsedMs)\n\tconsole.log('Semantic score:', ev.semanticProbability)\n})\n\nawait detector.startLine({ text: 'To be or not to be, that is the question.' })\n```\n\n## Configuration\n\n### Core Options\n\n- **pauseMs**: number (default 800) – silence duration to trigger completion\n- **suffixWords**: number (default 3) – trailing words to compare for suffix\n  match\n- **suffixMaxDist**: number (default 2) – max edit distance for suffix match\n- **baseWPS**: number (default 2.7) – initial words/sec for timing estimate\n- **maxMultiplier**: number (default 2.0) – timeout multiplier cap\n- **useWebSpeech**: boolean (default true) – prefer Web Speech API locally\n\n### AssemblyAI Options\n\n- **assemblyAiToken**: string (optional) – Temporary token for AssemblyAI\n  real-time streaming (required for browser)\n- **assemblyAiApiKey**: string (optional) – API key for AssemblyAI (Node.js\n  only, not for browser)\n- **Important**: For browser usage, you MUST use `assemblyAiToken` instead of\n  `assemblyAiApiKey`. Generate tokens server-side:\n  ```js\n  // Server-side (Node.js)\n  import { AssemblyAI } from 'assemblyai'\n  const client = new AssemblyAI({ apiKey: 'your-api-key' })\n  const token = await client.streaming.createTemporaryToken({\n  \texpires_in_seconds: 480,\n  })\n  // Return token to client\n  ```\n- Get your API key: [https://www.assemblyai.com](https://www.assemblyai.com)\n\n### Gemini Options\n\n- **geminiApiKey**: string (optional) – API key for Gemini 2.5 Flash Lite\n  semantic matching\n- **geminiProxyUrl**: string (optional) – **Recommended** – Server-side proxy\n  URL for Gemini API calls (keeps API key secure)\n- **semanticThreshold**: number (default 0.75) – probability threshold for\n  semantic completion (0-1)\n- **enableSemanticMatching**: boolean (default true) – enable semantic\n  verification\n\n**Security Best Practice**: Use `geminiProxyUrl` instead of `geminiApiKey` in\nbrowser applications to keep your API key secure on the server. See\n[Remix Example](./examples/remix/) for a complete implementation.\n\nGet your key: [https://aistudio.google.com](https://aistudio.google.com)\n\n### Legacy Options\n\n- **remoteURL**: string | null – optional remote STT (future)\n\n## Events\n\n- `lineStart`: `{ text }` – Fired when a new line starts\n- `lineComplete`:\n  `{ reason: 'pause' | 'suffix' | 'semantic', elapsedMs, transcript?, avgWPS?, semanticProbability?, semanticReason? }`\n  – Fired when line is successfully detected\n- `lineTimeout`:\n  `{ reason: 'timeout', elapsedMs, transcript?, avgWPS?, semanticProbability?, semanticReason? }`\n  – Fired if no completion detected in time\n- `transcript`: `{ text, isFinal }` – Fired on each transcript update\n- `semanticMatch`: `{ probability, reason }` – Fired when semantic match is\n  calculated (Gemini only)\n\n## Speech-to-Text Adapters\n\nThe library supports multiple STT backends with automatic fallback:\n\n1. **AssemblyAI** (recommended): Real-time streaming transcription with\n   WebSocket\n   - Best accuracy and latency\n   - Requires temporary token (for browser) or API key (for Node.js)\n   - Generate tokens server-side for browser security\n   - Automatic fallback if unavailable\n\n2. **Web Speech API**: Browser-native STT (default fallback)\n   - No API key required\n   - Good for simple use cases\n   - Browser support varies\n\n3. **Vosk-WASM** (offline, WIP): Local on-device transcription\n   - No internet required\n   - Scaffolding included; full integration pending\n\n## Semantic Verification\n\nWhen `geminiApiKey` is provided, the library uses **Gemini 2.5 Flash Lite** to\nverify semantic meaning:\n\n- **Probability-based matching**: Returns 0-1 score for how well transcript\n  matches target\n- **Smart threshold**: Configurable matching strictness (default 75%)\n- **Paraphrase-friendly**: Understands synonyms and alternate phrasings\n- **Low cost**: $0.10/1M input tokens\n- **Fast**: ~180ms time-to-first-token\n\n## Examples\n\n- **React**: `examples/react-basic` – Complete demo with AssemblyAI + Gemini\n  integration\n- Remix (client-only): `examples/remix`\n\n## How It Works\n\n1. **VAD Detection**: Voice Activity Detection monitors audio for speech/silence\n   boundaries\n2. **Transcription**: AssemblyAI streams real-time transcription (or Web Speech\n   as fallback)\n3. **Semantic Verification**: Gemini compares meaning between spoken and target\n   text\n4. **Completion**: Line completes when:\n   - Pause detected (configurable duration), OR\n   - Suffix match found, OR\n   - Semantic probability crosses threshold (Gemini only)\n\n## Roadmap\n\n- [x] AssemblyAI real-time streaming\n- [x] Gemini semantic verification\n- [ ] Vosk-WASM offline model + caching\n- [ ] Web Worker/AudioWorklet offload\n- [ ] Multi-language support\n","readmeFilename":"README.md"}