{"_id":"@andrewhoyle/voice","_rev":"3-8d5b12216ad8109a368fe8500c6dfd82","name":"@andrewhoyle/voice","dist-tags":{"latest":"0.5.0"},"versions":{"0.4.0":{"name":"@andrewhoyle/voice","version":"0.4.0","_id":"@andrewhoyle/voice@0.4.0","maintainers":[{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"}],"dist":{"shasum":"1b6ad2a11297af3f587f8fe92073e5594c51449e","tarball":"https://registry.npmjs.org/@andrewhoyle/voice/-/voice-0.4.0.tgz","fileCount":32,"integrity":"sha512-eMDh04df/fUlq4Yyn/lN5OsR2cgFLeHy2dbfroUoFNY1bQLI3Gt3f983Fmi0RckN/qahIikR66QXbi8m48qrZw==","signatures":[{"sig":"MEUCIQCD3jsrmC9N7Vb3DAfcnj7WGdR4DsQ9siSyh/4mV+IhWAIgbdaCM1qlKdIItk19mQfGmjAFZiQBgOfk3gXQhLmgJrg=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":176163},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./react":{"types":"./dist/react.d.ts","import":"./dist/react.js"},"./mindtale":{"types":"./dist/mindtale.d.ts","import":"./dist/mindtale.js"},"./universal":{"types":"./dist/universal.d.ts","import":"./dist/universal.js"}},"gitHead":"7745707ae9169ecb5b00b402f44f19ed12690157","scripts":{"test":"vitest run","build":"tsc -p tsconfig.json","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"},"_npmVersion":"10.9.2","description":"TypeScript SDK for the Imperium Voice gateway (client + MindTale compat shim + React voice picker).","directories":{},"_nodeVersion":"22.15.0","publishConfig":{"access":"public","registry":"https://registry.npmjs.org"},"_hasShrinkwrap":false,"devDependencies":{"react":"^18.3.0","vitest":"^2.0.0","typescript":"^5.5.0","@types/react":"^18.3.0"},"peerDependencies":{"react":">=18"},"peerDependenciesMeta":{"react":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.4.0_1783201129295_0.9339445083609761","host":"s3://npm-registry-packages-npm-production"}},"0.4.1":{"name":"@andrewhoyle/voice","version":"0.4.1","_id":"@andrewhoyle/voice@0.4.1","maintainers":[{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"}],"dist":{"shasum":"7fbb06047fa69300612b24432d386681ac1750e7","tarball":"https://registry.npmjs.org/@andrewhoyle/voice/-/voice-0.4.1.tgz","fileCount":39,"integrity":"sha512-b/MRcTeKgSeiBmefj/wVJtOcqbS3WZCAJCZowacn1QKZD1TcrM/siVTG2JZNVUrI04t0ZPm148mCyCNTXW22/g==","signatures":[{"sig":"MEQCIFSNeYPvTatA4wehoaXO+FhT0eLHCV4ryX1aPSYXZYinAiAQByYVez0ZB6AOsk/61uFDFGiycVSPJDavUqAB+Y9VWQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":223081},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./react":{"types":"./dist/react.d.ts","import":"./dist/react.js"},"./mindtale":{"types":"./dist/mindtale.d.ts","import":"./dist/mindtale.js"},"./universal":{"types":"./dist/universal.d.ts","import":"./dist/universal.js"}},"gitHead":"ed5e74d3bc62703821d3faf951390ccc0a0b0c60","scripts":{"test":"vitest run","build":"tsc -p tsconfig.json","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"},"_npmVersion":"10.9.2","description":"TypeScript SDK for the Imperium Voice gateway (client + MindTale compat shim + React voice picker).","directories":{},"_nodeVersion":"22.15.0","publishConfig":{"access":"public","registry":"https://registry.npmjs.org"},"_hasShrinkwrap":false,"devDependencies":{"react":"^18.3.0","vitest":"^2.0.0","typescript":"^5.5.0","@types/react":"^18.3.0"},"peerDependencies":{"react":">=18"},"peerDependenciesMeta":{"react":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/voice_0.4.1_1783858970166_0.27385384025409265","host":"s3://npm-registry-packages-npm-production"}},"0.5.0":{"name":"@andrewhoyle/voice","version":"0.5.0","description":"TypeScript SDK for the Imperium Voice gateway (client + MindTale compat shim + React voice picker).","type":"module","main":"./dist/index.js","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./react":{"types":"./dist/react.d.ts","import":"./dist/react.js"},"./mindtale":{"types":"./dist/mindtale.d.ts","import":"./dist/mindtale.js"},"./universal":{"types":"./dist/universal.d.ts","import":"./dist/universal.js"}},"publishConfig":{"registry":"https://registry.npmjs.org","access":"public"},"scripts":{"build":"tsc -p tsconfig.json","typecheck":"tsc -p tsconfig.json --noEmit","test":"vitest run"},"engines":{"node":">=18"},"peerDependencies":{"react":">=18"},"peerDependenciesMeta":{"react":{"optional":true}},"devDependencies":{"@types/react":"^19.2.17","react":"^19.2.7","typescript":"^7.0.2","vitest":"^2.0.0"},"_id":"@andrewhoyle/voice@0.5.0","gitHead":"73b01d1d6da1ebb265a10495b1325b5e01713e75","_nodeVersion":"22.23.1","_npmVersion":"10.9.8","dist":{"integrity":"sha512-tmC/0ejtkuPH62WzECNWhdPGBKw+Yc8z94uQE+aK25LyJHC6Sp7zscKMTgameg2VLZm725VVPrxbRiS3EUIvMA==","shasum":"fa2574a5bf68d140366d2e36dbbbf3771a5079dc","tarball":"https://registry.npmjs.org/@andrewhoyle/voice/-/voice-0.5.0.tgz","fileCount":39,"unpackedSize":234656,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQCxHfQ0Sr6HD5bL9NhyykzJKpjOX0BVdHTA+YOHI2Z3ZQIhAKQJy6R1MLPCm8ACCqB208S4FSFCB+jjI3TZnmd27H57"}]},"_npmUser":{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"},"directories":{},"maintainers":[{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/voice_0.5.0_1784363825614_0.4040827990968652"},"_hasShrinkwrap":false}},"time":{"created":"2026-07-04T21:38:49.065Z","modified":"2026-07-18T08:37:05.964Z","0.4.0":"2026-07-04T21:38:49.432Z","0.4.1":"2026-07-12T12:22:50.317Z","0.5.0":"2026-07-18T08:37:05.799Z"},"description":"TypeScript SDK for the Imperium Voice gateway (client + MindTale compat shim + React voice picker).","maintainers":[{"name":"andrewhoyle","email":"andrew.hoyle@imperium-bi.co.uk"}],"readme":"# @andrewhoyle/voice (TypeScript SDK)\n\nTyped `fetch`-based client for the [Imperium Voice](../../README.md) gateway,\nplus a React voice picker and a MindTale compatibility shim. Runs anywhere with a\nglobal `fetch` — Node ≥18, edge runtimes, Deno, browsers.\n\nThe full server-side REST contract for consuming apps (endpoints, auth, error\ncodes, learning-loop feedback) ships with this package:\n[GATEWAY-INTEGRATION.md](./GATEWAY-INTEGRATION.md).\n\n```bash\nnpm install @andrewhoyle/voice\n```\n\n## Quick start\n\n```ts\nimport { VoiceClient, QuotaError } from \"@andrewhoyle/voice\";\n\nconst voice = new VoiceClient({ baseUrl: \"https://voice.imperium.example\", apiKey });\n\n// One clause, content-addressed + cached (live voice-switching):\nconst clip = await voice.segment(\n  \"Once upon a time, in a kingdom by the sea.\",\n  { provider: \"elevenlabs\", elevenlabs_voice_id: \"pNInz6obpgDQGcFmaJgB\" },\n  { contentType: \"story\" },\n);\nconsole.log(clip.audioUrl, clip.durationMs, clip.cached);\n\n// A whole lesson:\nconst lesson = await voice.synthesize(\"black-holes-101\", [\n  { title: \"Intro\", text: \"Welcome.\" },\n  { text: \"Black holes bend light.\" },\n]);\n```\n\nEvery non-2xx response throws a typed error (all extend `VoiceError`):\n`AuthError` (401), `PolicyError` (403), `BadRequestError` (400/422),\n`QuotaError` (429), `StorageUnavailableError` (503), `ServerError` (5xx), and\n`TransportError` (no response). `ApiError` carries `.status` and `.detail`.\n\n```ts\ntry {\n  await voice.segment(text, { provider: \"elevenlabs\" });\n} catch (e) {\n  if (e instanceof QuotaError) { /* back off / upgrade */ }\n}\n```\n\n## React voice picker\n\n```tsx\nimport { useState } from \"react\";\nimport { VoicePicker, DEFAULT_VOICES, type VoiceOption } from \"@andrewhoyle/voice/react\";\n\nfunction Narrator() {\n  const [voice, setVoice] = useState<VoiceOption>(DEFAULT_VOICES[0]!);\n  return (\n    <>\n      <VoicePicker value={voice.id} onChange={setVoice} />\n      <button onClick={() => voice && client.segment(text, voice.voiceConfig)}>Play</button>\n    </>\n  );\n}\n```\n\n`VoicePicker` is data-driven: pass your own `voices` prop (from your catalogue) or\nuse the bundled `DEFAULT_VOICES`. `onChange` hands back the full option — read\n`option.voiceConfig` and pass it straight to a client call. React is an optional\npeer dependency, isolated to this subpath so non-React consumers don't pull it in.\n\n## Voice clones\n\nCreate a clone (needs the `clones:write` scope and a consent reference), then\nvoice it by `clone_id` on a Chatterbox config — the gateway resolves the stored\nsample server-side, so the raw URL never travels on the synthesis request:\n\n```ts\nconst clone = await voice.createClone({ name: \"Narrator\", sampleUrl, consentRef: \"consent-2026-01\" });\nawait voice.segment(\"Read this in my voice.\", {\n  provider: \"chatterbox\",\n  voice_name: \"chatterbox_default\",\n  clone_id: clone.id,\n});\n```\n\n## MindTale migration\n\nThe gateway's response shape is identical to the MindTale Lesson row, so the\ncompat shim lets existing call sites switch transport with minimal change:\n\n```ts\nimport { VoiceClient } from \"@andrewhoyle/voice\";\nimport { createMindTaleEngine } from \"@andrewhoyle/voice/mindtale\";\n\nconst engine = createMindTaleEngine(new VoiceClient({ baseUrl, apiKey }));\nconst audio = await engine.generateLessonAudio({ slug, sections, voiceConfig });\n// audio.audioUrl / audio.durationSeconds / audio.sectionTimestamps — unchanged\n```\n\nThe shim defaults `textFormat` to `\"raw\"` (MindTale fed unprocessed prose and let\nthe worker normalise), unlike the base client's `\"normalized\"` default.\n\n## Speech-to-text\n\n```ts\n// Sync transcription — upload a Blob/bytes, or pass { url }:\nconst t = await voice.transcribe(fileBlob, { diarize: true, wordTimestamps: true });\nconsole.log(t.text, t.language, t.speakerCount);\n\n// Long audio → async job (+ optional signed webhook):\nconst job = await voice.createTranscriptionJob(\"https://files/meeting.mp3\", {\n  diarize: true, webhookUrl: \"https://me/cb\", idempotencyKey: \"meeting-42\",\n});\nconst done = await voice.getTranscriptionJob(job.id);\n\n// Real-time (browser) — the key rides the apikey.<KEY> subprotocol, never the URL:\nconst session = voice.transcribeStream(\n  { onTranscript: (m) => console.log(m.type, m.transcript) },\n  { config: { sample_rate: 16000, diarize: true } },\n);\nsession.sendAudio(pcmChunk);\nsession.finish();\n```\n\n## Streaming TTS + discovery\n\n```ts\nfor await (const chunk of await voice.synthesizeStream(\"Once upon a time.\")) {\n  player.feed(chunk);              // low time-to-first-audio\n}\nconst voices = await voice.listVoices();   // VoiceInfo[] (each with a ready voice_config)\nconst models = await voice.listModels();   // ModelInfo[] (features + cost_per_minute)\n```\n\nThe React picker can load entitled voices live: `const { voices } = useVoices(client)`.\n\n## Live transcription — `listen()` (device-first)\n\n`@andrewhoyle/voice/universal` adds live speech-to-text that is **device-first**: when the\nbrowser has Web Speech it uses the **browser-native** recogniser — free, low-latency,\ninterim results, and with *no* round-trip through our gateway. Only when Web Speech is\nunavailable does it fall back to the gateway (over a short-lived token — the long-lived key\nnever enters the page). It's a drop-in replacement for hand-rolled `webkitSpeechRecognition`,\nwith the gateway fallback as a strict upgrade over a typing-only fallback.\n\n> **Residency note — Web Speech is *not* guaranteed on-device.** The browser controls where\n> recognition happens: some browsers process locally, but **Chrome streams audio to Google**\n> and Safari to Apple. So `listen()`'s device tier is \"no transfer through *our* infra and no\n> change versus the `webkitSpeechRecognition` an app already hand-rolls\" — **not** an offline\n> guarantee. If you need a hard on-device/UK-residency guarantee, that comes from the gateway\n> tier pointed at a UK/EU STT provider, not from Web Speech. Confirm behaviour per target\n> browser before making an \"offline\" claim to users.\n\n```ts\nimport { Voice } from \"@andrewhoyle/voice/universal\";\n\nconst voice = new Voice({ client /* optional: enables the gateway fallback */ });\nconst session = await voice.listen({\n  lang: \"en-GB\",\n  onPartial: (t) => setDraft(t),     // interim\n  onFinal: (t) => setValue(t),       // committed\n});\nconsole.log(session.tier);           // \"device\" on Chrome/Edge; \"gateway\" otherwise\n// …\nsession.stop();                      // graceful (deliver final, then end)\n```\n\nReact:\n\n```tsx\nimport { useListen } from \"@andrewhoyle/voice/react\";\n\nconst { listening, partial, final, start, stop } = useListen(voice, { lang: \"en-GB\" });\n<button onMouseDown={start} onMouseUp={stop}>{listening ? \"…\" : \"🎤\"}</button>\n<input value={final || partial} />\n```\n\nForce a tier with `preferTier: \"device\" | \"gateway\"` (default is device-first). The gateway\nfallback needs the gateway to allow your origin (`CORS_ALLOWED_ORIGINS`) and the tenant to\nhold the `transcribe:stream` scope.\n\nThe gateway tier captures mic audio as **anti-aliased 16 kHz PCM16** (a Kaiser windowed-sinc\nresampler runs in an AudioWorklet — design record: ADR-0019 in the platform repo) in fixed 20 ms frames,\nand `stop()` drains the buffered tail before finishing so the last word isn't clipped.\nBundler note: the worklet embeds SDK classes via `Function.toString()`, so do **not** enable\nproperty mangling (e.g. terser `mangle.properties`) on `@andrewhoyle/voice` — plain\nminification/renaming is fine.\n\n## Develop\n\n```bash\npnpm install\npnpm typecheck   # tsc --noEmit\npnpm test        # vitest\npnpm build       # emit dist/\n```\n\n## Streaming sessions: reconnection & keepalive\n\nLive WS sessions (`transcribeStream`, `converse`) are stateful — audio position\nand transcript context live server-side — so **reconnection is your\nresponsibility**: on `onClose`/`onError`, open a fresh session and resume from\nyour own application state (they cannot be resumed transparently). `sendAudio`\nis safe to call immediately: frames sent while the socket is still connecting\nqueue and flush in order once it opens. For long idle gaps (e.g. push-to-talk\npauses), keep the session warm by continuing to stream silence frames, or send\na small audio frame periodically as a keepalive — idle sockets are subject to\nintermediary timeouts.\n","readmeFilename":"README.md"}