{"_id":"@bandwidth-labs/bw-stt","_rev":"2-663c99aa309e379db89aed19bf1aa00f","name":"@bandwidth-labs/bw-stt","dist-tags":{"latest":"0.2.0"},"versions":{"0.1.0":{"name":"@bandwidth-labs/bw-stt","version":"0.1.0","keywords":["bandwidth","speech-to-text","stt","transcription","streaming","websocket"],"license":"MIT","_id":"@bandwidth-labs/bw-stt@0.1.0","maintainers":[{"name":"dlinsky","email":"dlinsky@bandwidth.com"}],"homepage":"https://labs.bandwidth.com/docs/speech-to-text","bugs":{"url":"https://github.com/Bandwidth/bw_labs_sdks/issues"},"dist":{"shasum":"73c86bedda75c4cb98313ded3d08aa4bb0b2cb9e","tarball":"https://registry.npmjs.org/@bandwidth-labs/bw-stt/-/bw-stt-0.1.0.tgz","fileCount":10,"integrity":"sha512-5QYL5n2t2+oZgjR2CIUZDLEe5sR0oFq1MiYC4XVTKoC19tFQO8BGlfIICeBYsE2QkCryzRtpmKCUqSac/6PSFw==","signatures":[{"sig":"MEUCIGJ7Tw5YMjXEaQOwUu0YP37uH15pE4bCZNh4mVV/fH2JAiEA3CkwofRW+m8jQOqWOeOaEhrSvskDnpiW2gl8zVcXB1A=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bandwidth-labs%2fbw-stt@0.1.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":396063},"main":"./dist/index.cjs","type":"module","_from":"file:bandwidth-labs-bw-stt-0.1.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22"},"exports":{".":{"import":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"require":{"types":"./dist/index.d.cts","default":"./dist/index.cjs"}},"./package.json":"./package.json"},"scripts":{"lint":"eslint .","test":"vitest run","build":"tsup","prepack":"npm run build","prepare":"npm run build","typecheck":"tsc --noEmit"},"_npmUser":{"name":"dlinsky","email":"dlinsky@bandwidth.com"},"_resolved":"/home/runner/work/bw_labs_sdks/bw_labs_sdks/typescript/bandwidth-labs-bw-stt-0.1.0.tgz","_integrity":"sha512-5QYL5n2t2+oZgjR2CIUZDLEe5sR0oFq1MiYC4XVTKoC19tFQO8BGlfIICeBYsE2QkCryzRtpmKCUqSac/6PSFw==","repository":{"url":"git+https://github.com/Bandwidth/bw_labs_sdks.git","type":"git","directory":"typescript"},"_npmVersion":"11.17.0","description":"TypeScript SDK for the Bandwidth Labs streaming speech-to-text API","directories":{},"sideEffects":false,"_nodeVersion":"24.19.0","dependencies":{"ws":"^8.18.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.2","tsup":"^8.3.5","eslint":"^9.17.0","vitest":"^2.1.8","@types/ws":"^8.5.13","typescript":"^5.7.2","@types/node":"^24","typescript-eslint":"^8.19.0"},"_npmOperationalInternal":{"tmp":"tmp/bw-stt_0.1.0_1788196588854_0.6517134008798569","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@bandwidth-labs/bw-stt","version":"0.2.0","description":"TypeScript SDK for the Bandwidth Labs streaming speech-to-text API","license":"MIT","repository":{"type":"git","url":"git+https://github.com/Bandwidth/bw_labs_sdks.git","directory":"typescript"},"homepage":"https://labs.bandwidth.com/docs/speech-to-text","bugs":{"url":"https://github.com/Bandwidth/bw_labs_sdks/issues"},"publishConfig":{"access":"public"},"type":"module","engines":{"node":">=22"},"main":"./dist/index.cjs","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"import":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"require":{"types":"./dist/index.d.cts","default":"./dist/index.cjs"}},"./package.json":"./package.json"},"sideEffects":false,"keywords":["bandwidth","speech-to-text","stt","transcription","streaming","websocket"],"scripts":{"build":"tsup","prepack":"npm run build","typecheck":"tsc --noEmit","lint":"eslint .","test":"vitest run","prepare":"npm run build"},"dependencies":{"ws":"^8.18.0"},"devDependencies":{"@types/node":"^24","@types/ws":"^8.5.13","esbuild":"^0.28.2","eslint":"^9.17.0","tsup":"^8.3.5","tsx":"^4.19.2","typescript":"^5.7.2","typescript-eslint":"^8.19.0","vitest":"^5.0.0"},"overrides":{"esbuild":"$esbuild"},"_id":"@bandwidth-labs/bw-stt@0.2.0","_integrity":"sha512-3PTN8ALLVWOFYEKhmt+qhh+6qZ+WAj9ptpfdtKPVtaQScqreBKQYqC2MORgcSueIdouShZRM51/LGmkVIS/L9g==","_resolved":"/home/runner/work/bw_labs_sdks/bw_labs_sdks/package/bandwidth-labs-bw-stt-0.2.0.tgz","_from":"file:package/bandwidth-labs-bw-stt-0.2.0.tgz","_nodeVersion":"24.20.0","_npmVersion":"11.19.0","dist":{"integrity":"sha512-3PTN8ALLVWOFYEKhmt+qhh+6qZ+WAj9ptpfdtKPVtaQScqreBKQYqC2MORgcSueIdouShZRM51/LGmkVIS/L9g==","shasum":"778e2cec0188f5fefbcb014ab567c507c556b6b4","tarball":"https://registry.npmjs.org/@bandwidth-labs/bw-stt/-/bw-stt-0.2.0.tgz","fileCount":10,"unpackedSize":543807,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bandwidth-labs%2fbw-stt@0.2.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIBsWNQs4Aav/Upr9P9ewPxaDBynsyIrRZ5qMcfKrRBVXAiB5orsWJXgZRk4k3jJ2/QbAIkxiF600aRt5s9bfTeZ+uw=="}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:c38b9e3f-2cc7-42b3-ac28-2888ee029194"}},"directories":{},"maintainers":[{"name":"dlinsky","email":"dlinsky@bandwidth.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/bw-stt_0.2.0_1788978143939_0.3792271487843246"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-31T17:16:28.360Z","modified":"2026-09-09T18:22:24.424Z","0.1.0":"2026-08-31T17:16:29.004Z","0.2.0":"2026-09-09T18:22:24.104Z"},"bugs":{"url":"https://github.com/Bandwidth/bw_labs_sdks/issues"},"license":"MIT","homepage":"https://labs.bandwidth.com/docs/speech-to-text","keywords":["bandwidth","speech-to-text","stt","transcription","streaming","websocket"],"repository":{"type":"git","url":"git+https://github.com/Bandwidth/bw_labs_sdks.git","directory":"typescript"},"description":"TypeScript SDK for the Bandwidth Labs streaming speech-to-text API","maintainers":[{"name":"dlinsky","email":"dlinsky@bandwidth.com"}],"readme":"# @bandwidth-labs/bw-stt\n\nTypeScript SDK for the Bandwidth Labs streaming speech-to-text API. Requires Node >=22. Browser support is separate: use a modern browser with fetch, AbortController and WebSocket support. File-path APIs are Node-only.\n\nFull protocol details are in the [API reference](https://labs.bandwidth.com/docs/speech-to-text).\n\n## Install\n\nAfter 0.2.0 is published, install it from the package registry:\n\n```sh\nnpm install @bandwidth-labs/bw-stt@0.2.0\n```\n\nFor source setup, see [Contributing](https://github.com/Bandwidth/bw_labs_sdks/blob/main/CONTRIBUTING.md).\n\n## Quickstart\n\n```ts\nimport { BwSttClient } from \"@bandwidth-labs/bw-stt\";\n\nconst client = new BwSttClient(); // reads BW_STT_API_KEY from the environment\nconst session = await client.connect({ encoding: \"linear16\", sampleRate: 16000 });\n\nsession.on(\"segment\", (segment) => process.stdout.write(segment.text));\n\nfor await (const segment of session.streamFile(\"call.wav\")) void segment;\n\nconst closed = await session.closeStream();\nconsole.log(`\\naudio seconds: ${closed.audioDurationSeconds.toFixed(2)}`);\n```\n\nIn browsers, pass the key explicitly: `new BwSttClient({ apiKey })`. Browsers cannot set WebSocket headers, so the SDK automatically carries the key as an `api_key` query parameter there. In Node it uses the `X-BW-LABS-API-KEY` header. Override with `authCarrier: \"header\" | \"query\"` if needed.\n\n### Pointing at another endpoint\n\n`baseUrl` accepts `ws`, `wss`, `http`, or `https`; `http(s)` is converted to `ws(s)` for streaming and back to `http(s)` for transcribe. A baseUrl without a path gets the standard endpoint paths appended (`/audio/v1/listen`, `/audio/v1/transcribe`). A custom path is used verbatim for streaming; for transcribe, a trailing `/listen` is replaced with `/transcribe`, and any other custom path gets `/transcribe` appended.\n\n## Listen modes and Transcribe\n\nInstant and demand are modes of the `/audio/v1/listen` WebSocket endpoint.\nTranscribe uses `POST /audio/v1/transcribe` over HTTP and is not a\nWebSocket session mode.\n\n| Mode | Delivery | Best for |\n|---|---|---|\n| `instant` (default) | Final `Segment` events arrive as audio is decoded. | Live captions and continuously updating displays |\n| `demand` | Each `Finalize` gets one `Transcript` per channel, including empty transcripts. `CloseStream` delivers the remainder the same way, then `SessionClosed`. | Voice-agent turns and application-controlled boundaries |\n\n### Instant\n\nThe default. `Segment` events stream back the moment text is decoded. Every\nsegment is final and append-only; there are no interim results to reconcile.\n\n```ts\nconst session = await client.connect(); // mode: \"instant\" is the server default\nsession.on(\"segment\", (segment) => render(segment));\n```\n\n### Demand\n\nDemand buffers finalized results server-side. Each `Finalize` gets one\n`Transcript` per channel, even when that channel is empty. `CloseStream`\ndelivers the remainder as `Transcript` events, then sends `SessionClosed`.\nDemand never sends `Segment` events.\n\n```ts\nconst session = await client.connect({ mode: \"demand\" });\nfor (const turnFrames of callerTurns) {\n  for (const frame of turnFrames) session.sendAudio(frame);\n  const transcripts = await session.finalizeTranscript();\n  const turnText = transcripts.map((transcript) => transcript.text).join(\" \");\n  answerTurn(turnText);\n}\n\nconst closed = await session.closeStream(); // remainder Transcripts, then SessionClosed\n```\n\nUse `finalize()` instead when the control message should be fire-and-forget.\nFor an instant versus demand comparison, use `Segment` callbacks for\ncontinuously arriving final pieces, or `Transcript` callbacks and\n`finalizeTranscript()` for application-controlled voice-agent turns.\n\n### Transcribe\n\nWhole-recording transcription (up to five minutes) in one HTTP\ncall. No session to manage.\n\n```ts\nconst result = await client.transcribeFile(\"call.wav\");\nconsole.log(result.text, result.audioDurationSeconds);\n\n// or from bytes you already have:\nconst result2 = await client.transcribe(rawLinear16, { encoding: \"linear16\", sampleRate: 16000 });\n```\n\n`transcribeFile` uploads a WAV path as `audio/wav`, preserving its container\nand using the header's sample rate and channel count. `transcribe` and\n`transcribeFile(..., { raw: true })` upload headerless linear16 bytes as\n`application/octet-stream` with `encoding=linear16` and `sample_rate`; channels\nmay be 1 or 2, with 2 selecting downmix.\n\nA successful response has this shape:\n\n```json\n{\n  \"request_id\": \"6f58c1c6-7e0c-4bb8-9d72-3fb3d4c5c1aa\",\n  \"text\": \"i need a dry van\",\n  \"words\": [\n    {\"word\": \"i\", \"start\": 0.00, \"end\": 0.12},\n    {\"word\": \"need\", \"start\": 0.16, \"end\": 0.20}\n  ],\n  \"segments\": [\n    {\"start\": 0.00, \"end\": 0.72, \"text\": \"i need a dry van\"}\n  ],\n  \"audio_duration_seconds\": 0.72,\n  \"model_info\": {\"name\": \"bw-listen-en\", \"version\": \"current\"}\n}\n```\n\n`words` is a timestamped word list and may be empty. `segments` is a typed\nlist with `start`, `end`, and `text` fields.\n\nFor raw audio, pass `channels: 2` explicitly with `multichannel: true` because\nthe client cannot infer the channel count. WAV uploads and URL submissions let\nthe server infer the channel count. A stereo recording with `multichannel:\ntrue` returns an independent typed result for each channel. The single-channel\nresult keeps the same shape when `multichannel` is omitted:\n\n```ts\nconst stereo = await client.transcribe(rawLinear16, {\n  channels: 2,\n  multichannel: true,\n});\nfor (const channel of stereo.channels ?? []) console.log(channel.channel, channel.text);\n```\n\n### Asynchronous transcription jobs\n\nUse the `transcriptions` namespace for recordings that should be processed as\njobs. Byte uploads use raw linear16 by default. Set `raw: false` when the bytes\nare a complete WAV container, which is sent as `audio/wav` without raw-only\nformat parameters. URL submissions send the location in JSON:\n\n```ts\nconst job = await client.transcriptions.submit({ audio: rawLinear16 });\nconst result = await client.transcriptions.wait(job.id);\nconsole.log(result.text);\n\nconst urlJob = await client.transcriptions.submit({\n  audioUrl: \"https://media.example.com/call.wav\",\n  callbackUrl: \"https://hooks.example.com/stt\",\n  callbackAuthHeaderName: \"X-Callback-Key\",\n  callbackAuthHeaderValue: \"callback-secret\",\n});\nconsole.log(urlJob.status);\n```\n\nFor raw stereo jobs, pass `channels: 2, multichannel: true` explicitly. WAV and\nURL jobs let the server infer the channel count. Pass `channels: 2,\nmultichannel: true` when the client should include the stereo declaration.\nFor upload callbacks, `callbackUrl` is a query parameter and\n`callbackAuthHeaderName` and `callbackAuthHeaderValue` travel in the\n`X-Callback-Auth-Name` and `X-Callback-Auth-Value` request headers. URL\nsubmissions put all three callback options in the JSON `callback` object. The\nSDK never sends callback credentials in the query. `wait(id)` defaults to a\n600 second overall timeout and polls every 2 seconds. Pass `timeoutMs` and\n`pollIntervalMs` to change them, and use `signal` to cancel a wait while it is\npolling. A wait timeout throws `TranscriptionTimeoutError`. The namespace also\nprovides `get(id)` and `delete(id)`. Uploads are fully buffered in memory, up\nto 512 MiB, including audio downloaded from a URL.\n\n### Job lifecycle\n\nThe per-key job limit returns `Retry-After: 30`; a busy submission returns\n`Retry-After: 5` (seconds). The SDK surfaces these errors without retrying.\nThe service follows at most three redirects when fetching `audio_url`, with a\n60 s whole-download limit. Authenticated SDK API calls reject redirects.\nUploads are limited to 512 MiB and job audio to 1800 s. These job limits are\nseparate from the synchronous transcription endpoint's limits.\n\nCallbacks are delivered at least once. Retries occur no sooner than 30 s after\na failure, with up to ten recorded failures. Deduplicate callbacks by job id.\nCompleted status and callbacks describe the transcript only.\n\nThe job record expires seven days after its last update. Each stored object\nexpires seven days after it was written. DELETE removes the job record and\nstored audio/results. It does not remove separately retained captures or usage,\nand cancels an unfinished platform capture. Local timeout or cancellation does\nnot delete an accepted server job; call delete explicitly when needed.\n\n## Streaming audio\n\n`sendAudio` sends one binary frame per call and validates it: 20 to 1000 ms of complete interleaved samples. For Opus, send exactly one raw packet per call; the duration rule does not apply.\n\n`streamChunks` accepts chunks of any size and re-cuts them into exact 160 ms frames with a final 20 to 160 ms tail:\n\n```ts\nfor await (const segment of session.streamChunks(microphoneChunks)) {\n  process.stdout.write(segment.text);\n}\n```\n\n`streamFile` (Node only) streams a 16-bit PCM WAV file whose format must match the session, or headerless audio with `{ raw: true }`. File streaming is not paced to realtime: the file is sent as fast as the socket drains, so large files buffer according to socket drain rather than playing out at audio speed.\n\nDuring quiet periods the session sends `KeepAlive` automatically every 25 seconds of send-side silence, well inside the server's 60 second idle deadline. Tune with `keepAliveIntervalMs`; 0 or null disables it.\n\n## Displaying words live\n\nSegments carry raw decoded text deltas, often subword pieces. Two helpers turn them into display text. `TranscriptAssembler` builds the full transcript by plain concatenation. `WordAssembler` maintains a live word list: a piece starting with a space begins a new word, and a piece without one grows the previous word in place, so \"dr\" appears instantly and becomes \"dry\" when the next piece arrives.\n\n```ts\nimport { TranscriptAssembler, WordAssembler } from \"@bandwidth-labs/bw-stt\";\n\nconst transcript = new TranscriptAssembler();\nconst words = new WordAssembler();\nsession.on(\"segment\", (segment) => {\n  transcript.push(segment);\n  const line = words.push(segment).map((word) => word.text).join(\" \");\n  redraw(line); // \"i need a dr\" then \"i need a dry van\"\n});\n```\n\nSee `examples/transcribe-wav.mts` for a complete CLI:\n\n```sh\nBW_STT_API_KEY=bwa_key_... node --import tsx examples/transcribe-wav.mts call.wav\n```\n\n## PII redaction\n\nAsk the service to redact common US PII categories in results:\n\n| Option | Type | Default | Description |\n|---|---|---|---|\n| `redactPii` | `boolean` | `false` | Redact personally identifiable information. |\n| `redactPiiSub` | `\"entity_name\" \\| \"hash\"` | unset | Choose the replacement style. |\n| `redactPiiReturn` | `boolean` | `false` | Return redacted entity spans. Requires `redactPii: true` and hash substitution. |\n\n```ts\nconst session = await client.connect({\n  mode: \"demand\",\n  redactPii: true,\n  redactPiiReturn: true,                      // omit redactPiiSub for the server's hash default\n});\nconst transcripts = await session.finalizeTranscript();\nfor (const transcript of transcripts) {\n  const entitiesByToken = new Map(\n    (transcript.redactedEntities ?? []).map((entity) => [entity.token, entity]),\n  );\n  for (const token of transcript.text.split(/\\s+/)) {\n    const entity = entitiesByToken.get(token);\n    if (entity !== undefined) console.log(token, \"maps to\", entity.text, entity.kind);\n  }\n}\n```\n\nThis synthetic demand `Transcript` uses an invalid SSN placeholder:\n\n```json\n{\n  \"type\": \"Transcript\",\n  \"channel\": 0,\n  \"text\": \"my number is hash:v1:9f2c41d08ab37e15\",\n  \"words\": [],\n  \"redaction\": {\n    \"applied\": true,\n    \"entities_redacted\": 1\n  },\n  \"redacted_entities\": [\n    {\n      \"token\": \"hash:v1:9f2c41d08ab37e15\",\n      \"kind\": \"pii\",\n      \"text\": \"000-00-0000\",\n      \"start\": 2.10,\n      \"end\": 2.45\n    }\n  ]\n}\n```\n\n`redactedEntities` is `undefined` when the server omits the field and an empty\narray when the server sends an empty array. Each `token` is the exact hash\ntoken in the redacted text, so it can be joined to the corresponding entity.\n`start` and `end` are `null` when the server has no timestamps. The same\noptions work on `transcribe` and `transcribeFile`. When `redactPiiReturn: true`,\n`redactPiiSub: \"entity_name\"` is invalid. Redaction covers common US PII\ncategories selected by the service.\n\n## Keyword boosting\n\nBoost recognition of up to 100 domain terms with a combined limit of 4096 UTF-8\nbytes:\n\n```ts\nconst session = await client.connect({ keywords: [\"dry van\", \"reefer\", \"backhaul\"] });\n```\n\n## Error handling\n\nConnection-time and transcribe failures reject with typed errors: `AuthenticationError` (401/403), `RateLimitError` with `retryAfterSeconds` (429), `InvalidRequestError` (400, 413, and other unexpected 4xx on transcribe), and `ServiceUnavailableError` for 5xx and transport-level failures, including network errors and request timeouts. `TranscriptionTimeoutError` represents an expired asynchronous job `wait()` deadline. `ConnectionClosedError` covers a WebSocket that drops mid-session or an upgrade rejection the transport cannot classify (browsers only expose a generic close).\n\nJob requests use `JobLimitError` for `job_limit_reached` and\n`job_submission_busy`; both expose `retryAfterSeconds` when the server sends\n`Retry-After`. `JobPlatformUnavailableError` represents\n`job_platform_unavailable`, and `TranscriptionNotFoundError` represents an\nunknown or inaccessible job id.\n\nIn-band `Error` events, including `transcript_too_large`, do not throw; they\narrive through `session.on(\"error\", ...)` and `session.events()`. If the\nconnection then drops before `SessionClosed`, pending iterators and\n`closeStream()` reject with a `ConnectionClosedError` whose `lastErrorEvent`\ncarries that event. A failed final delivery is reported by\n`SessionClosed.deliveryFailed`.\n\n```ts\nimport { RateLimitError } from \"@bandwidth-labs/bw-stt\";\n\ntry {\n  const session = await client.connect();\n} catch (error) {\n  if (error instanceof RateLimitError) scheduleRetry(error.retryAfterSeconds);\n  else throw error;\n}\n```\n\nThere is no resume protocol: after an unexpected close, connect again and decide what audio to resend.\n\nInvalid local input (bad frame sizes, misaligned samples, too many keywords, or\nmore than 4096 combined keyword bytes) throws plain `RangeError` or `TypeError`\nat the call site.\n\n## Events\n\nAll server events are available as a typed union via `session.events()` or\n`session.on(\"event\", ...)`. Demand `Transcript` events are also available via\n`session.on(\"transcript\", ...)`. `events()` does not yield `SessionOpened`: it\nis consumed by the connect handshake and available as `session.opened`. Field\nnames are camelCase mappings of the wire names (`audio_duration_seconds`\nbecomes `audioDurationSeconds`), and every event keeps the original payload on\n`.raw`. Event types this SDK does not know yet are surfaced as `UnknownEvent`\nrather than dropped.\n\n## License\n\nMIT. See [LICENSE](./LICENSE).\n\n\n\n\nFetch rejects redirects with `ProtocolError` when the runtime identifies the\nredirect failure. Fetch does not expose the rejected redirect status. Browsers\nmay report an indistinguishable network failure as `ServiceUnavailableError`.\n","readmeFilename":"README.md"}