{"_id":"@darkhorseprojects/pi-vo","_rev":"3-80d3527700a94e576b698e74c2cb481d","name":"@darkhorseprojects/pi-vo","dist-tags":{"latest":"0.2.2"},"versions":{"0.2.0":{"name":"@darkhorseprojects/pi-vo","version":"0.2.0","keywords":["pi","voice","omnivoice","qwen3-asr","tts","asr"],"license":"Apache-2.0","_id":"@darkhorseprojects/pi-vo@0.2.0","maintainers":[{"name":"5arrio","email":"samarixyt@gmail.com"}],"pi":{"extensions":["./src/index.ts"]},"bin":{"pi-vo-install":"scripts/install.sh"},"dist":{"shasum":"251ed01408034b8f4ebb105fd2d9c5730440ce12","tarball":"https://registry.npmjs.org/@darkhorseprojects/pi-vo/-/pi-vo-0.2.0.tgz","fileCount":18,"integrity":"sha512-k3ev/pTq4cpDpv7CsTVHN9gjuLBQ+672P6GlbNE4VGTddgDaC6JDMLPEV2KTaBySK5KMc0V8QI6ZUNeAQw1wGA==","signatures":[{"sig":"MEUCIQCMG6ckn2ov1+wJKmDYKmOT/nqdyYadUNAZ0XCRduXbeQIgdnk6H3/yKFMaFHWW6DNYdpg8q3LwC78Icc7W1Q0d7Jg=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":108816},"main":"src/index.ts","type":"module","engines":{"node":">=20"},"gitHead":"62a9cdfec591dc3749ad0773fc32005554282be7","scripts":{"check":"npm run typecheck && npm run check:python","typecheck":"tsc --noEmit -p tsconfig.json","check:python":"python3 -m py_compile workers/*.py"},"_npmUser":{"name":"5arrio","email":"samarixyt@gmail.com"},"_npmVersion":"11.13.0","description":"Local low-resource voice extension for Pi featuring sub-6GB VRAM TTS/ASR with 4-bit quantization.","directories":{},"_nodeVersion":"24.14.1","_hasShrinkwrap":false,"devDependencies":{"typescript":"latest","@types/node":"latest","@earendil-works/pi-ai":"latest","@earendil-works/pi-tui":"latest","@earendil-works/pi-coding-agent":"latest"},"peerDependencies":{"@earendil-works/pi-ai":"*","@earendil-works/pi-tui":"*","@earendil-works/pi-coding-agent":"*"},"_npmOperationalInternal":{"tmp":"tmp/pi-vo_0.2.0_1778965098290_0.5385064778528013","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@darkhorseprojects/pi-vo","version":"0.2.1","keywords":["pi","voice","omnivoice","qwen3-asr","tts","asr"],"license":"Apache-2.0","_id":"@darkhorseprojects/pi-vo@0.2.1","maintainers":[{"name":"5arrio","email":"samarixyt@gmail.com"}],"pi":{"extensions":["./src/index.ts"]},"bin":{"pi-vo-install":"scripts/install.sh"},"dist":{"shasum":"8705e236da702d6da6f6f50981459dc9cdccf3e6","tarball":"https://registry.npmjs.org/@darkhorseprojects/pi-vo/-/pi-vo-0.2.1.tgz","fileCount":18,"integrity":"sha512-gIVnV7paTfwhANu5qTHdFmJmJlBfmMgSFw0jmwEhUxOXh/DfnIemWIqeE7pKtz7ZKh/6Fo9msxgfQ+L+ZBZHrQ==","signatures":[{"sig":"MEQCIBrzjajr7VvgDuonFziqDrOi6n30qi1DakOgDNXwrwaxAiBcXx3XVSdfgQdcY1MqodccvbowyYW4hxykFOcaj1wH+w==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":108827},"main":"src/index.ts","type":"module","engines":{"node":">=20"},"gitHead":"ea9651afc9fc5b56cc987eedf8d0d629aff3558f","scripts":{"check":"npm run typecheck && npm run check:python","typecheck":"tsc --noEmit -p tsconfig.json","check:python":"python3 -m py_compile workers/*.py"},"_npmUser":{"name":"5arrio","email":"samarixyt@gmail.com"},"_npmVersion":"11.13.0","description":"Local low-resource voice extension for Pi featuring sub-6GB VRAM TTS/ASR with 4-bit quantization.","directories":{},"_nodeVersion":"24.14.1","_hasShrinkwrap":false,"devDependencies":{"typescript":"latest","@types/node":"latest","@earendil-works/pi-ai":"latest","@earendil-works/pi-tui":"latest","@earendil-works/pi-coding-agent":"latest"},"peerDependencies":{"@earendil-works/pi-ai":"*","@earendil-works/pi-tui":"*","@earendil-works/pi-coding-agent":"*"},"_npmOperationalInternal":{"tmp":"tmp/pi-vo_0.2.1_1778965333201_0.1392309278502173","host":"s3://npm-registry-packages-npm-production"}},"0.2.2":{"name":"@darkhorseprojects/pi-vo","version":"0.2.2","description":"Local low-resource voice extension for Pi featuring sub-6GB VRAM TTS/ASR with 4-bit quantization.","type":"module","main":"src/index.ts","bin":{"pi-vo-install":"scripts/install.sh"},"scripts":{"typecheck":"tsc --noEmit -p tsconfig.json","check:python":"python3 -m py_compile workers/*.py","check":"npm run typecheck && npm run check:python"},"keywords":["pi","voice","omnivoice","qwen3-asr","tts","asr"],"pi":{"extensions":["./src/index.ts"]},"engines":{"node":">=20"},"peerDependencies":{"@earendil-works/pi-ai":"*","@earendil-works/pi-coding-agent":"*","@earendil-works/pi-tui":"*"},"devDependencies":{"@earendil-works/pi-ai":"latest","@earendil-works/pi-coding-agent":"latest","@earendil-works/pi-tui":"latest","@types/node":"latest","typescript":"latest"},"license":"Apache-2.0","_id":"@darkhorseprojects/pi-vo@0.2.2","gitHead":"04ebbb7156ef838ecc1c6454d9d2bf9c4ce57d0f","_nodeVersion":"22.22.2","_npmVersion":"10.9.7","dist":{"integrity":"sha512-I4tx96WllXx9DUIkR4UNOKAyXOCMuRm9JKK3y7+ja0iWSMFpBvN4WvrDBebbx2nn59woUX3E/M83IO6X6iEMTw==","shasum":"e61a186155ef7a94ad55efbb69a9513fd0b64fdf","tarball":"https://registry.npmjs.org/@darkhorseprojects/pi-vo/-/pi-vo-0.2.2.tgz","fileCount":17,"unpackedSize":88421,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIEkfGCLalP2xSikQPRVAzJKkXuvtR0ovL19yCEX7rWZuAiAUT+2e2n4pfDkRxLNl3eEGCxY+kiJfN2hBQfJ3l1SCCw=="}]},"_npmUser":{"name":"5arrio","email":"samarixyt@gmail.com"},"directories":{},"maintainers":[{"name":"5arrio","email":"samarixyt@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/pi-vo_0.2.2_1778965478419_0.6423970148768556"},"_hasShrinkwrap":false}},"time":{"created":"2026-05-16T20:58:18.167Z","modified":"2026-05-16T21:04:38.771Z","0.2.0":"2026-05-16T20:58:18.432Z","0.2.1":"2026-05-16T21:02:13.341Z","0.2.2":"2026-05-16T21:04:38.624Z"},"license":"Apache-2.0","keywords":["pi","voice","omnivoice","qwen3-asr","tts","asr"],"description":"Local low-resource voice extension for Pi featuring sub-6GB VRAM TTS/ASR with 4-bit quantization.","maintainers":[{"name":"5arrio","email":"samarixyt@gmail.com"}],"readme":"# pi-vo\n\n[![Watch Video](https://img.shields.io/badge/▶-Watch%20Video-red?style=for-the-badge)](https://chaosdiscovery.s-ul.eu/iB5hNOsX)\n\nLocal voice extension for [Pi](https://github.com/earendil-works/pi-coding-agent) featuring resident ASR/TTS workers for low-latency speech-to-text and text-to-speech.\n\n[![npm version](https://img.shields.io/npm/v/@darkhorseprojects/pi-vo)](https://www.npmjs.com/package/@darkhorseprojects/pi-vo)&nbsp;\n[![CI](https://github.com/darkhorseprojects/pi-vo/actions/workflows/ci.yml/badge.svg)](https://github.com/darkhorseprojects/pi-vo/actions/workflows/ci.yml)&nbsp;\n[![License: Apache-2.0](https://img.shields.io/badge/License-Apache--2.0-blue.svg)](https://opensource.org/licenses/Apache-2.0)\n\n## Features\n\n- **Live Speech-to-Text**: Cohere-ASR transcribes microphone input in real-time\n- **Text-to-Speech**: OmniVoice generates expressive speech with voice design\n- **Zero-latency**: Models run as persistent workers, no per-request loading\n- **Low VRAM**: 4-bit quantization with CPU offload reduces VRAM to ~6GB\n- **Configurable**: Voice personality via voice design parameters\n\n## Usage\n\n```\n/v              Warm models and toggle live microphone\nctrl+space      Same as /v (toggle recording)\n/v 0-100        Set volume (0-100%)\n/v stop         Stop all recording, playback, and workers\n/v unload       Stop models and hide the orb\n/v stt [model]  Show or switch ASR model\n/v tts          Show TTS backend settings\n/v i [n]        List/select input devices\n/v o [n]        List/select output devices\nesc             Stop current speech/playback\nenter           Clear transcript preview\n```\n\n## Installation\n\n```bash\nnpm install\nnpm run typecheck\nscripts/install.sh\n```\n\nThe installer creates `.venv-voice`, installs PyTorch, OmniVoice, Cohere-ASR, bitsandbytes, and writes `~/.pi/pi-vo.json`.\n\n## Voice Cloning\n\nFor voice cloning, provide a reference audio file and its transcript. The audio should be 15-30 seconds of expressive speech that matches the style you want:\n\n```json\n{\n  \"ttsReferenceAudio\": \"/path/to/voice-sample.wav\",\n  \"ttsVoiceDesign\": \"female, young adult, moderate pitch\"\n}\n```\n\nThe reference audio must be 16kHz PCM WAV.\n\n### Voice Style Guidance\n\nUse `ttsVoiceDesign` to guide the delivery style (emotional tone, pace, character). Valid attributes are: gender, age, pitch, style, and accent. Combine with comma + space:\n\n```json\n{\n  \"ttsVoiceDesign\": \"female, young adult, high pitch, american accent\"\n}\n```\n\nThis parameter works independently or alongside voice cloning to shape how text is delivered.\n\n## Configuration\n\nDefault config at `~/.pi/pi-vo.json`:\n\n```json\n{\n  \"voicePython\": \"/path/to/pi-vo/.venv-voice/bin/python\",\n  \"asrModel\": \"cstr/cohere-transcribe-onnx-int4\",\n  \"asrDeviceMap\": \"cuda:0\",\n  \"asrDtype\": \"float32\",\n  \"ttsModel\": \"k2-fsa/OmniVoice\",\n  \"ttsDeviceMap\": \"cuda:0\",\n  \"ttsDtype\": \"bfloat16\",\n  \"ttsLoadIn4bit\": true,\n  \"ttsQuantType\": \"nf4\",\n  \"ttsComputeDtype\": \"bfloat16\",\n  \"ttsCpuOffload\": true,\n  \"ttsOffloadFolder\": \"~/.pi/pi-vo-offload\",\n  \"ttsReferenceAudio\": \"\",\n  \"ttsVoiceDesign\": \"\",\n  \"ttsNumSteps\": 32,\n  \"voiceSpeed\": 1.15,\n  \"voiceVolume\": 0.85,\n  \"recordSampleRate\": 16000,\n  \"audioSampleRate\": 24000\n}\n```\n\n### Environment Variables\n\n```bash\nPI_VO_VOICE_PYTHON=/path/to/.venv-voice/bin/python\nPI_VO_ASR_MODEL=cstr/cohere-transcribe-onnx-int4\nPI_VO_ASR_DEVICE_MAP=cuda:0\nPI_VO_ASR_DTYPE=float32\nPI_VO_TTS_MODEL=k2-fsa/OmniVoice\nPI_VO_TTS_DEVICE_MAP=cuda:0\nPI_VO_TTS_DTYPE=bfloat16\nPI_VO_TTS_LOAD_IN_4BIT=1\nPI_VO_TTS_QUANT_TYPE=nf4\nPI_VO_TTS_COMPUTE_DTYPE=bfloat16\nPI_VO_TTS_CPU_OFFLOAD=1\n```\n\n## Requirements\n\n- Node.js 20+\n- Python 3.12+\n- CUDA-compatible GPU\n- PipeWire or PulseAudio\n- ~6GB VRAM (with 4-bit quantization and CPU offload)\n\n---\n\n**Inspired by p8n-ai/pi-listens**\n\n## License\n\nApache 2.0 - see [LICENSE](LICENSE) file for details.","readmeFilename":"README.md"}