{"_id":"@bachstudio/mcp-asr-diarize","_rev":"10-590efcf0dfe2a2cc002f5a3904ddef1d","name":"@bachstudio/mcp-asr-diarize","dist-tags":{"latest":"2.0.1"},"versions":{"1.0.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.0.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.0.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"2d3b07787af5c9bc49685f46cdb77355f805f281","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.0.0.tgz","fileCount":4,"integrity":"sha512-ZeNzhef72Epkpn2GrFtRUnsSI9GgpSxdBYtXYFQI3ChZoGlNrY7WA8zpKgoqniaPj9Ixr6mKJUJl2fuc9F1HCg==","signatures":[{"sig":"MEYCIQC1ShDSI7xKJiHJ+etDnALf6HqB5QefNK9+4wFtpnqyUgIhAPQ7HyVEp9IJ0tBHZJtv7cNbomIZlpN5PzYEGUfDfj0r","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":14311},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.0.0_1788751708776_0.18653677100717303","host":"s3://npm-registry-packages-npm-production"}},"1.1.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.1.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.1.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"f48fed476b607e0697ada4ba228671154b2a8942","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.1.0.tgz","fileCount":4,"integrity":"sha512-dFNapF6kyYM/nv5enFbabmYbDZjvxglIxu3UbRcSEZXq5L4mteQYTevtEHDykVhlb5vwAnP1SySoFO3ArkNyCQ==","signatures":[{"sig":"MEUCIQDx3xcMc6MgmyUlGilBKL+Bo4YmoI67etbOr9odAViS0wIgXBNHoD36IuiQ/vL9oE7XyuUgO1JKxQiaj5J9RZKa9C0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEQCID3aAn0GmgG+o5L8GizRNT0U5MmMthKJN3tGZTdn4XEjAiAvplWaasY1z0ydnbAoDiM6Doh+qrYBcDIcogbkseyv+Q==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":15689},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.1.0_1788759164609_0.8867181040163457","host":"s3://npm-registry-packages-npm-production"}},"1.2.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.2.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.2.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"8887a53036d82b2aabe428bda4d0bc4616b2e46e","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.2.0.tgz","fileCount":4,"integrity":"sha512-IjZxyttD4CKCTWVfTrZxmGTRt07ryLhSGfQ/BNM9NYfDDJRaf3kObH5cF8xAyfZA4a7az0g1sTl+TjqO8dn0gg==","signatures":[{"sig":"MEYCIQC0caTwtDNkNhx74mK6rlx/iqNfIJy5YHVspLOQeB2NPQIhALKiyVnyVpbhWTwK+9Pk+t3W5kwoJylkHiGohlXK8uBy","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEYCIQDha4ky1PWqXx93xvD0Ql7U/XCzf4T/9y/FnSfVeJ0v0AIhAP8RtzRv91rsMKICrRCVS0NfO8deZQALnGIL5unklwF+","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":21662},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.2.0_1788762469859_0.3003753824984223","host":"s3://npm-registry-packages-npm-production"}},"1.3.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.3.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.3.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"1387e0d74208b3dc3168873f4814a6af2e2c4a78","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.3.0.tgz","fileCount":4,"integrity":"sha512-t78l/JbcjzLqcbsBRLgaBooqF0moJtvAl6sg0mNHSpCYzVE5NcRww9iQfP4Y/9eZVBwOQtJU+Qq+b92a1hZy1w==","signatures":[{"sig":"MEQCIGOkIEo3JCwvshYCSdBaOzLBrpev3F3LH1G7h/6y+w2RAiAiN/bDVzHBHXA3j9u0ts7jrnJc7tD9Rw4MM6kHiVf7yw==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEUCIQCW08XEhnrx59euS4JDnH3XGEocWLCU2dLrJsYydT6YfgIgBRIWYuEgviGl9/Rx8zpiaRGGqlkyzSpjeQbBKvyfFN8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":28082},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.3.0_1788764422132_0.5785473537472907","host":"s3://npm-registry-packages-npm-production"}},"1.4.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.4.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.4.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"53332b17d5c7c92706562ef636064e19ea6203d3","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.4.0.tgz","fileCount":4,"integrity":"sha512-XbspOLuyEngtwrs1WUdGeIu2fBWEaF+S0v+T7+I0LF1pxH0Kfs6f9UBmHbiu417HltaBKmhpQOIpKLGfcrEeQw==","signatures":[{"sig":"MEQCIFPQ/s9qs/DPj0UFIk6j/I1VLbPXk8Ne/AmoHKwBU+oNAiBF46hfd8+XA2p/x/7sioW3IqVluSD6D8c+MK25VoX1Jg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEUCIH1tqrCprBczecdoHqMbYiyJsw14aYEbhqta4anSQYw9AiEAsBdOBzczV0anS4l6+0Ri1FuJzsjsjaMAsh63UoliJcg=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":29416},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.4.0_1788766031348_0.08034260351881839","host":"s3://npm-registry-packages-npm-production"}},"1.5.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.5.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.5.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"26dc1b81090a90b6570927ecc053d3ebfed94ca3","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.5.0.tgz","fileCount":4,"integrity":"sha512-KWNWuUDoR7ab+OJVq7m2u+lMtUk3G7vkFZVP4vir7LfdAPKuxHyvsFM8CJrx/K6Qp1M6nLBU3yO14rR6fVTv/w==","signatures":[{"sig":"MEYCIQC2cRhetk+Qlja/OAUbyCjFSAPdsniebUs6eBxH24Y+bQIhANXyHw/gn0C5DPowljMIKfoJtQmOHeok7w7dRmOm8Icc","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEQCIGMXtO3YxAWcwdhOjsh2T3Xuq+nGFLPx9RvIdJ+evLJkAiBdVWytsredciWW/Lg4ZKWCVzsoyPcaqzN2QIB2b8ROkw==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":31042},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.5.0_1788767638343_0.5247546370969476","host":"s3://npm-registry-packages-npm-production"}},"1.6.0":{"name":"@bachstudio/mcp-asr-diarize","version":"1.6.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.6.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"cd042983e3b8713d78f1bc036d7854464bea7dbd","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.6.0.tgz","fileCount":4,"integrity":"sha512-+zHJeB32ymn3e++Iu/eX6IL8t6p65i7SXP2VfZhSsLF9kAEtBVFFaMGmS8fDOyinS92XFoNADTPVPQ3+MwLbuQ==","signatures":[{"sig":"MEUCIQDmONw39L22nl7xq6tvkvNIjNTErRVI6Syh0NJgGUkrowIgODjr2HmAxjRK5Slqt/DJAYMuot0Jxf/bq1sNA7odZJY=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEYCIQDFg3i+QgjKsO8o3L+BOKJYp0AQaYa7ZdBcEJsvkzKSfQIhAO90DBIs6/OGDB3w7bq49eOURbYnujYvMkDsEYdYfI12","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":32025},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.6.0_1788774940335_0.5311277100092091","host":"s3://npm-registry-packages-npm-production"}},"1.6.1":{"name":"@bachstudio/mcp-asr-diarize","version":"1.6.1","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@1.6.1","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"efc67644809cb5c6693bbc6df352abfb127eb1c9","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-1.6.1.tgz","fileCount":4,"integrity":"sha512-kONARyaz++gjOIrMCRPpNsFmJmttP76dWSxEjB2ftvcSfoLa9H7ceDKR0vN+a45iqw9CjkQlD9HNr4rU2K9OdQ==","signatures":[{"sig":"MEYCIQDJOC2UI/LF9vwzZibf00wg0OSjE1SkIC4dypaTWzVjMAIhAIU9dQoMwzRh20ElrDuhkPLdDzxdgjB9E1GHyjwkc085","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEYCIQDx+Fj591ptuh/ZobL+IDWmTxZoUqJ9hILUj0eYjd+LkQIhANO6gkLqTfR9+e0INJVCmgbUsp1++YSqJ8YdzhLkoxP/","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":31964},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_1.6.1_1788775101700_0.26749600439529275","host":"s3://npm-registry-packages-npm-production"}},"2.0.0":{"name":"@bachstudio/mcp-asr-diarize","version":"2.0.0","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"author":{"name":"bachstudio"},"license":"MIT","_id":"@bachstudio/mcp-asr-diarize@2.0.0","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"0562b89d963c34754820cd642eb6fc7fc09401d3","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-2.0.0.tgz","fileCount":4,"integrity":"sha512-CEOAfoCd0Gz5fBwrPodDRwE+b4XUjNQDjzynHwbhD1nlttOKphAlhNbk4wcWQiV7cr8p1CezsN4uS/D+gWtRFg==","signatures":[{"sig":"MEUCIQC82ioWVCp5gGdwpqFDmzjUuy2Jr3IBn85myE6zumI9oQIgNALjdi3K2Vgy55vNaQ/usr6FtMUmXL/MtpVenw0j5OI=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEUCIQDwLlaHnFRoR+RxqdQbgCnTgJSSuyRTwXR+uNVR8kTXAAIgNGp15zKX1pmju24Gc0NLjHQAG56ZqaBxl2P/RDIBtOM=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":32801},"main":"src/index.js","type":"commonjs","engines":{"node":">=18"},"scripts":{"start":"node src/index.js"},"_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/mcp-asr-diarize_2.0.0_1788938555729_0.503137497955839","host":"s3://npm-registry-packages-npm-production"}},"2.0.1":{"_id":"@bachstudio/mcp-asr-diarize@2.0.1","bin":{"mcp-asr-diarize":"src/index.js"},"dist":{"shasum":"f2f5bc6aad7be9aa63530dd33274cfa6a92478d2","tarball":"https://registry.npmjs.org/@bachstudio/mcp-asr-diarize/-/mcp-asr-diarize-2.0.1.tgz","fileCount":4,"integrity":"sha512-g/8oeJhNGiCjpKcS6lC6ixJKMY/Jji0TUYi0aW6K77Je7s9lXGWSNA0j+FJdUc2s0T7pcHWqm/y6KeU0YzxYDQ==","signatures":[{"sig":"MEUCIB5iQvHk5cELvfy8TpwyHfec4SHGOs4ZhOC2NsQz/dgkAiEAs+4IxHMBam0fXbyGgR+FUlFgCohjwSDcCId87f0fxUI=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIGphOZBCM5cTE7hMSN3p1cyRAO2KcerihzvfLeXYTiPGAiBWEe7GFsW1LgG4jIE067LpdMOvT8iFBMaBhUrtDBcL1A=="}],"unpackedSize":33490},"main":"src/index.js","name":"@bachstudio/mcp-asr-diarize","type":"commonjs","author":{"name":"bachstudio"},"engines":{"node":">=18"},"license":"MIT","scripts":{"start":"node src/index.js"},"version":"2.0.1","_npmUser":{"name":"bachstudio","email":"bach_mcp_studio@bach.team"},"keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"_npmVersion":"10.8.2","description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","directories":{},"maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/mcp-asr-diarize_2.0.1_1788940181171_0.5870263509270717"}}},"time":{"created":"2026-09-07T03:28:28.629Z","modified":"2026-09-09T07:49:41.405Z","1.0.0":"2026-09-07T03:28:28.895Z","1.1.0":"2026-09-07T05:32:44.693Z","1.2.0":"2026-09-07T06:27:49.946Z","1.3.0":"2026-09-07T07:00:22.211Z","1.4.0":"2026-09-07T07:27:11.440Z","1.5.0":"2026-09-07T07:53:58.441Z","1.6.0":"2026-09-07T09:55:40.420Z","1.6.1":"2026-09-07T09:58:21.869Z","2.0.0":"2026-09-09T07:22:35.819Z","2.0.1":"2026-09-09T07:49:41.250Z"},"author":{"name":"bachstudio"},"license":"MIT","keywords":["mcp","modelcontextprotocol","asr","speech-to-text","speaker-diarization","voiceprint","whisper","transcription","meeting","realtime","live-transcription","streaming"],"description":"MCP server for speaker-attributed speech-to-text: tells apart who is speaking (voiceprint / diarization) and transcribes what they said. Point it at any backend with --url.","maintainers":[{"name":"bachstudio","email":"bach_mcp_studio@bach.team"}],"readme":"# @bachstudio/mcp-asr-diarize\n\nMCP server for **speaker-attributed speech-to-text** — it tells apart *who* is\nspeaking and transcribes *what* they said, in one call.\n\nPoint it at a meeting recording and you get back:\n\n```\n模式=diarize  时长=1832.4s  语种=zh  说话人=张三、李四、说话人3\n耗时=41.7s（decode:1.2s asr:33.8s diar:6.7s）  加速比=43.9x 实时\n--------------------------------------------------------\n[00:00:03] 张三: 各位好，今天的会议主要讨论第三季度的产品排期。\n[00:00:11] 李四: 我这边的结论是，视频生成模块要往后推两周。\n[00:00:19] 说话人3: 预算方面没有问题，人力是主要瓶颈。\n```\n\nSpeakers whose voiceprint you have enrolled come back with their real name;\neveryone else gets `说话人1 / 2 / 3`.\n\n## Install\n\nBackend on the same machine (default `http://127.0.0.1:8811`):\n\n```bash\nclaude mcp add asr -- npx -y @bachstudio/mcp-asr-diarize\n```\n\nBackend somewhere else — pass `--url`:\n\n```bash\nclaude mcp add asr -- npx -y @bachstudio/mcp-asr-diarize --url https://your-host/asr\n```\n\nOr in any MCP client's config:\n\n```json\n{\n  \"mcpServers\": {\n    \"asr\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@bachstudio/mcp-asr-diarize\", \"--url\", \"https://your-host/asr\"]\n    }\n  }\n}\n```\n\nZero dependencies, Node >= 18. `--help` prints every flag.\n\n| Flag | Env | Default | Meaning |\n| --- | --- | --- | --- |\n| `--url` | `ASR_BASE_URL` | `http://127.0.0.1:8811` | Backend address. A path prefix is fine (`https://host/asr`) |\n| `--token` | `ASR_TOKEN` | — | Bearer token, if the backend requires one |\n| `--timeout` | `ASR_TIMEOUT` | `1800` | Per-request timeout, seconds |\n\nWhen the backend is on `127.0.0.1`, files are handed over by absolute path (no copy).\nFor any other address the file is uploaded over `POST /upload` first, so a remote GPU box\nbehind a tunnel works with no shared filesystem.\n\n## Requires a backend\n\nThis package is the **MCP front end only**. It talks to an HTTP backend that\nowns the models. The reference backend runs faster-whisper (CTranslate2, fp16)\non the GPU and sherpa-onnx (pyannote segmentation-3.0 + 3D-Speaker CAM++\nembeddings) on the CPU, in parallel — on an RTX 4090 that lands around **40x\nrealtime** for a Chinese meeting, diarization included.\n\n### Backend HTTP contract\n\nAny server implementing these four routes works:\n\n| Route | Body | Returns |\n| --- | --- | --- |\n| `GET /health` | — | `{ok, loaded[], gpu{}, voiceprints[]}` |\n| `POST /transcribe` | `{audio_id \\| path, preset, language, mode, num_speakers, channel_names[], hotwords, identify}` | `{ok, mode, duration, language, speakers[], segments[{start,end,speaker,text}], text, timing{}, speedup}` |\n| `POST /enroll` | `{name, path}` | `{ok, name, seconds, enrolled[]}` |\n| `POST /upload?name=&audio_id=&final=` | raw bytes | `{ok, audio_id, path, bytes, complete, audio{}}` |\n| `POST /fetch` | `{url, name}` | same as `/upload` |\n| `POST /audio/delete` | `{audio_id}` | `{ok}` |\n\n## Tools\n\n| Tool | What it does |\n| --- | --- |\n| `upload_audio` | Hand the backend an audio file and get an `audio_id` |\n| `transcribe_meeting` | Audio/video → speaker-labelled transcript |\n| `delete_audio` | Drop an uploaded file early (they expire after 24h anyway) |\n| `enroll_voiceprint` | Register one person's voice so they get named |\n| `list_voiceprints` | List enrolled speakers |\n| `delete_voiceprint` | Remove one |\n| `asr_health` | Backend status, loaded models, free VRAM |\n| `start_live_session` | Open a live (as-you-speak) transcription session |\n| `get_live_transcript` | Pull what has been said so far, incrementally |\n| `stop_live_session` | Close it, and get a cleaned-up offline pass over the whole recording |\n| `list_live_sessions` | Sessions still open |\n\n### Live transcription\n\nMCP is request/response — it cannot stream tokens back. So the live part happens between\nyour client and the backend, and these tools let the model **look in on it** at any moment:\n\n```\naudio:  your page ──16kHz int16 PCM, chunked POST──▶ /live/push\nMCP:    start_live_session()            → session_id\n        get_live_transcript(session_id, since)  → only what is new\n        stop_live_session(session_id)   → live transcript + a polished offline re-run\n```\n\nMeasured on an RTX 4090 over a public tunnel: **text appears ~1.5s after the speaker stops**\n(0.35s end-of-utterance detection + polling interval + inference + round trip).\n\nSomeone who talks for two minutes without pausing would otherwise produce nothing until the\nforced cut, so `get_live_transcript` also returns `partials`: the sentence **currently being\nspoken**, re-transcribed roughly every 1.5s. On a 40-second stretch of unbroken speech the\nfirst words showed up at **2.4s** instead of waiting 18s for the cut. Partials are provisional —\nthey are replaced by the committed version once the sentence ends, and they never enter the\nspeaker clustering.\n\n**Nobody has to enrol first.** Strangers are separated automatically into 说话人1 / 2 / 3.\nAfter every sentence the backend re-clusters *all* voiceprints collected so far with\nagglomerative clustering, and reads the speaker count off the dendrogram's largest gap —\nso it never depends on an absolute similarity threshold, which drifts with recording quality.\nBecause every sentence is re-judged against the full picture, **earlier labels can be rewritten**\nas evidence accumulates (two clusters turning out to be one person, say). The `rev` field\nincrements whenever that happens: when it changes, re-fetch with `since: 0` instead of\ntaking the incremental slice. Pass `num_speakers` when you know it — it beats any auto-detection.\n\nAnyone who *is* enrolled gets their real name from their first sentence instead of a number.\n\n`stop_live_session` re-runs the *whole* recording through the offline pipeline. That second\npass has better punctuation and speaker boundaries than anything produced sentence-by-sentence\n— keep that one. The recording is also saved as an `audio_id` you can re-run with other settings.\n\n**Pass `language` only when the meeting really is in one language.** Sentence-at-a-time\ntranscription is far more sensitive to a wrong language hint than whole-file transcription is:\nforcing `zh` onto an English sentence makes Whisper hallucinate fluent Chinese. Leave it empty\nand each utterance is detected on its own.\n\n### `upload_audio`\n\nThree ways in — pick whichever you already have:\n\n| Param | When to use |\n| --- | --- |\n| `path` | You have the file on the machine running the MCP server |\n| `content` | Base64 bytes inline. **No address of any kind is needed**, so CORS and auth on wherever the audio came from stop mattering |\n| `url` | The backend fetches it itself — useful when the client is a browser blocked by CORS |\n\nReturns an `audio_id`. Uploads are **chunked at 8MB**, so a two-hour recording goes\nthrough reverse proxies that cap request bodies (Cloudflare's free tier stops at 100MB).\nThe file is decoded once on arrival, so a corrupt or unsupported file fails right there\ninstead of halfway through a transcription.\n\nPass the `audio_id` to `transcribe_meeting`. Re-running with different settings\n(another preset, a different speaker count) reuses the same upload — no second transfer.\n\n### `transcribe_meeting`\n\n| Param | Default | Notes |\n| --- | --- | --- |\n| `audio_id` | — | From `upload_audio`. Either this or `path` |\n| `path` | — | Local file, any container: webm/opus/m4a/mp3/wav/mp4 |\n| `preset` | `balanced` | `fast` / `balanced` / `accurate` |\n| `language` | auto | Pass `zh` / `en` when known — faster *and* more accurate |\n| `mode` | `auto` | `split` / `diarize` / `single` |\n| `num_speakers` | auto | Pass it when known; materially better clustering |\n| `speaker_names` | `[\"对方\",\"我\"]` | Channel labels for `split` mode |\n| `hotwords` | — | Proper nouns, space separated |\n| `identify` | `true` | Match clusters against enrolled voiceprints |\n\n### The `split` shortcut, and why it isn't the whole story\n\nIf the file is stereo and the two channels differ, `mode: auto` transcribes each\nchannel separately. That is the fast path for recordings captured as\n*left = them, right = me* — the \"me vs them\" split is physical, so no clustering\nerror is possible there.\n\nBut **\"them\" is usually several people**. So the far-end channel is *also* run\nthrough voiceprint clustering, and comes back as `对方1 / 对方2 / …`\n(just `对方` when there really is only one). The near-end channel is your own\nmicrophone, assumed to be one person — set `mic_multi` if a room full of people\nshares it. Turn the whole thing off with `split_diarize: false`.\n\nMeasured on a 45s stereo scene with three distinct voices on the far end and one\non the near end: 11 of 12 utterances labelled correctly, speaker count auto-detected\nexactly. The single miss was a 1.4-second \"Thanks.\" — utterances under about two\nseconds carry too little voice for a reliable embedding.\n\n## License\n\nMIT © bachstudio\n","readmeFilename":"README.md"}