{"_id":"@ruvector/ruvllm-wasm","_rev":"5-0c1f1b6fb49b4c9c9fc11a7c56fc4d00","name":"@ruvector/ruvllm-wasm","dist-tags":{"latest":"2.0.2"},"versions":{"0.1.0":{"name":"@ruvector/ruvllm-wasm","version":"0.1.0","keywords":["llm","wasm","webassembly","browser","inference","webgpu","ai","machine-learning","edge","offline","ruvector","ruvllm","transformers"],"author":{"name":"rUv Team","email":"team@ruv.io"},"license":"MIT OR Apache-2.0","_id":"@ruvector/ruvllm-wasm@0.1.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/ruvector/tree/main/crates/ruvllm-wasm","bugs":{"url":"https://github.com/ruvnet/ruvector/issues"},"dist":{"shasum":"35d2dbce1700812b6d33071a778098baa5a656c0","tarball":"https://registry.npmjs.org/@ruvector/ruvllm-wasm/-/ruvllm-wasm-0.1.0.tgz","fileCount":10,"integrity":"sha512-6l6i/N6GKzL/yQlk1h1ChEb8zmd+qb3/Phv4U1gQuxmxEvjg9wCcgTD5Chv98PiMEKTawifw/5CKz9utSAh3yQ==","signatures":[{"sig":"MEUCIQD5ZTdY36P9IVpqSMjSjGYDMACa+9Hp6zn8faNihtBEfwIgaoAap86bmYlUjrKLmiPN/lz9zY0QD7/aFBjFty/t0+M=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":32550},"main":"dist/index.js","types":"dist/index.d.ts","engines":{"node":">= 18"},"exports":{".":{"import":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"require":{"types":"./dist/index.d.ts","default":"./dist/index.js"}}},"gitHead":"8f5b2bdb033b0f86da80c8ea1c6da22e1cbf8535","scripts":{"test":"node --test test/*.test.js","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","prepublishOnly":"npm run build"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"deprecated":"This version is a placeholder — WASM binary not compiled. See https://github.com/ruvnet/ruvector/issues/238","repository":{"url":"git+https://github.com/ruvnet/ruvector.git","type":"git","directory":"npm/packages/ruvllm-wasm"},"_npmVersion":"11.6.2","description":"WASM bindings for browser-based LLM inference - run AI models directly in the browser with WebGPU acceleration","directories":{},"_nodeVersion":"25.3.0","publishConfig":{"access":"public","registry":"https://registry.npmjs.org/"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.3","@types/node":"^20.19.30","@webgpu/types":"^0.1.69"},"_npmOperationalInternal":{"tmp":"tmp/ruvllm-wasm_0.1.0_1769058338034_0.9791810697491701","host":"s3://npm-registry-packages-npm-production"}},"2.0.0":{"name":"@ruvector/ruvllm-wasm","version":"2.0.0","keywords":["wasm","llm","inference","browser","webgpu"],"license":"MIT","_id":"@ruvector/ruvllm-wasm@2.0.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/ruvector#readme","bugs":{"url":"https://github.com/ruvnet/ruvector/issues"},"dist":{"shasum":"5dec34ad4cc390120b75fdda78a2508d884bd5c9","tarball":"https://registry.npmjs.org/@ruvector/ruvllm-wasm/-/ruvllm-wasm-2.0.0.tgz","fileCount":5,"integrity":"sha512-9wCN+RhipGVs6j7npEMeU6iEtiyVDCveqg50frB/xth0cQBF57u7XZ7ctpcnSQmHWZjkNcIa0dhx8i8Tr6I36A==","signatures":[{"sig":"MEUCIALgH8dtrJutMxpbbS3b0/u/5zmyei6crJpwBnVTOoApAiEAm2fCiGotXlv7KCAJN/F3TpQDElzKrp2IhfQwwF7ljbU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":613227},"main":"ruvllm_wasm.js","type":"module","types":"ruvllm_wasm.d.ts","gitHead":"55b9ab3bc7bb126e686aa360832873de4b86587a","_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/ruvector.git","type":"git"},"_npmVersion":"9.8.1","description":"WASM bindings for RuvLLM - browser-compatible LLM inference runtime with WebGPU acceleration","directories":{},"sideEffects":["./snippets/*"],"_nodeVersion":"22.21.1","collaborators":["Ruvector Team"],"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/ruvllm-wasm_2.0.0_1772808903216_0.1784961066096904","host":"s3://npm-registry-packages-npm-production"}},"2.0.1":{"name":"@ruvector/ruvllm-wasm","version":"2.0.1","keywords":["wasm","llm","inference","browser","webgpu"],"license":"MIT","_id":"@ruvector/ruvllm-wasm@2.0.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/ruvector#readme","bugs":{"url":"https://github.com/ruvnet/ruvector/issues"},"dist":{"shasum":"7b3bd2ef0640a7104fefb51fef09d2beda7446b6","tarball":"https://registry.npmjs.org/@ruvector/ruvllm-wasm/-/ruvllm-wasm-2.0.1.tgz","fileCount":5,"integrity":"sha512-OPaEslLEG6gyTH9C9AHuS5o8h4rUI8f9lozc4Ca3jQtNIXepokBZIz6eOuy+vJxMda5MUxtXK8NbQ3tPltuPzA==","signatures":[{"sig":"MEYCIQC4XNHO9kzzi4z/tMAkv3sPo1/ciBMlxPImkxIgmzj2bAIhAJJriyLXvxkCYdhtERNti3oIbGqjX6rZE7tEqWE6yqrq","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":572642},"main":"ruvllm_wasm.js","type":"module","types":"ruvllm_wasm.d.ts","gitHead":"d35ea335b4bf028701ccae0216821e3495345f82","_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/ruvector.git","type":"git"},"_npmVersion":"10.9.4","description":"WASM bindings for RuvLLM - browser-compatible LLM inference runtime with WebGPU acceleration","directories":{},"sideEffects":["./snippets/*"],"_nodeVersion":"22.22.1","collaborators":["Ruvector Team"],"_hasShrinkwrap":false,"_npmOperationalInternal":{"tmp":"tmp/ruvllm-wasm_2.0.1_1773774501368_0.4616984905727002","host":"s3://npm-registry-packages-npm-production"}},"2.0.2":{"name":"@ruvector/ruvllm-wasm","type":"module","collaborators":["Ruvector Team"],"description":"WASM bindings for RuvLLM - browser-compatible LLM inference runtime with WebGPU acceleration","version":"2.0.2","license":"MIT","repository":{"type":"git","url":"git+https://github.com/ruvnet/ruvector.git"},"main":"ruvllm_wasm.js","types":"ruvllm_wasm.d.ts","sideEffects":["./snippets/*"],"keywords":["wasm","llm","inference","browser","webgpu"],"_id":"@ruvector/ruvllm-wasm@2.0.2","gitHead":"084954f4d273bfe6d30628219c22a47d7d49a793","bugs":{"url":"https://github.com/ruvnet/ruvector/issues"},"homepage":"https://github.com/ruvnet/ruvector#readme","_nodeVersion":"22.22.1","_npmVersion":"10.9.4","dist":{"integrity":"sha512-jVu+t710JvEzU6UZZvKrTp+dWsm3hKzOqB+yxrcqFsVaccVfeKyQuvCx13qVENvuntbnfslryErn9QbydH76yQ==","shasum":"28cbbc5f1dcdae0e896bfd1361a6e29696b4c2d8","tarball":"https://registry.npmjs.org/@ruvector/ruvllm-wasm/-/ruvllm-wasm-2.0.2.tgz","fileCount":5,"unpackedSize":573244,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIFDAuEZ2E0a0uLES2sD63w1hANTlsYKr6okfObJrytaVAiBHs2rpmjKHhmanyR1K5U6FBHvMqp/zx0Qb6ARrS9ssJg=="}]},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"directories":{},"maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/ruvllm-wasm_2.0.2_1773780223945_0.4041795610772845"},"_hasShrinkwrap":false}},"time":{"created":"2026-01-22T05:05:37.964Z","modified":"2026-03-17T20:43:44.251Z","0.1.0":"2026-01-22T05:05:38.166Z","2.0.0":"2026-03-06T14:55:03.446Z","2.0.1":"2026-03-17T19:08:21.557Z","2.0.2":"2026-03-17T20:43:44.144Z"},"bugs":{"url":"https://github.com/ruvnet/ruvector/issues"},"license":"MIT","homepage":"https://github.com/ruvnet/ruvector#readme","keywords":["wasm","llm","inference","browser","webgpu"],"repository":{"type":"git","url":"git+https://github.com/ruvnet/ruvector.git"},"description":"WASM bindings for RuvLLM - browser-compatible LLM inference runtime with WebGPU acceleration","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"readme":"# @ruvector/ruvllm-wasm\n\n[![npm](https://img.shields.io/npm/v/@ruvector/ruvllm-wasm.svg)](https://www.npmjs.com/package/@ruvector/ruvllm-wasm)\n[![License](https://img.shields.io/crates/l/ruvllm-wasm.svg)](https://github.com/ruvnet/ruvector/blob/main/LICENSE)\n\nBrowser-compatible LLM inference runtime with WebAssembly. Semantic routing, adaptive learning, KV cache management, and chat template formatting — directly in the browser, no server required.\n\n## Features\n\n- **KV Cache Management** — Two-tier cache (FP32 tail + u8 quantized store) for efficient token storage\n- **Memory Pooling** — Arena allocator + buffer pool for minimal allocation overhead\n- **Chat Templates** — Llama3, Mistral, Qwen, ChatML, Phi, Gemma format support\n- **HNSW Semantic Router** — 150x faster pattern matching with bidirectional graph search\n- **MicroLoRA** — Sub-millisecond model adaptation (rank 1-4)\n- **SONA Instant Learning** — EMA quality tracking + adaptive rank adjustment\n- **Web Workers** — Parallel inference with SharedArrayBuffer detection\n- **Full TypeScript** — Complete `.d.ts` type definitions for all exports\n\n## Install\n\n```bash\nnpm install @ruvector/ruvllm-wasm\n```\n\n## Quick Start\n\n```javascript\nimport init, {\n  RuvLLMWasm,\n  ChatTemplateWasm,\n  ChatMessageWasm,\n  HnswRouterWasm,\n  healthCheck\n} from '@ruvector/ruvllm-wasm';\n\n// Initialize WASM module\nawait init();\n\n// Verify module loaded\nconsole.log(healthCheck()); // true\n\n// Format chat conversations\nconst template = ChatTemplateWasm.llama3();\nconst messages = [\n  ChatMessageWasm.system(\"You are a helpful assistant.\"),\n  ChatMessageWasm.user(\"What is WebAssembly?\"),\n];\nconst prompt = template.format(messages);\n\n// Semantic routing with HNSW\nconst router = new HnswRouterWasm(384, 1000);\nrouter.addPattern(new Float32Array(384).fill(0.1), \"coder\", \"code tasks\");\nconst result = router.route(new Float32Array(384).fill(0.1));\nconsole.log(result.name, result.score); // \"coder\", 1.0\n```\n\n## API\n\n### Core Types\n\n| Type | Description |\n|------|-------------|\n| `RuvLLMWasm` | Main inference engine with KV cache + buffer pool |\n| `GenerateConfig` | Generation parameters (temperature, top_k, top_p, repetitionPenalty) |\n| `KvCacheWasm` | Two-tier KV cache for token management |\n| `InferenceArenaWasm` | O(1) bump allocator for inference temporaries |\n| `BufferPoolWasm` | Pre-allocated buffer pool (1KB-256KB size classes) |\n\n### Chat Templates\n\n```javascript\n// Auto-detect from model ID\nconst template = ChatTemplateWasm.detectFromModelId(\"meta-llama/Llama-3-8B\");\n// Or use directly\nconst template = ChatTemplateWasm.mistral();\nconst prompt = template.format([\n  ChatMessageWasm.system(\"You are helpful.\"),\n  ChatMessageWasm.user(\"Hello!\"),\n]);\n```\n\nSupported: `llama3()`, `mistral()`, `chatml()`, `phi()`, `gemma()`, `custom(name, pattern)`\n\n### HNSW Semantic Router\n\n```javascript\nconst router = new HnswRouterWasm(384, 1000); // dimensions, max_patterns\nrouter.addPattern(embedding, \"agent-name\", \"metadata\");\nconst result = router.route(queryEmbedding);\nconsole.log(result.name, result.score);\n\n// Persistence\nconst json = router.toJson();\nconst restored = HnswRouterWasm.fromJson(json);\n```\n\n### MicroLoRA Adaptation\n\n```javascript\nconst config = new MicroLoraConfigWasm();\nconfig.rank = 2;\nconfig.inFeatures = 384;\nconfig.outFeatures = 384;\n\nconst lora = new MicroLoraWasm(config);\nconst adapted = lora.apply(inputVector);\nlora.adapt(new AdaptFeedbackWasm(0.9)); // quality score\n```\n\n### SONA Instant Learning\n\n```javascript\nconst config = new SonaConfigWasm();\nconfig.hiddenDim = 384;\nconst sona = new SonaInstantWasm(config);\n\nconst result = sona.instantAdapt(inputVector, 0.85); // quality\nconsole.log(result.applied, result.qualityEma);\n\nsona.recordPattern(embedding, \"agent\", true); // success pattern\nconst suggestion = sona.suggestAction(queryEmbedding);\n```\n\n### Parallel Inference (Web Workers)\n\n```javascript\nimport { ParallelInference, feature_summary } from '@ruvector/ruvllm-wasm';\n\nconsole.log(feature_summary()); // browser capability report\n\nconst engine = await new ParallelInference(4); // 4 workers\nconst result = await engine.matmul(a, b, m, n, k);\nengine.terminate();\n```\n\n## Build from Source\n\n```bash\n# Install prerequisites\nrustup target add wasm32-unknown-unknown\ncargo install wasm-pack\n\n# Release build (workaround for Rust 1.91 codegen bug)\nCARGO_PROFILE_RELEASE_CODEGEN_UNITS=256 CARGO_PROFILE_RELEASE_LTO=off \\\n  wasm-pack build crates/ruvllm-wasm --target web --scope ruvector --release\n\n# Dev build\nwasm-pack build crates/ruvllm-wasm --target web --scope ruvector --dev\n\n# With WebGPU support\nCARGO_PROFILE_RELEASE_CODEGEN_UNITS=256 CARGO_PROFILE_RELEASE_LTO=off \\\n  wasm-pack build crates/ruvllm-wasm --target web --scope ruvector --release -- --features webgpu\n```\n\n## Browser Compatibility\n\n| Browser | Version | Notes |\n|---------|---------|-------|\n| Chrome | 57+ | Full support |\n| Edge | 79+ | Full support |\n| Firefox | 52+ | Full support |\n| Safari | 11+ | Full support |\n\nOptional enhancements:\n- **SharedArrayBuffer**: Requires `Cross-Origin-Opener-Policy: same-origin` + `Cross-Origin-Embedder-Policy: require-corp`\n- **WebGPU**: Available with `webgpu` feature flag (Chrome 113+)\n\n## Size\n\n~435 KB release WASM (~178 KB gzipped)\n\n## Related\n\n- [`@ruvector/ruvllm`](https://www.npmjs.com/package/@ruvector/ruvllm) — Node.js LLM orchestration\n- [`ruvector`](https://www.npmjs.com/package/ruvector) — Full RuVector CLI + MCP tools\n- [ADR-084](../../docs/adr/ADR-084-ruvllm-wasm-publish.md) — Build documentation and known limitations\n\n## License\n\nMIT\n","readmeFilename":"README.md"}