{"_id":"@aviallon/wllama","_rev":"7-b744d8b03d16190b5c086e4a75e25fbb","name":"@aviallon/wllama","dist-tags":{"latest":"2.3.7-e-slow"},"versions":{"2.3.7":{"name":"@aviallon/wllama","version":"2.3.7","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Xuan Son NGUYEN","email":"contact@ngxson.com"},"license":"MIT","_id":"@aviallon/wllama@2.3.7","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"homepage":"https://github.com/ngxson/wllama#readme","bugs":{"url":"https://github.com/ngxson/wllama/issues"},"dist":{"shasum":"61c5fdbd4221da9dd8e0eb90d003add344f5b77b","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7.tgz","fileCount":67,"integrity":"sha512-TGLX47GlENkKa02S2O8xfsJI93wmYRQNI1Tn6dDooKqireL0ms4N9s6q43q6Sr+AKmBTfrk00V+1qqQoKvCs8g==","signatures":[{"sig":"MEYCIQDh/Di9Q5OeYTjMKCkhJUn8TQ6fIpixhThUMFXH6rX6xgIhAPOwrEiHv8o0DPcMBADeKn+Ll8qUk2LnY5poEaiMheiQ","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":10652425},"main":"index.js","type":"module","gitHead":"8778d7bf2aed90b40be778eaef35f3f3ad3fd0ae","scripts":{"docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","test":"vitest","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","serve":"node ./scripts/http_server.js","format":"prettier --write .","upload":"npm run format && npm run build && npm publish --access public","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","postbuild":"./scripts/post_build.sh && npm run docs","build:glue":"node ./cpp/generate_glue_prototype.js","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","test:safari":"BROWSER=safari vitest","build:worker":"./scripts/build_worker.sh","test:firefox":"BROWSER=firefox vitest","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map"},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"prettier":{"semi":true,"tabWidth":2,"singleQuote":true,"trailingComma":"es5","bracketSameLine":false},"repository":{"url":"git+https://github.com/ngxson/wllama.git","type":"git"},"_npmVersion":"10.9.4","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","directories":{"example":"examples"},"_nodeVersion":"22.21.1","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.4.0","terser":"^5.39.0","express":"^4.18.3","typedoc":"^0.27.2","prettier":"^3.3.3","mime-types":"^2.1.35","playwright":"^1.49.0","typescript":"^5.4.2","webdriverio":"^9.4.1","@vitest/browser":"^2.1.6"},"_npmOperationalInternal":{"tmp":"tmp/wllama_2.3.7_1768408581741_0.66001172905278","host":"s3://npm-registry-packages-npm-production"}},"2.3.7-a":{"name":"@aviallon/wllama","version":"2.3.7-a","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Xuan Son NGUYEN","email":"contact@ngxson.com"},"license":"MIT","_id":"@aviallon/wllama@2.3.7-a","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"homepage":"https://github.com/ngxson/wllama#readme","bugs":{"url":"https://github.com/ngxson/wllama/issues"},"dist":{"shasum":"0da183ce93454645278dbdf5f4b3e35ecc5d20a5","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7-a.tgz","fileCount":67,"integrity":"sha512-rV6g+YZgDj1RR1iAdA0de7jwHyWED1Z+gNBaLyzRqSIUCGaey23tf6QBCa/CSCMN0PN18cQ5yKsDoNC3/3kStw==","signatures":[{"sig":"MEUCIQDz8ucHPGJ1voThzS925RcgcLu1qVbKJ26378U1/jgchwIgFpHRnJGUo4DCT/nAV7epmk0pcpdqnFlJNtwcxfT2tpA=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":10652428},"main":"index.js","type":"module","gitHead":"8778d7bf2aed90b40be778eaef35f3f3ad3fd0ae","scripts":{"docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","test":"vitest","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","serve":"node ./scripts/http_server.js","format":"prettier --write .","upload":"npm run format && npm run build && npm publish --access public","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","postbuild":"./scripts/post_build.sh && npm run docs","build:glue":"node ./cpp/generate_glue_prototype.js","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","test:safari":"BROWSER=safari vitest","build:worker":"./scripts/build_worker.sh","test:firefox":"BROWSER=firefox vitest","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map"},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"prettier":{"semi":true,"tabWidth":2,"singleQuote":true,"trailingComma":"es5","bracketSameLine":false},"repository":{"url":"git+https://github.com/ngxson/wllama.git","type":"git"},"_npmVersion":"10.9.4","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","directories":{"example":"examples"},"_nodeVersion":"22.21.1","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.4.0","terser":"^5.39.0","express":"^4.18.3","typedoc":"^0.27.2","prettier":"^3.3.3","mime-types":"^2.1.35","playwright":"^1.49.0","typescript":"^5.4.2","webdriverio":"^9.4.1","@vitest/browser":"^2.1.6"},"_npmOperationalInternal":{"tmp":"tmp/wllama_2.3.7-a_1768408818386_0.09597837671832798","host":"s3://npm-registry-packages-npm-production"}},"2.3.7-b":{"name":"@aviallon/wllama","version":"2.3.7-b","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Xuan Son NGUYEN","email":"contact@ngxson.com"},"license":"MIT","_id":"@aviallon/wllama@2.3.7-b","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"homepage":"https://github.com/ngxson/wllama#readme","bugs":{"url":"https://github.com/ngxson/wllama/issues"},"dist":{"shasum":"ac39b5fa3eee38a21b26391899bde5b27160c626","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7-b.tgz","fileCount":67,"integrity":"sha512-l3ssWUlxOS54FR2AKq3A0Dyw7iIA7me/SiW2FVMU0BznnP8RQBCpL1LgMEZIgdMQIlAmhG5+z0+FMbQ9xrG+sw==","signatures":[{"sig":"MEYCIQDezyDac51uAz0jZtpDmQkn9JOiCQzb+TNmZ365yAdL2wIhALGGFcyNE0NU5c2gRkgv1FQ/Z5RD1ArgMJs/EKoikBj5","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":11712040},"main":"index.js","type":"module","gitHead":"8778d7bf2aed90b40be778eaef35f3f3ad3fd0ae","scripts":{"docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","test":"vitest","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","serve":"node ./scripts/http_server.js","format":"prettier --write .","upload":"npm run format && npm run build && npm publish --access public","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","postbuild":"./scripts/post_build.sh && npm run docs","build:glue":"node ./cpp/generate_glue_prototype.js","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","test:safari":"BROWSER=safari vitest","build:worker":"./scripts/build_worker.sh","test:firefox":"BROWSER=firefox vitest","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map"},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"prettier":{"semi":true,"tabWidth":2,"singleQuote":true,"trailingComma":"es5","bracketSameLine":false},"repository":{"url":"git+https://github.com/ngxson/wllama.git","type":"git"},"_npmVersion":"10.9.4","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","directories":{"example":"examples"},"_nodeVersion":"22.21.1","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.4.0","terser":"^5.39.0","express":"^4.18.3","typedoc":"^0.27.2","prettier":"^3.3.3","mime-types":"^2.1.35","playwright":"^1.49.0","typescript":"^5.4.2","webdriverio":"^9.4.1","@vitest/browser":"^2.1.6"},"_npmOperationalInternal":{"tmp":"tmp/wllama_2.3.7-b_1768411141829_0.0600411046966669","host":"s3://npm-registry-packages-npm-production"}},"2.3.7-c":{"name":"@aviallon/wllama","version":"2.3.7-c","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Xuan Son NGUYEN","email":"contact@ngxson.com"},"license":"MIT","_id":"@aviallon/wllama@2.3.7-c","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"homepage":"https://github.com/ngxson/wllama#readme","bugs":{"url":"https://github.com/ngxson/wllama/issues"},"dist":{"shasum":"6238930e8dd7dd90fde3af3911ef0a4d054bb42d","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7-c.tgz","fileCount":69,"integrity":"sha512-xf60IPqjO0h0jKfMG2FY8M9YmDekhwOBw2Zp1hASknlM8diaC8HqtRTN/IK01NljojWs8hZGJiW/5LYUikKGTw==","signatures":[{"sig":"MEUCIQDLH6BkauTT8NELqONC24Eut0ItVD3IdkjIJrGS9smm1wIgWfwd24ywJWnchQ6M/1D1DTglBNwzOERvTL5xZWMMUAw=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":11715321},"main":"index.js","type":"module","gitHead":"8778d7bf2aed90b40be778eaef35f3f3ad3fd0ae","scripts":{"docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","test":"vitest","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","serve":"node ./scripts/http_server.js","format":"prettier --write .","upload":"npm run format && npm run build && npm publish --access public","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","postbuild":"./scripts/post_build.sh && npm run docs","build:glue":"node ./cpp/generate_glue_prototype.js","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","test:docker":"docker-compose -f ./scripts/docker-compose.test.yml up --build --abort-on-container-exit --exit-code-from wllama-test --remove-orphans wllama-test","test:safari":"BROWSER=safari vitest","build:worker":"./scripts/build_worker.sh","test:firefox":"BROWSER=firefox vitest","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map"},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"prettier":{"semi":true,"tabWidth":2,"singleQuote":true,"trailingComma":"es5","bracketSameLine":false},"repository":{"url":"git+https://github.com/ngxson/wllama.git","type":"git"},"_npmVersion":"10.9.4","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","directories":{"example":"examples"},"_nodeVersion":"22.21.1","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.4.0","terser":"^5.39.0","express":"^4.18.3","typedoc":"^0.27.2","prettier":"^3.3.3","mime-types":"^2.1.35","playwright":"^1.49.0","typescript":"^5.4.2","webdriverio":"^9.4.1","@vitest/browser":"^2.1.6"},"_npmOperationalInternal":{"tmp":"tmp/wllama_2.3.7-c_1768417933160_0.8321157167918987","host":"s3://npm-registry-packages-npm-production"}},"2.3.7-d":{"name":"@aviallon/wllama","version":"2.3.7-d","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Antoine VIALLON","email":"antoine.viallon@justai.co"},"license":"MIT","_id":"@aviallon/wllama@2.3.7-d","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"homepage":"https://git.justai.co/JustAI/brio-wllama#readme","bugs":{"url":"https://git.justai.co/JustAI/brio-wllama/issues"},"dist":{"shasum":"80b216c7710387f452b941b6014ad667fad3c8f6","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7-d.tgz","fileCount":88,"integrity":"sha512-ZsZIKYYQJtVXRyxp5ZllHLXNHjkIjpehaIL/v693q03eyvpSO3YEykcwB5XQc/+1Ur8mttdIi0JJZ1xyAZQiLg==","signatures":[{"sig":"MEQCICfNSpAR8ogxcPL/23wAWARKo4/jnN+1dd0vkBWm1jKQAiAnDTNqGNW5jzPKrxXKVRmUADBMCvVptG5dtq5dZfDV1A==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":11345926},"main":"index.js","type":"module","gitHead":"e467d4f1ff7070d308119ff032b188850d5329af","scripts":{"docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","test":"vitest","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","serve":"node ./scripts/http_server.js","format":"prettier --write .","upload":"npm run format && npm run build && npm publish --access public","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","postbuild":"./scripts/post_build.sh && npm run docs","build:glue":"node ./cpp/generate_glue_prototype.js","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","test:docker":"./scripts/test_in_docker.sh","test:safari":"BROWSER=safari vitest","build:worker":"./scripts/build_worker.sh","test:firefox":"BROWSER=firefox vitest","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map"},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"prettier":{"semi":true,"tabWidth":2,"singleQuote":true,"trailingComma":"es5","bracketSameLine":false},"repository":{"url":"git+https://git.justai.co/JustAI/brio-wllama.git","type":"git"},"_npmVersion":"10.9.4","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","directories":{"example":"examples"},"_nodeVersion":"22.21.1","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.4.0","terser":"^5.39.0","express":"^4.18.3","typedoc":"^0.27.2","prettier":"^3.3.3","mime-types":"^2.1.35","playwright":"^1.49.0","typescript":"^5.4.2","webdriverio":"^9.4.1","@vitest/browser":"^2.1.6"},"_npmOperationalInternal":{"tmp":"tmp/wllama_2.3.7-d_1768426201189_0.23399908687583904","host":"s3://npm-registry-packages-npm-production"}},"2.3.7-e":{"name":"@aviallon/wllama","version":"2.3.7-e","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Antoine VIALLON","email":"antoine.viallon@justai.co"},"license":"MIT","_id":"@aviallon/wllama@2.3.7-e","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"homepage":"https://git.justai.co/JustAI/brio-wllama#readme","bugs":{"url":"https://git.justai.co/JustAI/brio-wllama/issues"},"dist":{"shasum":"f518648e43561170c9e99ec98c3f4642afac3f70","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7-e.tgz","fileCount":89,"integrity":"sha512-5mV2qMZ0lhM4yVZQvO+mj5X3Pvalzzs4RUceIlCUB9aPSQifpwcW7t2NptbyjbVtVYxsYRu0tCTh6nWWLrB2ww==","signatures":[{"sig":"MEUCIQCTRw77HlxKp6YlLT9xa8eIqtgt5G9cZctkV9OK9foAUgIgF2VX0Qqxic0/okD0///GFNucZZYVG+2xrAzzm2Z+Lbs=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":11388985},"main":"index.js","type":"module","gitHead":"e467d4f1ff7070d308119ff032b188850d5329af","scripts":{"docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","test":"vitest","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","serve":"node ./scripts/http_server.js","format":"prettier --write .","upload":"npm run format && npm run build && npm publish --access public","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","postbuild":"./scripts/post_build.sh && npm run docs","test:wasm":"./scripts/test_wasm.sh","build:glue":"node ./cpp/generate_glue_prototype.js","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","test:docker":"./scripts/test_in_docker.sh","test:safari":"BROWSER=safari vitest","build:worker":"./scripts/build_worker.sh","test:firefox":"BROWSER=firefox vitest","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map"},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"prettier":{"semi":true,"tabWidth":2,"singleQuote":true,"trailingComma":"es5","bracketSameLine":false},"repository":{"url":"git+https://git.justai.co/JustAI/brio-wllama.git","type":"git"},"_npmVersion":"10.9.4","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","directories":{"example":"examples"},"_nodeVersion":"22.21.1","_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.4.0","terser":"^5.39.0","express":"^4.18.3","typedoc":"^0.27.2","prettier":"^3.3.3","mime-types":"^2.1.35","playwright":"^1.49.0","typescript":"^5.4.2","webdriverio":"^9.4.1","@vitest/browser":"^2.1.6"},"_npmOperationalInternal":{"tmp":"tmp/wllama_2.3.7-e_1768480151031_0.6465341875398418","host":"s3://npm-registry-packages-npm-production"}},"2.3.7-e-slow":{"name":"@aviallon/wllama","version":"2.3.7-e-slow","description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","main":"index.js","type":"module","directories":{"example":"examples"},"scripts":{"serve":"node ./scripts/http_server.js","serve:mt":"MULTITHREAD=1 node ./scripts/http_server.js","clean":"rm -rf ./esm && rm -rf ./docs && rm -rf ./wasm","build:worker":"./scripts/build_worker.sh","build:glue":"node ./cpp/generate_glue_prototype.js","build:wasm":"./scripts/build_wasm.sh && npm run build:glue","build:tsup":"tsup src/index.ts --format cjs,esm --clean","build:minified":"terser esm/index.js -o esm/index.min.js --compress --mangle --source-map","build:typedef":"tsc --emitDeclarationOnly --declaration -p tsconfig.build.json","build":"npm run clean && npm run build:worker && npm run build:tsup && npm run build:minified && npm run build:typedef","postbuild":"./scripts/post_build.sh && npm run docs","docs":"typedoc --tsconfig tsconfig.build.json src/index.ts","upload":"npm run format && npm run build && npm publish --access public","format":"prettier --write .","test":"vitest","test:firefox":"BROWSER=firefox vitest","test:safari":"BROWSER=safari vitest","test:docker":"./scripts/test_in_docker.sh","test:wasm":"./scripts/test_wasm.sh"},"repository":{"type":"git","url":"git+https://git.justai.co/JustAI/brio-wllama.git"},"keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"author":{"name":"Antoine VIALLON","email":"antoine.viallon@justai.co"},"license":"MIT","bugs":{"url":"https://git.justai.co/JustAI/brio-wllama/issues"},"homepage":"https://git.justai.co/JustAI/brio-wllama#readme","devDependencies":{"@vitest/browser":"^2.1.6","express":"^4.18.3","mime-types":"^2.1.35","playwright":"^1.49.0","prettier":"^3.3.3","terser":"^5.39.0","tsup":"^8.4.0","typedoc":"^0.27.2","typescript":"^5.4.2","webdriverio":"^9.4.1"},"prettier":{"trailingComma":"es5","tabWidth":2,"semi":true,"singleQuote":true,"bracketSameLine":false},"_id":"@aviallon/wllama@2.3.7-e-slow","gitHead":"59c30f3fb799cb4c673d845f4310a3d50d9092f8","_nodeVersion":"22.21.1","_npmVersion":"10.9.4","dist":{"integrity":"sha512-uYSA6UjWf9eRSyNT0gbgFJAU2yRfalABOFYRe5uUxVBd0kHK94uuphLwdFdV83QEkTmLz9ZTUEZ0lIHC5Xw8xQ==","shasum":"fe7bc7fb3dfd27480ac0d2fe01e3f774c5964cf0","tarball":"https://registry.npmjs.org/@aviallon/wllama/-/wllama-2.3.7-e-slow.tgz","fileCount":89,"unpackedSize":11326838,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIGFyB2BNAZiR+Z2owqkoJN9JsAdla0S86sGpD/pV9VnCAiEAnkCE8Jl4EpYvXofokVcJT2LsdHgnGOV8yZt+uGZMMOw="}]},"_npmUser":{"name":"aviallon","email":"antoine-npm@lesviallon.fr"},"maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/wllama_2.3.7-e-slow_1768494516451_0.06920653990183534"},"_hasShrinkwrap":false}},"time":{"created":"2026-01-14T16:36:21.624Z","modified":"2026-01-15T16:28:36.855Z","2.3.7":"2026-01-14T16:36:22.041Z","2.3.7-a":"2026-01-14T16:40:18.620Z","2.3.7-b":"2026-01-14T17:19:02.157Z","2.3.7-c":"2026-01-14T19:12:13.525Z","2.3.7-d":"2026-01-14T21:30:01.475Z","2.3.7-e":"2026-01-15T12:29:11.350Z","2.3.7-e-slow":"2026-01-15T16:28:36.750Z"},"bugs":{"url":"https://git.justai.co/JustAI/brio-wllama/issues"},"author":{"name":"Antoine VIALLON","email":"antoine.viallon@justai.co"},"license":"MIT","homepage":"https://git.justai.co/JustAI/brio-wllama#readme","keywords":["wasm","webassembly","llama","llm","ai","rag","embeddings","generation"],"repository":{"type":"git","url":"git+https://git.justai.co/JustAI/brio-wllama.git"},"description":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference (custom version from aviallon)","maintainers":[{"name":"aviallon","email":"antoine-npm@lesviallon.fr"}],"readme":"# wllama - Wasm binding for llama.cpp\n\n![](./README_banner.png)\n\nWebAssembly binding for [llama.cpp](https://github.com/ggerganov/llama.cpp)\n\n👉 [Try the demo app](https://huggingface.co/spaces/ngxson/wllama)\n\n📄 [Documentation](https://github.ngxson.com/wllama/docs/)\n\nFor changelog, please visit [releases page](https://github.com/ngxson/wllama/releases)\n\n> [!IMPORTANT]  \n> Version 2.0 is released 👉 [read more](./guides/intro-v2.md)\n\n![](./assets/screenshot_0.png)\n\n## Features\n\n- Typescript support\n- Can run inference directly on browser (using [WebAssembly SIMD](https://emscripten.org/docs/porting/simd.html)), no backend or GPU is needed!\n- No runtime dependency (see [package.json](./package.json))\n- High-level API: completions, embeddings\n- Low-level API: (de)tokenize, KV cache control, sampling control,...\n- Ability to split the model into smaller files and load them in parallel (same as `split` and `cat`)\n- Auto switch between single-thread and multi-thread build based on browser support\n- Inference is done inside a worker, does not block UI render\n- Pre-built npm package [@wllama/wllama](https://www.npmjs.com/package/@wllama/wllama)\n\nLimitations:\n- To enable multi-thread, you must add `Cross-Origin-Embedder-Policy` and `Cross-Origin-Opener-Policy` headers. See [this discussion](https://github.com/ffmpegwasm/ffmpeg.wasm/issues/106#issuecomment-913450724) for more details.\n- No WebGPU support, but maybe possible in the future\n- Max file size is 2GB, due to [size restriction of ArrayBuffer](https://stackoverflow.com/questions/17823225/do-arraybuffers-have-a-maximum-length). If your model is bigger than 2GB, please follow the **Split model** section below.\n\n## Code demo and documentation\n\n📄 [Documentation](https://github.ngxson.com/wllama/docs/)\n\nDemo:\n- Basic usages with completions and embeddings: https://github.ngxson.com/wllama/examples/basic/\n- Embedding and cosine distance: https://github.ngxson.com/wllama/examples/embeddings/\n- For more advanced example using low-level API, have a look at test file: [wllama.test.ts](./src/wllama.test.ts)\n\n## How to use\n\n### Use Wllama inside React Typescript project\n\nInstall it:\n\n```bash\nnpm i @wllama/wllama\n```\n\nThen, import the module:\n\n```ts\nimport { Wllama } from '@wllama/wllama';\nlet wllamaInstance = new Wllama(WLLAMA_CONFIG_PATHS, ...);\n// (the rest is the same with earlier example)\n```\n\nFor complete code example, see [examples/main/src/utils/wllama.context.tsx](./examples/main/src/utils/wllama.context.tsx)\n\nNOTE: this example only covers completions usage. For embeddings, please see [examples/embeddings/index.html](./examples/embeddings/index.html)\n\n### Prepare your model\n\n- It is recommended to split the model into **chunks of maximum 512MB**. This will result in slightly faster download speed (because multiple splits can be downloaded in parallel), and also prevent some out-of-memory issues.  \n  See the \"Split model\" section below for more details.\n- It is recommended to use quantized Q4, Q5 or Q6 for balance among performance, file size and quality. Using IQ (with imatrix) is **not** recommended, may result in slow inference and low quality.\n\n### Simple usage with ES6 module\n\nFor complete code, see [examples/basic/index.html](./examples/basic/index.html)\n\n```javascript\nimport { Wllama } from './esm/index.js';\n\n(async () => {\n  const CONFIG_PATHS = {\n    'single-thread/wllama.wasm': './esm/single-thread/wllama.wasm',\n    'multi-thread/wllama.wasm' : './esm/multi-thread/wllama.wasm',\n  };\n  // Automatically switch between single-thread and multi-thread version based on browser support\n  // If you want to enforce single-thread, add { \"n_threads\": 1 } to LoadModelConfig\n  const wllama = new Wllama(CONFIG_PATHS);\n  // Define a function for tracking the model download progress\n  const progressCallback =  ({ loaded, total }) => {\n    // Calculate the progress as a percentage\n    const progressPercentage = Math.round((loaded / total) * 100);\n    // Log the progress in a user-friendly format\n    console.log(`Downloading... ${progressPercentage}%`);\n  };\n  // Load GGUF from Hugging Face hub\n  // (alternatively, you can use loadModelFromUrl if the model is not from HF hub)\n  await wllama.loadModelFromHF(\n    'ggml-org/models',\n    'tinyllamas/stories260K.gguf',\n    {\n      progressCallback,\n    }\n  );\n  const outputText = await wllama.createCompletion(elemInput.value, {\n    nPredict: 50,\n    sampling: {\n      temp: 0.5,\n      top_k: 40,\n      top_p: 0.9,\n    },\n  });\n  console.log(outputText);\n})();\n```\n\nAlternatively, you can use the `*.wasm` files from CDN:\n\n```js\nimport WasmFromCDN from '@wllama/wllama/esm/wasm-from-cdn.js';\nconst wllama = new Wllama(WasmFromCDN);\n// NOTE: this is not recommended, only use when you can't embed wasm files in your project\n```\n\n### Split model\n\nCases where we want to split the model:\n- Due to [size restriction of ArrayBuffer](https://stackoverflow.com/questions/17823225/do-arraybuffers-have-a-maximum-length), the size limitation of a file is 2GB. If your model is bigger than 2GB, you can split the model into small files.\n- Even with a small model, splitting into chunks allows the browser to download multiple chunks in parallel, thus making the download process a bit faster.\n\nWe use `llama-gguf-split` to split a big gguf file into smaller files. You can download the pre-built binary via [llama.cpp release page](https://github.com/ggerganov/llama.cpp/releases):\n\n```bash\n# Split the model into chunks of 512 Megabytes\n./llama-gguf-split --split-max-size 512M ./my_model.gguf ./my_model\n```\n\nThis will output files ending with `-00001-of-00003.gguf`, `-00002-of-00003.gguf`, and so on.\n\nYou can then pass to `loadModelFromUrl` or `loadModelFromHF` the URL of the first file and it will automatically load all the chunks:\n\n```js\nconst wllama = new Wllama(CONFIG_PATHS, {\n  parallelDownloads: 5, // optional: maximum files to download in parallel (default: 3)\n});\nawait wllama.loadModelFromHF(\n  'ngxson/tinyllama_split_test',\n  'stories15M-q8_0-00001-of-00003.gguf'\n);\n```\n\n### Custom logger (suppress debug messages)\n\nWhen initializing Wllama, you can pass a custom logger to Wllama.\n\nExample 1: Suppress debug message\n\n```js\nimport { Wllama, LoggerWithoutDebug } from '@wllama/wllama';\n\nconst wllama = new Wllama(pathConfig, {\n  // LoggerWithoutDebug is predefined inside wllama\n  logger: LoggerWithoutDebug,\n});\n```\n\nExample 2: Add emoji prefix to log messages\n\n```js\nconst wllama = new Wllama(pathConfig, {\n  logger: {\n    debug: (...args) => console.debug('🔧', ...args),\n    log: (...args) => console.log('ℹ️', ...args),\n    warn: (...args) => console.warn('⚠️', ...args),\n    error: (...args) => console.error('☠️', ...args),\n  },\n});\n```\n\n## How to compile the binary yourself\n\nThis repository already come with pre-built binary from llama.cpp source code. However, in some cases you may want to compile it yourself:\n- You don't trust the pre-built one.\n- You want to try out latest - bleeding-edge changes from upstream llama.cpp source code.\n\nYou can use the commands below to compile it yourself:\n\n```shell\n# /!\\ IMPORTANT: Require having docker compose installed\n\n# Clone the repository with submodule\ngit clone --recurse-submodules https://github.com/ngxson/wllama.git\ncd wllama\n\n# Optionally, you can run this command to update llama.cpp to latest upstream version (bleeding-edge, use with your own risk!)\n# git submodule update --remote --merge\n\n# Install the required modules\nnpm i\n\n# Firstly, build llama.cpp into wasm\nnpm run build:wasm\n# Then, build ES module\nnpm run build\n```\n\n## TODO\n\n- Add support for LoRA adapter\n- Support GPU inference via WebGL\n- Support multi-sequences: knowing the resource limitation when using WASM, I don't think having multi-sequences is a good idea\n- Multi-modal: Waiting for refactoring LLaVA implementation from llama.cpp\n","readmeFilename":"README.md"}