{"_id":"@altiplano/usellama","_rev":"2-9bde7554c6953f892a0eac75bd05c57e","name":"@altiplano/usellama","dist-tags":{"latest":"0.0.3"},"versions":{"0.0.1":{"name":"@altiplano/usellama","version":"0.0.1","description":"Altiplano Llama.cpp composable with Llama-node","main":"dist/index.js","types":"dist/index.d.ts","scripts":{"build":"tsc -p ."},"dependencies":{"llama-node":"0.1.6","@llama-node/llama-cpp":"^0.1.6"},"devDependencies":{"@altiplano/types":"^0.0.1","@types/node":"^16.18.23","tslib":"^2.5.0","typescript":"^4.9.5"},"type":"module","publishConfig":{"access":"public","registry":"https://registry.npmjs.org/"},"license":"MIT","licenseText":"MIT License\n\nCopyright (c) 2021 Emencia\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n","_id":"@altiplano/usellama@0.0.1","dist":{"shasum":"335fa75171e579676c28605e3d35bed76ef74234","integrity":"sha512-vuTBiMnhkKVoVeEOniRRlPfVdvGp/4sGTZSfkLNzOSUn9Km1PsV2m5jLILN9VCWh2vYxNqL8TJxVHCiPPO/b0Q==","tarball":"https://registry.npmjs.org/@altiplano/usellama/-/usellama-0.0.1.tgz","fileCount":13,"unpackedSize":16893,"signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEQCIFcZrxw5cLIoKvQaT/fxWxz3swlMNdVYNyObOHJxddRTAiB/U/tp+4Cbx/TCrCKyOlq7ermI6utS7hvQXF79jiu3oA=="}]},"_npmUser":{"name":"synw","email":"synwx@pm.me"},"directories":{},"maintainers":[{"name":"synw","email":"synwx@pm.me"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/usellama_0.0.1_1687519346447_0.6205356288684607"},"_hasShrinkwrap":false},"0.0.2":{"name":"@altiplano/usellama","version":"0.0.2","description":"Altiplano Llama.cpp composable with Llama-node","main":"dist/index.js","types":"dist/index.d.ts","scripts":{"build":"tsc -p ."},"dependencies":{"llama-node":"0.1.6","@llama-node/llama-cpp":"^0.1.6"},"devDependencies":{"@altiplano/types":"^0.0.1","@types/node":"^16.18.23","tslib":"^2.5.0","typescript":"^4.9.5"},"type":"module","publishConfig":{"access":"public","registry":"https://registry.npmjs.org/"},"license":"MIT","repository":{"type":"git","url":"https://github.com/synw/altiplano.git"},"licenseText":"MIT License\n\nCopyright (c) 2021 Emencia\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n","_id":"@altiplano/usellama@0.0.2","dist":{"shasum":"0ed5d8a306c3676a3df0752c26f4419e455e8f9d","integrity":"sha512-WiRJGP55YQWCtanLDhhNgrvd4ml+GRSC6/wdgQNZuZ5ImUVguNGgyD9/C5nIBYrxnacL9WHQ2AiTo0wR5WIVDg==","tarball":"https://registry.npmjs.org/@altiplano/usellama/-/usellama-0.0.2.tgz","fileCount":13,"unpackedSize":18519,"signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEYCIQCAUQ+4exb3MIB5J63WINXX6cmNZ0I4KHD6abTn+qokOQIhAILhTZ46FNI/NMHo8ep4xjqbs4d0CtB+ofJwd7U8a/Hb"}]},"_npmUser":{"name":"synw","email":"synwx@pm.me"},"directories":{},"maintainers":[{"name":"synw","email":"synwx@pm.me"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/usellama_0.0.2_1687527760518_0.5932459305880811"},"_hasShrinkwrap":false},"0.0.3":{"name":"@altiplano/usellama","version":"0.0.3","description":"Altiplano Llama.cpp composable with Llama-node","main":"dist/index.js","types":"dist/index.d.ts","scripts":{"build":"tsc -p ."},"dependencies":{"llama-node":"0.1.6","@llama-node/llama-cpp":"^0.1.6"},"devDependencies":{"@altiplano/types":"^0.0.1","@types/node":"^16.18.23","tslib":"^2.5.0","typescript":"^4.9.5"},"type":"module","publishConfig":{"access":"public","registry":"https://registry.npmjs.org/"},"license":"MIT","repository":{"type":"git","url":"https://github.com/synw/altiplano.git"},"licenseText":"MIT License\n\nCopyright (c) 2021 Emencia\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n","_id":"@altiplano/usellama@0.0.3","dist":{"shasum":"3039e630f5427f1038732a705854da89142954ad","integrity":"sha512-avMpj6mSUIzhj6LdhKLFJb+5Z//ZLwENaC7RLNLXR2X1eDhFGzGnNTJN/mvy8UCYrOyhEo2qFTd+gK8C0g7c1Q==","tarball":"https://registry.npmjs.org/@altiplano/usellama/-/usellama-0.0.3.tgz","fileCount":15,"unpackedSize":19030,"signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEUCIFlQ4Jk4kTVXSverCC+NygKo0Hpdly1h5nVnrL8HCrW8AiEA6ZtI/n7BX0GFs48T8fZPiEJvShoakiKp+cb+Le7F4uk="}]},"_npmUser":{"name":"synw","email":"synwx@pm.me"},"directories":{},"maintainers":[{"name":"synw","email":"synwx@pm.me"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/usellama_0.0.3_1687531843623_0.5894034904045069"},"_hasShrinkwrap":false}},"time":{"created":"2023-06-23T11:22:26.353Z","0.0.1":"2023-06-23T11:22:26.665Z","modified":"2023-06-23T14:50:43.928Z","0.0.2":"2023-06-23T13:42:40.661Z","0.0.3":"2023-06-23T14:50:43.818Z"},"maintainers":[{"name":"synw","email":"synwx@pm.me"}],"description":"Altiplano Llama.cpp composable with Llama-node","license":"MIT","readme":"# Use Llama composable\n\nA composable to run [Llama.cpp](https://github.com/ggerganov/llama.cpp) with [Llama-node](https://github.com/Atome-FE/llama-node)\n\n## Install\n\n```bash\nnpm install @altiplano/usellama\n# or\nyarn add @altiplano/usellama\n```\n\n## Example usage\n\n```ts\nimport { useLlama } from \"@altiplano/usellama\";\n\nconst lm = useLlama({verbose: true});\nawait lm.loadModel(modelPath);\nconst template = `### Instruction: Fix this invalid json:\n\n{prompt}\n### Response: (answer in json)`;\nconst result = await lm.infer('{\"a\":1,}', template);\n```\n\n## Api\n\n### Initialization\n\nOptional initialization parameters:\n\n- `onToken`: *(message: any) => void* : a function executed on each token emission\n- `onStartInfer`: *(message: any) => void* : a function executed when the token emission starts\n- `onEndInfer`: *(message: any) => void* : a function executed when the token emission stops\n- `verbose`: *boolean* : output info and the inference text\n\nExample:\n\n```ts\nconst lm = useLlama({\n  onToken: (t) => process.stdout.write(t),\n  onEndInfer: () => doSomething()\n});\n```\n\n### Loading a model\n\nTo load a model use the `loadModel` function and pass it optional parameters:\n\n- `modelPath`: *string* : the absolute path to the model. Not necessary if it was preloaded with `useModel` (see below)\n- `params`: *OptionalModelParams*: the optional parameters\n\nDetail of the `OptionalModelParams`:\n\n- `nCtx`: *number* : context window size (default *2048*)\n- `nGpuLayers`: *number* : number of CPU layers to use (default *0*)\n- `seed`: *number* : (default *0*)\n- `f16Kv`: *boolean* : (default *false*)\n- `logitsAll`: *boolean* : (default *false*)\n- `vocabOnly`: *boolean* : (default *false*)\n- `useMlock`: *boolean* : (default *false*)\n- `embedding`: *boolean* : (default *false*)\n- `useMmap`: *boolean* : (default *true*)\n- `enableLogging`: *boolean* : (default *true*)\n\nExample:\n\n```ts\nawait lm.loadModel(\n  \"/an/absolute/path/open-llama-7B-open-instruct.ggmlv3.q5_1.bin\", \n  { nCtx: 1024 }\n);\n```\n\nTo use a model without actually loading it into the memory:\n\n```ts\nawait lm.useModel(\n  \"/an/absolute/path/open-llama-7B-open-instruct.ggmlv3.q5_1.bin\", \n  { nCtx: 1024 }\n);\n```\n\nIf the model is preloaded like this, no need to use parameters for `loadModel`. Also\nwhen using the `infer` function (see below) if the model is preloaded it will be loaded\nin the memory at the first inference request.\n\nAn `unloadModel` function is also available\n\n### Run inference\n\nTo run inference use the `infer` function with parameters:\n\n- `prompt`: *string* **required**: the prompt text\n- `template`: *string* : (default *{prompt}*): the template to use. A *{prompt}* template variable is available\n- `templateVars`: *Array<TemplateVar>* : extra template variables to use\n\nExample:\n\n```ts\nconst template = `### Instruction: Fix this invalid json:\n\n{prompt}\n### Response: (answer in json)`;\nconst result = await lm.infer('{\"a\":1,}', template);\n```\n\nThe inference result contains extra information:\n\n```js\n{\n  tokens: [ ' {\"', 'a', '\":', ' ', '1', '}', '\\n\\n<end>\\n' ],\n  completed: true,\n  text: ' {\"a\": 1}\\n\\n<end>\\n',\n  thinkingTime: 5.27,\n  inferenceTime: 1.33,\n  totalTime: 6.6,\n  tokensPerSeconds: 1.1\n}\n```\n\n### Inference parameters\n\nIt is possible to tune the inference parameters with the `params` function. Parameters:\n\n- `nThreads`: *number*: number of cpu threads to use (default *4*)\n- `nTokPredict`: *number*: max number of tokens to output (default *4*)\n- `logitBias`: *Array<LogitBias>*: logit bias for specific tokens (default *null*)\n- `topK`: *number*: top k tokens to sample from (default *40*, *1.0* = disabled)\n- `topP`: *number*: top p tokens to sample from (default *0.95*, *1.0* = disabled)\n- `tfsZ`: *number*: tail free sampling (default *1.0* - disabled)\n- `temp`: *number*: temperature (default *0.2*, *1.0* = disabled)\n- `typicalP`: *number*: locally typical sampling (default *1.0* - disabled)\n- `repeatPenalty`: *number*: repeat penalty (default *1.10*, *1.0* = disabled)\n- `repeatLastN`: *number*: last n tokens to penalize (default *64*, 0 = disable penalty, -1 = context size)\n- `frequencyPenalty`: *number*: frequency penalty (default *0*, *1.0* = disabled)\n- `presencePenalty`: *number*: presence penalty (default *0*, *1.0* = disabled)\n- `mirostat`: *number*: Mirostat 1.0 algorithm (default *0*, *0* = disabled)\n- `mirostatTau`: *number*: the target cross-entropy (or surprise) value you want to achieve for the generated text. A higher value corresponds to more surprising or less predictable text, while a lower value corresponds to less surprising or more predictable text (default *0.1*)\n- `mirostatEta`: *number*: the learning rate used to update `mu` based on the error between the target and observed surprisal of the sampled word. A larger learning rate will cause `mu` to be updated more quickly, while a smaller learning rate will result in slower updates (default *0.1*)\n- `stopSequence`: *string* : stop emiting sequence (default *null*)\n- `penalizeNl`: *number*: consider newlines as a repeatable token (default *true*)\n\n### Abort inference\n\nTo abort an inference running use the `abort` function:\n\n```ts\nlm.abort();\n```\n\n### Model info\n\nTo get information about the currently used model a readonly `model` getter is available:\n\n```js\n{\n  name: 'open-llama-7B-open-instruct.ggmlv3.q5_1',\n  path: '/path/to/models/open-llama-7B-open-instruct.ggmlv3.q5_1.bin',\n  isLoaded: true,\n  isInfering: false,\n  config: {\n    modelPath: '/path/to/models/open-llama-7B-open-instruct.ggmlv3.q5_1.bin',\n    nCtx: 1024,\n    enableLogging: true,\n    seed: 0,\n    f16Kv: false,\n    logitsAll: false,\n    vocabOnly: false,\n    useMlock: false,\n    embedding: false,\n    useMmap: true,\n    nGpuLayers: 0\n  },\n  inferenceParams: {\n    prompt: '',\n    nThreads: 4,\n    nTokPredict: 512,\n    logitBias: undefined,\n    topK: undefined,\n    topP: undefined,\n    tfsZ: undefined,\n    temp: 0.2,\n    typicalP: undefined,\n    repeatPenalty: 1,\n    repeatLastN: undefined,\n    frequencyPenalty: undefined,\n    presencePenalty: undefined,\n    mirostat: undefined,\n    mirostatTau: undefined,\n    mirostatEta: undefined,\n    stopSequence: undefined,\n    penalizeNl: undefined\n  }\n}\n```\n\n## Example\n\n```js\n#!/usr/bin/env node\n\nimport { argv, exit } from \"process\";\nimport { useLlama } from \"@altiplano/usellama\";\n\nasync function main(modelPath) {\n  // initialize the lm\n  const lm = useLlama({\n    onToken: (t) => process.stdout.write(t)\n  })\n  // load a model\n  await lm.loadModel(modelPath, { nCtx: 1024 });\n  // set some parameters\n  lm.params({\n    nTokPredict: 512,\n    repeatPenalty: 1,\n  })\n  // run inference\n  const template = \"### Instruction: Fix this invalid json:\\n\\n{prompt}\\n### Response: (answer in json)\"\n  const result = await lm.infer('{\"a\":1,}', template);\n  console.log(result)\n}\n\n\n(async () => {\n  try {\n    if (argv.length < 3) {\n      console.warn(\"Provide a model path as argument\");\n      exit(1);\n    }\n    await main(argv[2]);\n    console.log(\"Finished\");\n    exit(0);\n  }\n  catch (e) {\n    throw e;\n  }\n})();\n```","readmeFilename":"README.md","repository":{"type":"git","url":"https://github.com/synw/altiplano.git"}}