{"_id":"@bitkaio/langchain-kserve","_rev":"4-8e1a208b3b5dfcc75a1eaf74fa07f225","name":"@bitkaio/langchain-kserve","dist-tags":{"latest":"0.2.1"},"versions":{"0.1.0":{"name":"@bitkaio/langchain-kserve","version":"0.1.0","keywords":["langchain","kserve","llm","ai","machine-learning","kubernetes","vllm","openai-compatible"],"author":{"name":"bitkaio LLC"},"license":"MIT","_id":"@bitkaio/langchain-kserve@0.1.0","maintainers":[{"name":"bitkaio_team","email":"team@bitkaio.com"}],"homepage":"https://gitlab.com/bitkaio/langchain/kserve-provider#readme","bugs":{"url":"https://gitlab.com/bitkaio/langchain/kserve-provider/issues"},"dist":{"shasum":"dcafcce91eb757fbfed3fa704f51d2b18dc60235","tarball":"https://registry.npmjs.org/@bitkaio/langchain-kserve/-/langchain-kserve-0.1.0.tgz","fileCount":17,"integrity":"sha512-tvOAiFngpCXBxn16DDSOlF4+F7DY+gXTej6j6cD94VJluGGRYSx8H2heDHp4Ib5aFMCPwW6V98NtzHQ/ZEnr3A==","signatures":[{"sig":"MEUCIQDYvUAu4wCE5QJ0QV9k0xV/fWCxkHgZDejG5q6q962XtwIgNF1nFDZe3VqxRyyas/FQWQjT+tMR+rY3ShtwqmtY7mY=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":404871},"main":"./dist/index.cjs","_from":"file:bitkaio-langchain-kserve-0.1.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=18.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"scripts":{"lint":"eslint src/ tests/","test":"vitest run","build":"tsup","typecheck":"tsc --noEmit","test:watch":"vitest","build:watch":"tsup --watch","test:integration":"VITEST_INTEGRATION=1 vitest run tests/integration"},"_npmUser":{"name":"bitkaio_team","email":"team@bitkaio.com"},"_resolved":"/private/var/folders/r5/xf07fdqs66nb8vzm7vgtrkqm0000gn/T/68e9892ade8d2a31be1a7f654e11e559/bitkaio-langchain-kserve-0.1.0.tgz","_integrity":"sha512-tvOAiFngpCXBxn16DDSOlF4+F7DY+gXTej6j6cD94VJluGGRYSx8H2heDHp4Ib5aFMCPwW6V98NtzHQ/ZEnr3A==","repository":{"url":"git+https://gitlab.com/bitkaio/langchain/kserve-provider.git","type":"git"},"_npmVersion":"10.9.3","description":"LangChain.js integration for KServe inference services","directories":{},"_nodeVersion":"22.19.0","dependencies":{"uuid":"^9.0.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","vitest":"^1.6.0","typescript":"^5.4.0","@types/node":"^20.0.0","@types/uuid":"^9.0.0","@langchain/core":">=0.3.0"},"peerDependencies":{"@langchain/core":">=0.3.0"},"_npmOperationalInternal":{"tmp":"tmp/langchain-kserve_0.1.0_1772658406333_0.09387124145977621","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@bitkaio/langchain-kserve","version":"0.1.1","keywords":["langchain","kserve","llm","ai","machine-learning","kubernetes","vllm","openai-compatible"],"author":{"name":"bitkaio LLC"},"license":"MIT","_id":"@bitkaio/langchain-kserve@0.1.1","maintainers":[{"name":"bitkaio_team","email":"team@bitkaio.com"}],"homepage":"https://gitlab.com/bitkaio/langchain/kserve-provider#readme","bugs":{"url":"https://gitlab.com/bitkaio/langchain/kserve-provider/issues"},"dist":{"shasum":"e5bd5ebde26ad2fa0aaa4739d450ea9cdea1ed50","tarball":"https://registry.npmjs.org/@bitkaio/langchain-kserve/-/langchain-kserve-0.1.1.tgz","fileCount":17,"integrity":"sha512-f66F9e9P6PaC+GGXaPtA95lNr/lz43VEXJbvKwQ43Yr0uX9odpw7TnZtKVzipd2OXRzCUfk/bGEJMI8BekTzOg==","signatures":[{"sig":"MEUCIHUcw6JmYOIjOxxOIWCQKTXee6txXbLhe4y9cG/oNZ9IAiEAluny58IKQE5u+L9ePqn8HOjGZuw0+xvv2uIKXrTal6w=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bitkaio%2flangchain-kserve@0.1.1","provenance":{"predicateType":"https://slsa.dev/provenance/v0.2"}},"unpackedSize":405004},"main":"./dist/index.cjs","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.14.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"gitHead":"b363da79fb21069abfd82f8497f681470ca2c846","scripts":{"lint":"eslint src/ tests/","test":"vitest run","build":"tsup","typecheck":"tsc --noEmit","test:watch":"vitest","build:watch":"tsup --watch","prepublishOnly":"pnpm run build","test:integration":"VITEST_INTEGRATION=1 vitest run tests/integration"},"_npmUser":{"name":"GitLab CI/CD","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"gitlab","oidcConfigId":"oidc:02185c2a-1a1b-4d80-b461-2b1ed32d7f9e"}},"repository":{"url":"git+https://gitlab.com/bitkaio/langchain/kserve-provider.git","type":"git"},"_npmVersion":"11.11.0","description":"LangChain.js integration for KServe inference services","directories":{},"_nodeVersion":"22.22.0","dependencies":{"uuid":"^9.0.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","vitest":"^1.6.0","typescript":"^5.4.0","@types/node":"^20.0.0","@types/uuid":"^9.0.0","@langchain/core":">=0.3.0"},"peerDependencies":{"@langchain/core":">=0.3.0"},"_npmOperationalInternal":{"tmp":"tmp/langchain-kserve_0.1.1_1772665374196_0.9126438960175729","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@bitkaio/langchain-kserve","version":"0.2.0","keywords":["langchain","kserve","llm","ai","machine-learning","kubernetes","vllm","openai-compatible"],"author":{"name":"bitkaio LLC"},"license":"MIT","_id":"@bitkaio/langchain-kserve@0.2.0","maintainers":[{"name":"bitkaio_team","email":"team@bitkaio.com"}],"homepage":"https://gitlab.com/bitkaio/langchain/kserve-provider#readme","bugs":{"url":"https://gitlab.com/bitkaio/langchain/kserve-provider/issues"},"dist":{"shasum":"10dd81968fa6de1ef96d41fbb65de13e37cfe2d8","tarball":"https://registry.npmjs.org/@bitkaio/langchain-kserve/-/langchain-kserve-0.2.0.tgz","fileCount":18,"integrity":"sha512-/YWOjj+PXd1MOq6dObHpBevZFHC9oQbQ57K/nc8JieSaFwLL3iQamXeZVXVH/Y57BK/N6AGlJYLcJ06R7FNxDA==","signatures":[{"sig":"MEUCIQDx6gSxmjECDakPkGo6qu/r3PVUbQB2AhLC1CMfpAHAqgIgUk8pZItVafUkJSI/2kn0UsywzOH2/uUhEzLGVyTH3nU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bitkaio%2flangchain-kserve@0.2.0","provenance":{"predicateType":"https://slsa.dev/provenance/v0.2"}},"unpackedSize":568356},"main":"./dist/index.cjs","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.14.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"gitHead":"22e665717a84e80f1b4fc39e114179e3c265c3d5","scripts":{"lint":"eslint src/ tests/","test":"vitest run","build":"tsup","typecheck":"tsc --noEmit","test:watch":"vitest","build:watch":"tsup --watch","prepublishOnly":"pnpm run build","test:integration":"VITEST_INTEGRATION=1 vitest run tests/integration"},"_npmUser":{"name":"GitLab CI/CD","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"gitlab","oidcConfigId":"oidc:02185c2a-1a1b-4d80-b461-2b1ed32d7f9e"}},"repository":{"url":"git+https://gitlab.com/bitkaio/langchain/kserve-provider.git","type":"git"},"_npmVersion":"11.11.0","description":"LangChain.js integration for KServe inference services","directories":{},"_nodeVersion":"22.22.1","dependencies":{"zod":"^3.22.0","uuid":"^9.0.0","zod-to-json-schema":"^3.22.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","vitest":"^1.6.0","typescript":"^5.4.0","@types/node":"^20.0.0","@types/uuid":"^9.0.0","@langchain/core":">=0.3.0"},"peerDependencies":{"@langchain/core":">=0.3.0"},"_npmOperationalInternal":{"tmp":"tmp/langchain-kserve_0.2.0_1772901698170_0.826408370783287","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@bitkaio/langchain-kserve","version":"0.2.1","description":"LangChain.js integration for KServe inference services","author":{"name":"bitkaio LLC"},"license":"MIT","main":"./dist/index.cjs","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"import":"./dist/index.js","require":"./dist/index.cjs","types":"./dist/index.d.ts"}},"scripts":{"build":"tsup","build:watch":"tsup --watch","lint":"eslint src/ tests/","test":"vitest run","test:watch":"vitest","test:integration":"VITEST_INTEGRATION=1 vitest run tests/integration","typecheck":"tsc --noEmit","prepublishOnly":"pnpm run build"},"dependencies":{"uuid":"^9.0.0","zod":"^3.22.0","zod-to-json-schema":"^3.22.0"},"devDependencies":{"@langchain/core":">=1.0.0","@types/node":"^20.0.0","@types/uuid":"^9.0.0","tsup":"^8.0.0","typescript":"^5.4.0","vitest":"^1.6.0"},"peerDependencies":{"@langchain/core":">=1.0.0"},"engines":{"node":">=22.14.0"},"keywords":["langchain","kserve","llm","ai","machine-learning","kubernetes","vllm","openai-compatible"],"repository":{"type":"git","url":"git+https://gitlab.com/bitkaio/langchain/kserve-provider.git"},"gitHead":"5f3999e657769540cb63d57638d29d8b0c1569a9","_id":"@bitkaio/langchain-kserve@0.2.1","bugs":{"url":"https://gitlab.com/bitkaio/langchain/kserve-provider/issues"},"homepage":"https://gitlab.com/bitkaio/langchain/kserve-provider#readme","_nodeVersion":"22.22.1","_npmVersion":"11.11.0","dist":{"integrity":"sha512-6LoZgcd8DS0NhVwbDXk0Fw28NDfSWZDGzwxTmvGfrsh8ctd3ilamo6/S8Q3OOdjaMKQB7T9F9/btKLW4BpCcfA==","shasum":"8bbbd89ff941110e8100b87cfc4f05c0f380c434","tarball":"https://registry.npmjs.org/@bitkaio/langchain-kserve/-/langchain-kserve-0.2.1.tgz","fileCount":18,"unpackedSize":568356,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bitkaio%2flangchain-kserve@0.2.1","provenance":{"predicateType":"https://slsa.dev/provenance/v0.2"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQCZO/jCJ1ddz5AZBFnzhQsSibrHQDT0+nlrTP9DJ9qOQwIgRkpOavnWLhGWh7tqARFtixWZs0LHT0RUHQ5yJ0cFfaA="}]},"_npmUser":{"name":"GitLab CI/CD","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"gitlab","oidcConfigId":"oidc:02185c2a-1a1b-4d80-b461-2b1ed32d7f9e"}},"directories":{},"maintainers":[{"name":"bitkaio_team","email":"team@bitkaio.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/langchain-kserve_0.2.1_1772919077140_0.5676636507728869"},"_hasShrinkwrap":false}},"time":{"created":"2026-03-04T21:06:46.162Z","modified":"2026-03-07T21:31:17.633Z","0.1.0":"2026-03-04T21:06:46.544Z","0.1.1":"2026-03-04T23:02:54.342Z","0.2.0":"2026-03-07T16:41:38.310Z","0.2.1":"2026-03-07T21:31:17.301Z"},"bugs":{"url":"https://gitlab.com/bitkaio/langchain/kserve-provider/issues"},"author":{"name":"bitkaio LLC"},"license":"MIT","homepage":"https://gitlab.com/bitkaio/langchain/kserve-provider#readme","keywords":["langchain","kserve","llm","ai","machine-learning","kubernetes","vllm","openai-compatible"],"repository":{"type":"git","url":"git+https://gitlab.com/bitkaio/langchain/kserve-provider.git"},"description":"LangChain.js integration for KServe inference services","maintainers":[{"name":"bitkaio_team","email":"team@bitkaio.com"}],"readme":"# @bitkaio/langchain-kserve\n\nLangChain.js integration for [KServe](https://kserve.github.io/website/) inference services.\n\nConnect LangChain chains and agents to any model hosted on KServe — whether it's a vLLM-served Qwen2.5-Coder behind an Istio ingress, a Triton-served custom model, or a TGI-served Llama — with full streaming, tool calling, vision, and production-grade connection handling.\n\n## Features\n\n- **Two model classes**: `ChatKServe` for chat/instruct models, `KServeLLM` for base completion models\n- **Dual protocol support**: OpenAI-compatible API (`/v1/chat/completions`) and V2 Inference Protocol (`/v2/models/{model}/infer`), with auto-detection\n- **Full streaming**: SSE for OpenAI-compat, NDJSON for V2\n- **Tool calling**: Full OpenAI function-calling format with `tool_choice`, `parallel_tool_calls`, and proper `invalid_tool_calls` handling for malformed responses\n- **Vision / multimodal**: Send images alongside text via OpenAI content blocks\n- **Token usage tracking**: `llmOutput` and `generationInfo` populated from vLLM responses, including streaming\n- **Logprobs**: Optional per-token log-probabilities in `generationInfo`\n- **Finish reason**: Always propagated in `generationInfo` and `response_metadata`\n- **Model introspection**: `getModelInfo()` returns unified `KServeModelInfo` for both protocols\n- **Production-ready**: TLS/CA bundles, bearer token auth, async token providers (K8s service account tokens), exponential backoff with jitter, 120s default timeout for cold starts\n\n## Installation\n\n```bash\nnpm install @bitkaio/langchain-kserve @langchain/core\n# or\npnpm add @bitkaio/langchain-kserve @langchain/core\n```\n\n**Requirements**: Node.js 22.14+\n\n## Quick Start\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://qwen-coder.my-cluster.example.com\",\n  modelName: \"qwen2.5-coder-32b-instruct\",\n  temperature: 0.2,\n});\n\nconst response = await llm.invoke(\"Write a TypeScript binary search function.\");\nconsole.log(response.content);\n```\n\n## Usage\n\n### Basic invocation\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\nimport { HumanMessage, SystemMessage } from \"@langchain/core/messages\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://qwen-coder.my-cluster.example.com\",\n  modelName: \"qwen2.5-coder-32b-instruct\",\n});\n\n// Single string (shorthand)\nconst result = await llm.invoke(\"Explain KServe in one sentence.\");\n\n// With messages array\nconst result2 = await llm.invoke([\n  new SystemMessage(\"You are an expert TypeScript developer.\"),\n  new HumanMessage(\"What is a monad?\"),\n]);\n```\n\n### Streaming\n\n```typescript\nconst stream = await llm.stream(\"Implement a red-black tree in TypeScript.\");\n\nfor await (const chunk of stream) {\n  process.stdout.write(chunk.content as string);\n}\n```\n\n### In a chain\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\nimport { ChatPromptTemplate } from \"@langchain/core/prompts\";\nimport { StringOutputParser } from \"@langchain/core/output_parsers\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://qwen-coder.my-cluster.example.com\",\n  modelName: \"qwen2.5-coder-32b-instruct\",\n  temperature: 0.2,\n  streaming: true,\n});\n\nconst prompt = ChatPromptTemplate.fromMessages([\n  [\"system\", \"You are an expert {language} developer.\"],\n  [\"human\", \"{input}\"],\n]);\n\nconst chain = prompt.pipe(llm).pipe(new StringOutputParser());\n\nconst response = await chain.invoke({\n  language: \"TypeScript\",\n  input: \"Implement a LRU cache\",\n});\n```\n\n### Token usage tracking\n\nToken usage is available in both `llmOutput` (request-level) and `generationInfo` (per-generation):\n\n```typescript\nconst result = await llm._generate([new HumanMessage(\"Hello\")], {});\n\n// Request-level\nconsole.log(result.llmOutput);\n// { tokenUsage: { promptTokens: 5, completionTokens: 12, totalTokens: 17 } }\n\n// Per-generation\nconst info = result.generations[0].generationInfo;\nconsole.log(info?.tokenUsage);    // { promptTokens: 5, completionTokens: 12, totalTokens: 17 }\nconsole.log(info?.finishReason);  // \"stop\"\n```\n\n### Logprobs\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://qwen-coder.my-cluster.example.com\",\n  modelName: \"qwen2.5-coder-32b-instruct\",\n  protocol: \"openai\",\n  logprobs: true,\n  topLogprobs: 5,\n});\n\nconst result = await llm.invoke(\"Hello\");\nconst info = result.response_metadata;\nconsole.log(info.logprobs); // { content: [{ token: \"Hi\", logprob: -0.3, top_logprobs: [...] }] }\n```\n\n### Tool calling\n\nTool calling requires the OpenAI-compatible protocol (vLLM, TGI, etc.).\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\nimport { tool } from \"@langchain/core/tools\";\nimport { z } from \"zod\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://qwen-coder.my-cluster.example.com\",\n  modelName: \"qwen2.5-coder-32b-instruct\",\n  protocol: \"openai\",\n  parallelToolCalls: true,\n});\n\nconst searchTool = tool(\n  async ({ query }) => `Results for: ${query}`,\n  {\n    name: \"search_codebase\",\n    description: \"Search the codebase for relevant code\",\n    schema: z.object({ query: z.string() }),\n  }\n);\n\nconst llmWithTools = llm.bindTools([searchTool], { toolChoice: \"auto\" });\n\nconst result = await llmWithTools.invoke(\n  \"Find all usages of the AuthService class\"\n);\n\nif (result.tool_calls && result.tool_calls.length > 0) {\n  console.log(\"Tool call:\", result.tool_calls[0]);\n}\n\n// Malformed arguments from the model are captured in invalid_tool_calls\nif (result.invalid_tool_calls && result.invalid_tool_calls.length > 0) {\n  console.log(\"Invalid call:\", result.invalid_tool_calls[0]);\n}\n```\n\n### Vision / multimodal\n\nSend images alongside text. Works with OpenAI-compatible runtimes that support vision (e.g., vLLM with a multimodal model).\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\nimport { HumanMessage } from \"@langchain/core/messages\";\nimport { readFile } from \"node:fs/promises\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://llava.my-cluster.example.com\",\n  modelName: \"llava-1.6\",\n  protocol: \"openai\",\n});\n\n// Base64-encoded image (preferred for cluster-internal use)\nconst imageData = await readFile(\"chart.png\", { encoding: \"base64\" });\nconst message = new HumanMessage({\n  content: [\n    { type: \"text\", text: \"Describe what you see in this chart:\" },\n    {\n      type: \"image_url\",\n      image_url: {\n        url: `data:image/png;base64,${imageData}`,\n        detail: \"high\",\n      },\n    },\n  ],\n});\n\nconst response = await llm.invoke([message]);\nconsole.log(response.content);\n```\n\nURL-based images are also supported (the model pod fetches the image):\n\n```typescript\nconst message = new HumanMessage({\n  content: [\n    { type: \"text\", text: \"What's in this image?\" },\n    { type: \"image_url\", image_url: { url: \"https://example.com/image.jpg\" } },\n  ],\n});\n```\n\n### JSON mode / Response format\n\n```typescript\n// Force valid JSON output\nconst llm = new ChatKServe({\n  baseUrl: \"...\",\n  modelName: \"...\",\n  protocol: \"openai\",\n  responseFormat: { type: \"json_object\" },\n});\n\n// Force output matching a specific JSON schema (vLLM grammar-constrained decoding)\nconst llm = new ChatKServe({\n  baseUrl: \"...\",\n  modelName: \"...\",\n  protocol: \"openai\",\n  responseFormat: {\n    type: \"json_schema\",\n    json_schema: {\n      name: \"person\",\n      strict: true,\n      schema: {\n        type: \"object\",\n        properties: { name: { type: \"string\" }, age: { type: \"integer\" } },\n        required: [\"name\", \"age\"],\n      },\n    },\n  },\n});\n```\n\n### Structured output\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\nimport { z } from \"zod\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"...\",\n  modelName: \"...\",\n  protocol: \"openai\",\n});\n\nconst Person = z.object({ name: z.string(), age: z.number() });\n\n// functionCalling (default) — most reliable\nconst structured = llm.withStructuredOutput(Person);\nconst result = await structured.invoke(\"Extract: John is 30 years old.\");\nconsole.log(result.name, result.age); // \"John\", 30\n\n// jsonSchema — vLLM grammar-constrained decoding\nconst structured2 = llm.withStructuredOutput(Person, { method: \"jsonSchema\" });\n\n// jsonMode — json_object with schema instruction\nconst structured3 = llm.withStructuredOutput(Person, { method: \"jsonMode\" });\n\n// includeRaw — get both raw AIMessage and parsed output\nconst structured4 = llm.withStructuredOutput(Person, { includeRaw: true });\nconst result4 = await structured4.invoke(\"Extract: John is 30.\");\nconsole.log(result4.parsed);       // { name: \"John\", age: 30 }\nconsole.log(result4.raw);          // AIMessage(...)\nconsole.log(result4.parsingError); // null\n```\n\n### Embeddings\n\n```typescript\nimport { KServeEmbeddings } from \"@bitkaio/langchain-kserve\";\n\nconst embeddings = new KServeEmbeddings({\n  baseUrl: \"https://my-embedding-model.cluster.example.com\",\n  modelName: \"Qwen/Qwen3-Embedding-0.6B\",\n});\n\n// Embed documents (batched automatically)\nconst vectors = await embeddings.embedDocuments([\"Hello world\", \"How are you?\"]);\nconsole.log(vectors.length, vectors[0].length); // 2, <embedding_dim>\n\n// Embed a single query\nconst queryVector = await embeddings.embedQuery(\"What is KServe?\");\n\n// With dimensions (Matryoshka models)\nconst embeddingsWithDims = new KServeEmbeddings({\n  baseUrl: \"...\",\n  modelName: \"...\",\n  dimensions: 512,\n  chunkSize: 500, // max texts per API call (default: 1000)\n});\n```\n\n### Model introspection\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://qwen-coder.my-cluster.example.com\",\n  modelName: \"qwen2.5-coder-32b-instruct\",\n});\n\nconst info = await llm.getModelInfo();\n\nconsole.log(info.modelName);    // \"qwen2.5-coder-32b-instruct\"\nconsole.log(info.platform);     // \"openai-compat\" or V2 platform string\nconsole.log(info.raw);          // full response from the endpoint\n```\n\n### V2 Inference Protocol\n\nUse `protocol: \"v2\"` for runtimes that expose the native KServe V2/Open Inference Protocol (e.g., Triton Inference Server):\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://triton.my-cluster.example.com\",\n  modelName: \"llama-3-8b-instruct\",\n  protocol: \"v2\",\n  chatTemplate: \"llama\",\n});\n\nconst result = await llm.invoke(\"Explain transformers in simple terms.\");\n```\n\n> **Note:** Tool calling and vision are not supported on V2. Attempting either throws a `KServeInferenceError` immediately (before any HTTP call) with a clear message.\n\n### Base (non-chat) models with KServeLLM\n\n```typescript\nimport { KServeLLM } from \"@bitkaio/langchain-kserve\";\n\nconst llm = new KServeLLM({\n  baseUrl: \"https://base-model.my-cluster.example.com\",\n  modelName: \"llama-3-base\",\n  protocol: \"v2\",\n  temperature: 0.8,\n  maxTokens: 256,\n});\n\nconst completion = await llm.invoke(\"Once upon a time\");\n\n// Token usage (from streaming or non-streaming)\nconst result = await llm.generate([\"Once upon a time\"]);\nconsole.log(result.llmOutput); // { tokenUsage: {...} } if vLLM, else undefined\n```\n\n### Environment variable configuration\n\nAll constructor options can be set via environment variables:\n\n| Environment Variable | Constructor Field |\n|----------------------|-------------------|\n| `KSERVE_BASE_URL`    | `baseUrl`         |\n| `KSERVE_MODEL_NAME`  | `modelName`       |\n| `KSERVE_API_KEY`     | `apiKey`          |\n| `KSERVE_PROTOCOL`    | `protocol`        |\n| `KSERVE_CA_BUNDLE`   | `caBundle`        |\n\n```bash\nexport KSERVE_BASE_URL=https://qwen-coder.my-cluster.example.com\nexport KSERVE_MODEL_NAME=qwen2.5-coder-32b-instruct\nexport KSERVE_API_KEY=my-bearer-token\n```\n\n### Authentication\n\n**Static API key / bearer token:**\n\n```typescript\nconst llm = new ChatKServe({\n  baseUrl: \"https://my-model.cluster.example.com\",\n  modelName: \"my-model\",\n  apiKey: \"my-bearer-token\",\n});\n```\n\n**Dynamic token provider** (e.g., Kubernetes service account tokens):\n\n```typescript\nimport { readFile } from \"node:fs/promises\";\n\nconst llm = new ChatKServe({\n  baseUrl: \"https://my-model.cluster.example.com\",\n  modelName: \"my-model\",\n  tokenProvider: async () => {\n    const token = await readFile(\n      \"/var/run/secrets/kubernetes.io/serviceaccount/token\",\n      \"utf-8\"\n    );\n    return token.trim();\n  },\n});\n```\n\n### Custom TLS / CA bundle\n\n```typescript\nconst llm = new ChatKServe({\n  baseUrl: \"https://my-model.internal.example.com\",\n  modelName: \"my-model\",\n  caBundle: \"/etc/ssl/certs/my-internal-ca.pem\",\n});\n```\n\n### Connection tuning\n\n```typescript\nconst llm = new ChatKServe({\n  baseUrl: \"https://my-model.cluster.example.com\",\n  modelName: \"my-model\",\n  timeout: 300_000,  // 5 minutes for cold starts\n  maxRetries: 5,\n});\n```\n\n### DeepAgents integration\n\n```typescript\nimport { ChatKServe } from \"@bitkaio/langchain-kserve\";\nimport { tool } from \"@langchain/core/tools\";\nimport { z } from \"zod\";\n\nconst readFile = tool(\n  async ({ path }) => { /* ... */ },\n  {\n    name: \"read_file\",\n    description: \"Read a file from the codebase\",\n    schema: z.object({ path: z.string() }),\n  }\n);\n\nconst llm = new ChatKServe({\n  baseUrl: process.env.KSERVE_BASE_URL!,\n  modelName: \"qwen2.5-coder-32b-instruct\",\n  temperature: 0.2,\n  streaming: true,\n});\n\nimport { Agent } from \"deepagents\";\n\nconst agent = new Agent({ llm, tools: [readFile] });\n\nconst result = await agent.invoke({\n  input: \"Refactor the authentication module to use JWT tokens\",\n});\n```\n\n## API Reference\n\n### `ChatKServe`\n\nExtends `BaseChatModel`. Use for chat/instruct models.\n\n#### Constructor options\n\n| Option | Type | Default | Description |\n|--------|------|---------|-------------|\n| `baseUrl` | `string` | `KSERVE_BASE_URL` env | KServe inference service URL |\n| `modelName` | `string` | `KSERVE_MODEL_NAME` env | Model name as registered in KServe |\n| `protocol` | `\"openai\" \\| \"v2\" \\| \"auto\"` | `\"auto\"` | Inference protocol |\n| `apiKey` | `string` | `KSERVE_API_KEY` env | Static bearer token |\n| `tokenProvider` | `() => Promise<string>` | — | Dynamic token provider |\n| `verifySsl` | `boolean` | `true` | Verify SSL certificates |\n| `caBundle` | `string` | `KSERVE_CA_BUNDLE` env | Path to CA certificate bundle |\n| `temperature` | `number` | — | Sampling temperature |\n| `maxTokens` | `number` | — | Max tokens to generate |\n| `topP` | `number` | — | Top-p nucleus sampling |\n| `stop` | `string[]` | — | Stop sequences |\n| `streaming` | `boolean` | `false` | Enable streaming |\n| `logprobs` | `boolean` | — | Return per-token log-probabilities (OpenAI-compat only) |\n| `topLogprobs` | `number` | — | Number of top logprobs per token (OpenAI-compat only) |\n| `parallelToolCalls` | `boolean` | — | Allow multiple tool calls in one turn (OpenAI-compat only) |\n| `responseFormat` | `OpenAIResponseFormat` | — | Response format constraint (e.g. `{ type: \"json_object\" }`) (OpenAI-compat only) |\n| `timeout` | `number` | `120000` | Request timeout (ms) |\n| `maxRetries` | `number` | `3` | Max retry attempts |\n| `chatTemplate` | `\"chatml\" \\| \"llama\" \\| \"custom\"` | `\"chatml\"` | Template for V2 protocol |\n| `customChatTemplate` | `string` | — | Custom template string |\n\n#### Methods\n\n| Method | Returns | Description |\n|--------|---------|-------------|\n| `invoke(input)` | `Promise<AIMessage>` | Single-turn generation |\n| `stream(input)` | `AsyncIterable<AIMessageChunk>` | Streaming generation |\n| `bindTools(tools, kwargs?)` | `Runnable` | Bind tools for function calling |\n| `withStructuredOutput(schema, config?)` | `Runnable` | Structured output via function calling, JSON schema, or JSON mode |\n| `getModelInfo()` | `Promise<KServeModelInfo>` | Fetch model metadata from endpoint |\n\n### `KServeLLM`\n\nExtends `BaseLLM`. Use for base/completion models. Same options as `ChatKServe`, minus `chatTemplate` and `customChatTemplate`.\n\n### `KServeModelInfo`\n\n```typescript\ninterface KServeModelInfo {\n  modelName: string;\n  modelVersion?: string;\n  platform?: string;\n  inputs?: Array<Record<string, unknown>>;\n  outputs?: Array<Record<string, unknown>>;\n  raw: Record<string, unknown>;\n}\n```\n\n### Error classes\n\n```typescript\nimport {\n  KServeError,                // base class\n  KServeConnectionError,      // network/DNS failures\n  KServeAuthenticationError,  // 401/403\n  KServeModelNotFoundError,   // 404, model not loaded\n  KServeInferenceError,       // 4xx/5xx during inference, or V2 unsupported feature\n  KServeTimeoutError,         // timeout exceeded\n} from \"@bitkaio/langchain-kserve\";\n```\n\n## API Reference\n\n### `KServeEmbeddings`\n\n```typescript\nconst embeddings = new KServeEmbeddings({\n  baseUrl: \"...\",           // or KSERVE_EMBEDDINGS_BASE_URL / KSERVE_BASE_URL env\n  modelName: \"...\",         // or KSERVE_EMBEDDINGS_MODEL_NAME env\n  apiKey: \"...\",            // optional, or KSERVE_API_KEY env\n  dimensions: 512,          // optional, for Matryoshka models\n  encodingFormat: \"float\",  // \"float\" (default) or \"base64\"\n  chunkSize: 1000,          // max texts per API call\n  timeout: 120_000,         // ms\n  maxRetries: 3,\n});\n```\n\n| Method | Returns | Description |\n|--------|---------|-------------|\n| `embedDocuments(docs)` | `Promise<number[][]>` | Embed a list of documents |\n| `embedQuery(doc)` | `Promise<number[]>` | Embed a single query |\n\n## Protocol Capability Matrix\n\n| Feature | OpenAI-compat | V2 |\n|---|:---:|:---:|\n| Text generation | ✅ | ✅ |\n| Streaming | ✅ | ✅ |\n| Tool calling | ✅ | ❌ |\n| Vision / multimodal | ✅ | ❌ |\n| Logprobs | ✅ | ❌ |\n| Token usage tracking | ✅ | ❌ |\n| Finish reason | ✅ | partial |\n| JSON mode / responseFormat | ✅ | ❌ |\n| Structured output | ✅ | ❌ |\n| Embeddings | ✅ | — |\n\nAttempting tool calling, vision, or responseFormat with V2 throws `KServeInferenceError` immediately (before any HTTP call) with a clear message.\n\n## Protocol auto-detection\n\nWhen `protocol: \"auto\"` (the default), the client probes `GET /v1/models`:\n\n- **200 response** → OpenAI-compatible protocol (`/v1/chat/completions`)\n- **Non-200 / connection error** → V2 Inference Protocol (`/v2/models/{model}/infer`)\n\nThe detected protocol is cached for the lifetime of the model instance. Pin a protocol explicitly to skip detection:\n\n```typescript\nconst llm = new ChatKServe({\n  baseUrl: \"...\",\n  modelName: \"...\",\n  protocol: \"openai\", // or \"v2\"\n});\n```\n\n## Development\n\n```bash\n# Install dependencies\npnpm install\n\n# Build\npnpm build\n\n# Run unit tests\npnpm test\n\n# Run with watch mode\npnpm test:watch\n\n# Type check\npnpm typecheck\n\n# Integration tests (requires live KServe endpoint)\nKSERVE_BASE_URL=https://... KSERVE_MODEL_NAME=... pnpm test:integration\n```\n\n## License\n\nMIT\n","readmeFilename":"README.md"}