{"_id":"@aos-agent/ai","_rev":"4-1edb3e73d2336db31c811129c3f8b640","name":"@aos-agent/ai","dist-tags":{"latest":"0.85.1"},"versions":{"0.84.2":{"name":"@aos-agent/ai","version":"0.84.2","keywords":["ai","llm","openai","anthropic","gemini","bedrock","unified","api"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@aos-agent/ai@0.84.2","maintainers":[{"name":"aosagent","email":"guanwenpeng2001@gmail.com"}],"homepage":"https://github.com/guanwenpeng2001-bot/AOS-Agent#readme","bugs":{"url":"https://github.com/guanwenpeng2001-bot/AOS-Agent/issues"},"bin":{"aos-agent-ai":"dist/cli.js"},"dist":{"shasum":"803f9368e934989b9dd901894d82e72e020665b9","tarball":"https://registry.npmjs.org/@aos-agent/ai/-/ai-0.84.2.tgz","fileCount":742,"integrity":"sha512-k+re5a1sA75ZJk70NMYvRFOxx/pZVVnYKaDp0jMh0vU7QoI5YtLVbjfhUEKmrzSAeMNWQHS8fC7SJfJFvVkgmQ==","signatures":[{"sig":"MEQCIDFLbznZr5sHI9OsgXXelNBvfeS48hQ05NNACVQpplxpAiB32kmj5xRPOfx4yCTK6K/5rfrVScInrMwWJkWcB1EjLA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":4090493},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.19.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./api/*":{"types":"./dist/api/*.d.ts","import":"./dist/api/*.js"},"./oauth":{"types":"./dist/oauth.d.ts","import":"./dist/oauth.js"},"./compat":{"types":"./dist/compat.d.ts","import":"./dist/compat.js"},"./bun-oauth":{"types":"./dist/bun-oauth.d.ts","import":"./dist/bun-oauth.js"},"./providers/*":{"types":"./dist/providers/*.d.ts","import":"./dist/providers/*.js"},"./bedrock-provider":{"types":"./dist/bedrock-provider.d.ts","import":"./dist/bedrock-provider.js"}},"scripts":{"test":"vitest --run","build":"npm run generate-models && npm run generate-image-models && npm run build:offline","clean":"shx rm -rf dist","build:offline":"npm run generate-aos-model-registry && npm run check:model-data && tsgo -p tsconfig.build.json && shx rm -rf dist/providers/data && shx cp -r src/providers/data dist/providers/data","prepublishOnly":"npm run clean && npm run build","generate-models":"node scripts/generate-models.ts --strict","check:model-data":"node scripts/check-model-data.ts","hydrate-model-data":"node scripts/generate-models.ts --strict --data-only","generate-image-models":"node scripts/generate-image-models.ts --strict","generate-model-catalog":"node scripts/generate-models.ts --strict --json-only --json-output ../../.artifacts/model-catalog","update-aos-model-registry":"npm run generate-models && npm run generate-aos-model-registry","generate-aos-model-registry":"node scripts/generate-aos-model-registry.ts"},"_npmUser":{"name":"aosagent","email":"guanwenpeng2001@gmail.com"},"repository":{"url":"git+https://github.com/guanwenpeng2001-bot/AOS-Agent.git","type":"git","directory":"packages/ai"},"_npmVersion":"12.0.2","description":"Unified LLM API with automatic model discovery and provider configuration","directories":{},"sideEffects":["./dist/compat.js","./dist/images.js","./dist/providers/images/register-builtins.js"],"_nodeVersion":"24.18.1","dependencies":{"openai":"6.40.0","typebox":"1.3.7","partial-json":"0.1.7","@google/genai":"1.52.0","http-proxy-agent":"7.0.2","@anthropic-ai/sdk":"0.91.1","https-proxy-agent":"7.0.6","@opentelemetry/api":"1.9.0","@aos-agent/telemetry":"^0.84.2","@smithy/node-http-handler":"4.7.3","@aws-sdk/client-bedrock-runtime":"3.1048.0"},"_hasShrinkwrap":false,"devDependencies":{"canvas":"3.2.3","vitest":"4.1.9","@types/node":"24.12.4"},"_npmOperationalInternal":{"tmp":"tmp/ai_0.84.2_1786323537387_0.5425606170361383","host":"s3://npm-registry-packages-npm-production"}},"0.84.3":{"name":"@aos-agent/ai","version":"0.84.3","keywords":["ai","llm","openai","anthropic","gemini","bedrock","unified","api"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@aos-agent/ai@0.84.3","maintainers":[{"name":"aosagent","email":"guanwenpeng2001@gmail.com"}],"homepage":"https://github.com/guanwenpeng2001-bot/AOS-Agent#readme","bugs":{"url":"https://github.com/guanwenpeng2001-bot/AOS-Agent/issues"},"bin":{"aos-agent-ai":"dist/cli.js"},"dist":{"shasum":"df7cc013257d3741fdbae65416fdea34e63f7c4c","tarball":"https://registry.npmjs.org/@aos-agent/ai/-/ai-0.84.3.tgz","fileCount":742,"integrity":"sha512-88M76PHRBzkcWpb9wRbK2ksAgsd8LeCxez/btKokxctKmHdxY0eh/MaXaAW3kjhDlAcH4PyjYC25rfBIejgozw==","signatures":[{"sig":"MEUCIClwYAvjzbf6wZC7cK8S16yFsmJpd7e+uFBTdq1UKIqoAiEAyIyJ9GY5Cq5MA1/lHM1eFWrhpFolniMK4Oc+7ArzYHM=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@aos-agent%2fai@0.84.3","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":4088632},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.19.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./api/*":{"types":"./dist/api/*.d.ts","import":"./dist/api/*.js"},"./oauth":{"types":"./dist/oauth.d.ts","import":"./dist/oauth.js"},"./compat":{"types":"./dist/compat.d.ts","import":"./dist/compat.js"},"./bun-oauth":{"types":"./dist/bun-oauth.d.ts","import":"./dist/bun-oauth.js"},"./providers/*":{"types":"./dist/providers/*.d.ts","import":"./dist/providers/*.js"},"./bedrock-provider":{"types":"./dist/bedrock-provider.d.ts","import":"./dist/bedrock-provider.js"}},"gitHead":"d92430128236cc1f285dcbb4f03ddd56ed81f8da","scripts":{"test":"vitest --run","build":"npm run generate-models && npm run generate-image-models && npm run build:offline","clean":"shx rm -rf dist","build:offline":"npm run generate-aos-model-registry && npm run check:model-data && tsgo -p tsconfig.build.json && shx rm -rf dist/providers/data && shx cp -r src/providers/data dist/providers/data","prepublishOnly":"npm run clean && npm run build","generate-models":"node scripts/generate-models.ts --strict","check:model-data":"node scripts/check-model-data.ts","hydrate-model-data":"node scripts/generate-models.ts --strict --data-only","generate-image-models":"node scripts/generate-image-models.ts --strict","generate-model-catalog":"node scripts/generate-models.ts --strict --json-only --json-output ../../.artifacts/model-catalog","update-aos-model-registry":"npm run generate-models && npm run generate-aos-model-registry","generate-aos-model-registry":"node scripts/generate-aos-model-registry.ts"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:3558bf2c-3ba5-4215-8837-cd27022012d4"}},"repository":{"url":"git+https://github.com/guanwenpeng2001-bot/AOS-Agent.git","type":"git","directory":"packages/ai"},"_npmVersion":"11.5.1","description":"Unified LLM API with automatic model discovery and provider configuration","directories":{},"sideEffects":["./dist/compat.js","./dist/images.js","./dist/providers/images/register-builtins.js"],"_nodeVersion":"24.18.0","dependencies":{"openai":"6.40.0","typebox":"1.3.7","partial-json":"0.1.7","@google/genai":"1.52.0","http-proxy-agent":"7.0.2","@anthropic-ai/sdk":"0.91.1","https-proxy-agent":"7.0.6","@opentelemetry/api":"1.9.0","@aos-agent/telemetry":"^0.84.3","@smithy/node-http-handler":"4.7.3","@aws-sdk/client-bedrock-runtime":"3.1048.0"},"_hasShrinkwrap":false,"devDependencies":{"canvas":"3.2.3","vitest":"4.1.9","@types/node":"24.12.4"},"_npmOperationalInternal":{"tmp":"tmp/ai_0.84.3_1786383892902_0.9555718166988281","host":"s3://npm-registry-packages-npm-production"}},"0.85.0":{"name":"@aos-agent/ai","version":"0.85.0","keywords":["ai","llm","openai","anthropic","gemini","bedrock","unified","api"],"author":{"name":"AOS Agent"},"license":"MIT","_id":"@aos-agent/ai@0.85.0","maintainers":[{"name":"aosagent","email":"guanwenpeng2001@gmail.com"}],"homepage":"https://github.com/guanwenpeng2001-bot/AOS-Agent#readme","bugs":{"url":"https://github.com/guanwenpeng2001-bot/AOS-Agent/issues"},"bin":{"aos-agent-ai":"dist/cli.js"},"dist":{"shasum":"8bc43236fae802b4da3fff1f27839b2bbff4acb9","tarball":"https://registry.npmjs.org/@aos-agent/ai/-/ai-0.85.0.tgz","fileCount":754,"integrity":"sha512-jroU3yIqzahUefFxux+20ntEdYpnlz/341GA6Ijgnk2OuyQ94DRWYzYsapi8yTJu/XHgt7nNQ5Ixmx+xIcYAZw==","signatures":[{"sig":"MEUCIDR0EyJ/Gh5L0MzQsgZ6JwgfD6U+khR05KLR41VpOjCCAiEAoISwy6v6hV2Sa2fPsWoKUpxkIPDJiyKF50zovGTgRv4=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@aos-agent%2fai@0.85.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":4110515},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.19.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./api/*":{"types":"./dist/api/*.d.ts","import":"./dist/api/*.js"},"./oauth":{"types":"./dist/oauth.d.ts","import":"./dist/oauth.js"},"./compat":{"types":"./dist/compat.d.ts","import":"./dist/compat.js"},"./bun-oauth":{"types":"./dist/bun-oauth.d.ts","import":"./dist/bun-oauth.js"},"./providers/*":{"types":"./dist/providers/*.d.ts","import":"./dist/providers/*.js"},"./bedrock-provider":{"types":"./dist/bedrock-provider.d.ts","import":"./dist/bedrock-provider.js"}},"gitHead":"a305bd6d376b447cbdc0b92711a54e9778d24695","scripts":{"test":"vitest --run","build":"npm run prepare-test-catalog && npm run build:offline","clean":"shx rm -rf dist","pretest":"npm run prepare-test-catalog","build:offline":"npm run generate-aos-model-registry && npm run check:model-data && tsgo -p tsconfig.build.json && shx rm -rf dist/providers/data && shx cp -r src/providers/data dist/providers/data","prepublishOnly":"npm run clean && npm run build","generate-models":"node scripts/generate-models.ts --strict","check:model-data":"node scripts/check-model-data.ts","hydrate-model-data":"node scripts/generate-models.ts --strict --data-only","prepare-test-catalog":"node scripts/generate-models.ts --strict --snapshot test/fixtures/model-catalog && node scripts/generate-image-models.ts --strict --snapshot test/fixtures/image-models.json","generate-image-models":"node scripts/generate-image-models.ts --strict","generate-model-catalog":"node scripts/generate-models.ts --strict --json-only --json-output ../../.artifacts/model-catalog","update-aos-model-registry":"npm run generate-models && npm run generate-aos-model-registry","generate-aos-model-registry":"node scripts/generate-aos-model-registry.ts"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:3558bf2c-3ba5-4215-8837-cd27022012d4"}},"repository":{"url":"git+https://github.com/guanwenpeng2001-bot/AOS-Agent.git","type":"git","directory":"packages/ai"},"_npmVersion":"11.5.1","description":"Unified LLM API with automatic model discovery and provider configuration","directories":{},"sideEffects":["./dist/compat.js","./dist/images.js","./dist/providers/images/register-builtins.js"],"_nodeVersion":"24.19.0","dependencies":{"openai":"6.40.0","typebox":"1.3.7","partial-json":"0.1.7","@google/genai":"1.52.0","http-proxy-agent":"7.0.2","@anthropic-ai/sdk":"0.91.1","https-proxy-agent":"7.0.6","@opentelemetry/api":"1.9.0","@aos-agent/telemetry":"^0.85.0","@smithy/node-http-handler":"4.7.3","@aws-sdk/client-bedrock-runtime":"3.1048.0"},"_hasShrinkwrap":false,"devDependencies":{"canvas":"3.2.3","vitest":"4.1.9","@types/node":"24.12.4"},"_npmOperationalInternal":{"tmp":"tmp/ai_0.85.0_1788326516245_0.5228547152718757","host":"s3://npm-registry-packages-npm-production"}},"0.85.1":{"name":"@aos-agent/ai","version":"0.85.1","description":"Unified LLM API with automatic model discovery and provider configuration","type":"module","main":"./dist/index.js","types":"./dist/index.d.ts","sideEffects":["./dist/compat.js","./dist/images.js","./dist/providers/images/register-builtins.js"],"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./compat":{"types":"./dist/compat.d.ts","import":"./dist/compat.js"},"./providers/*":{"types":"./dist/providers/*.d.ts","import":"./dist/providers/*.js"},"./api/*":{"types":"./dist/api/*.d.ts","import":"./dist/api/*.js"},"./oauth":{"types":"./dist/oauth.d.ts","import":"./dist/oauth.js"},"./bedrock-provider":{"types":"./dist/bedrock-provider.d.ts","import":"./dist/bedrock-provider.js"},"./bun-oauth":{"types":"./dist/bun-oauth.d.ts","import":"./dist/bun-oauth.js"}},"bin":{"aos-agent-ai":"dist/cli.js"},"scripts":{"clean":"shx rm -rf dist","generate-models":"node scripts/generate-models.ts --strict","prepare-test-catalog":"node scripts/generate-models.ts --strict --snapshot test/fixtures/model-catalog && node scripts/generate-image-models.ts --strict --snapshot test/fixtures/image-models.json","hydrate-model-data":"node scripts/generate-models.ts --strict --data-only","generate-model-catalog":"node scripts/generate-models.ts --strict --json-only --json-output ../../.artifacts/model-catalog","generate-aos-model-registry":"node scripts/generate-aos-model-registry.ts","update-aos-model-registry":"npm run generate-models && npm run generate-aos-model-registry","generate-image-models":"node scripts/generate-image-models.ts --strict","check:model-data":"node scripts/check-model-data.ts","build":"npm run prepare-test-catalog && npm run build:offline","build:offline":"npm run generate-aos-model-registry && npm run check:model-data && tsgo -p tsconfig.build.json && shx rm -rf dist/providers/data && shx cp -r src/providers/data dist/providers/data","pretest":"npm run prepare-test-catalog","test":"vitest --run","prepublishOnly":"npm run clean && npm run build"},"dependencies":{"@anthropic-ai/sdk":"0.91.1","@aws-sdk/client-bedrock-runtime":"3.1048.0","@aos-agent/telemetry":"^0.85.1","@google/genai":"1.52.0","@opentelemetry/api":"1.9.0","@smithy/node-http-handler":"4.7.3","http-proxy-agent":"7.0.2","https-proxy-agent":"7.0.6","openai":"6.40.0","partial-json":"0.1.7","typebox":"1.3.7"},"keywords":["ai","llm","openai","anthropic","gemini","bedrock","unified","api"],"author":{"name":"AOS Agent"},"license":"MIT","repository":{"type":"git","url":"git+https://github.com/guanwenpeng2001-bot/AOS-Agent.git","directory":"packages/ai"},"engines":{"node":">=22.19.0"},"devDependencies":{"@types/node":"24.12.4","canvas":"3.2.3","vitest":"4.1.9"},"_id":"@aos-agent/ai@0.85.1","gitHead":"8fc01dea402bced66d545cf6febddcd944d8943b","bugs":{"url":"https://github.com/guanwenpeng2001-bot/AOS-Agent/issues"},"homepage":"https://github.com/guanwenpeng2001-bot/AOS-Agent#readme","_nodeVersion":"24.20.0","_npmVersion":"11.5.1","dist":{"integrity":"sha512-piahRWQrsJp2EtphuMRjuR0n7r0RGmXRe9tVr2x4BxVs7VK9YhSkpbazDlmNw9QD3DQK4FnHuN97xiTSskBmew==","shasum":"db244e888e4e0c93dc0714c6a06c7b87a0c847bc","tarball":"https://registry.npmjs.org/@aos-agent/ai/-/ai-0.85.1.tgz","fileCount":754,"unpackedSize":4110515,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@aos-agent%2fai@0.85.1","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQCqn1xUw5cMQgGr3HK44c/aYHNjDG9PbGdWYYsBJuBTgwIgZX8H1Qg7YhhFMm4+7Q90POMwDSy/oyK/1KGKr5/bpxE="}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:3558bf2c-3ba5-4215-8837-cd27022012d4"}},"directories":{},"maintainers":[{"name":"aosagent","email":"guanwenpeng2001@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/ai_0.85.1_1788451880723_0.9021453161391988"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-10T00:58:57.215Z","modified":"2026-09-03T16:11:21.249Z","0.84.2":"2026-08-10T00:58:57.551Z","0.84.3":"2026-08-10T17:44:53.107Z","0.85.0":"2026-09-02T05:21:56.372Z","0.85.1":"2026-09-03T16:11:20.896Z"},"bugs":{"url":"https://github.com/guanwenpeng2001-bot/AOS-Agent/issues"},"author":{"name":"AOS Agent"},"license":"MIT","homepage":"https://github.com/guanwenpeng2001-bot/AOS-Agent#readme","keywords":["ai","llm","openai","anthropic","gemini","bedrock","unified","api"],"repository":{"type":"git","url":"git+https://github.com/guanwenpeng2001-bot/AOS-Agent.git","directory":"packages/ai"},"description":"Unified LLM API with automatic model discovery and provider configuration","maintainers":[{"name":"aosagent","email":"guanwenpeng2001@gmail.com"}],"readme":"# @aos-agent/ai\n\nUnified LLM API with provider collections, automatic auth resolution, token and cost tracking, and simple context persistence and hand-off to other models mid-session.\n\n**Note**: This library only includes models that support tool calling (function calling), as this is essential for agentic workflows.\n\n## AOS model registry\n\nThe standalone AOS Agent build owns its generated provider-neutral registry under `.artifacts/aos-model-registry/`. Default `build` hydrates ignored wrappers from the tracked `test/fixtures` snapshot so CI does not fetch live catalogs. Run `npm run update-aos-model-registry` to refresh reviewed public metadata inputs, or `npm run generate-aos-model-registry` to reproduce output from an existing hydrated input snapshot. The generation boundary records source URL/name, retrieval timestamp, input hashes, normalization policy, and review status; see [`aos-model-registry.md`](aos-model-registry.md). Public availability alone is not permission to redistribute metadata.\n\n## Table of Contents\n\n- [Supported Providers](#supported-providers)\n- [Installation](#installation)\n- [Quick Start](#quick-start)\n- [Providers and Models](#providers-and-models)\n  - [Provider Factories](#provider-factories)\n  - [All Built-in Providers](#all-built-in-providers)\n  - [Querying Models](#querying-models)\n  - [Static Catalog Reads](#static-catalog-reads)\n  - [Dynamic Providers](#dynamic-providers)\n- [Auth](#auth)\n  - [How Auth Resolves](#how-auth-resolves)\n  - [Transforming Request Headers](#transforming-request-headers)\n  - [Credential Store](#credential-store)\n  - [Environment Variables](#environment-variables)\n- [Tools](#tools)\n  - [Defining Tools](#defining-tools)\n  - [Handling Tool Calls](#handling-tool-calls)\n  - [Streaming Tool Calls with Partial JSON](#streaming-tool-calls-with-partial-json)\n  - [Validating Tool Arguments](#validating-tool-arguments)\n  - [Complete Event Reference](#complete-event-reference)\n- [Image Input](#image-input)\n- [Image Generation](#image-generation)\n- [Thinking/Reasoning](#thinkingreasoning)\n  - [Unified Interface](#unified-interface-streamsimplecompletesimple)\n  - [Provider-Specific Options](#provider-specific-options-streamcomplete)\n  - [Streaming Thinking Content](#streaming-thinking-content)\n- [Stop Reasons](#stop-reasons)\n- [Error Handling](#error-handling)\n  - [Aborting Requests](#aborting-requests)\n  - [Continuing After Abort](#continuing-after-abort)\n  - [Debugging Provider Payloads](#debugging-provider-payloads)\n- [Custom Providers](#custom-providers)\n  - [createProvider()](#createprovider)\n  - [Calling API Implementations Directly](#calling-api-implementations-directly)\n  - [OpenAI Compatibility Settings](#openai-compatibility-settings)\n- [Fake Provider for Tests](#fake-provider-for-tests)\n- [Cross-Provider Handoffs](#cross-provider-handoffs)\n- [Context Serialization](#context-serialization)\n- [Browser Usage](#browser-usage)\n- [Bundling and Tree Shaking](#bundling-and-tree-shaking)\n- [OAuth Providers](#oauth-providers)\n  - [Vertex AI](#vertex-ai)\n  - [CLI Login](#cli-login)\n  - [Programmatic OAuth](#programmatic-oauth)\n- [Migrating from the Old Global API](#migrating-from-the-old-global-api)\n- [Development](#development)\n- [License](#license)\n\n## Supported Providers\n\n- **OpenAI**\n- **Ant Ling**\n- **Azure OpenAI (Responses)**\n- **OpenAI Codex** (ChatGPT Plus/Pro subscription, requires OAuth, see below)\n- **DeepSeek**\n- **NVIDIA NIM**\n- **Anthropic**\n- **Google**\n- **Vertex AI** (Gemini via Vertex AI)\n- **Mistral**\n- **Groq**\n- **Cerebras**\n- **Cloudflare AI Gateway**\n- **Cloudflare Workers AI**\n- **xAI**\n- **OpenRouter**\n- **Vercel AI Gateway**\n- **ZAI Coding Plan (Global)** (with separate China provider)\n- **MiniMax** (with separate China provider)\n- **Together AI**\n- **Baseten**\n- **Hugging Face**\n- **Moonshot AI** (with separate China provider)\n- **GitHub Copilot** (requires OAuth, see below)\n- **Amazon Bedrock**\n- **OpenCode Zen**\n- **OpenCode Go**\n- **Fireworks** (uses OpenAI- and Anthropic-compatible APIs)\n- **Kimi For Coding** (Moonshot AI subscription endpoint, uses Anthropic-compatible API)\n- **Qwen Token Plan** (separate Individual and existing catalogs, with a separate China provider)\n- **Xiaomi MiMo** (defaults to API billing endpoint, with separate Token Plan providers for `cn`/`ams`/`sgp` regions)\n- **Any OpenAI-compatible API**: Ollama, vLLM, LM Studio, etc.\n\n## Installation\n\n```bash\nnpm install @aos-agent/ai\n```\n\nTypeBox exports are re-exported from `@aos-agent/ai`: `Type`, `Static`, and `TSchema`.\n\n## Quick Start\n\nYou build a `Models` collection of providers and stream through it. The quickest start registers every built-in provider; apps that care about bundle size register individual providers instead (see [Provider Factories](#provider-factories) and [Bundling and Tree Shaking](#bundling-and-tree-shaking)).\n\n```typescript\nimport { Type, type Context, type Tool } from '@aos-agent/ai';\nimport { builtinModels } from '@aos-agent/ai/providers/all';\n\n// A Models collection with every built-in provider registered\nconst models = builtinModels();\n\n// Sync lookup against the collection\nconst model = models.getModel('openai', 'gpt-4o-mini')!;\n\n// Define tools with TypeBox schemas for type safety and validation\nconst tools: Tool[] = [{\n  name: 'get_time',\n  description: 'Get the current time',\n  parameters: Type.Object({\n    timezone: Type.Optional(Type.String({ description: 'Optional timezone (e.g., America/New_York)' }))\n  })\n}];\n\n// Build a conversation context (easily serializable and transferable between models)\nconst context: Context = {\n  systemPrompt: 'You are a helpful assistant.',\n  messages: [{ role: 'user', content: 'What time is it?', timestamp: Date.now() }],\n  tools\n};\n\n// Option 1: Streaming with all event types.\n// Auth resolves through the provider (OPENAI_API_KEY from the environment here).\nconst s = models.stream(model, context);\n\nfor await (const event of s) {\n  switch (event.type) {\n    case 'start':\n      console.log(`Starting with ${event.partial.model}`);\n      break;\n    case 'text_start':\n      console.log('\\n[Text started]');\n      break;\n    case 'text_delta':\n      process.stdout.write(event.delta);\n      break;\n    case 'text_end':\n      console.log('\\n[Text ended]');\n      break;\n    case 'thinking_start':\n      console.log('[Model is thinking...]');\n      break;\n    case 'thinking_delta':\n      process.stdout.write(event.delta);\n      break;\n    case 'thinking_end':\n      console.log('[Thinking complete]');\n      break;\n    case 'toolcall_start':\n      console.log(`\\n[Tool call started: index ${event.contentIndex}]`);\n      break;\n    case 'toolcall_delta':\n      // Partial tool arguments are being streamed\n      const partialCall = event.partial.content[event.contentIndex];\n      if (partialCall.type === 'toolCall') {\n        console.log(`[Streaming args for ${partialCall.name}]`);\n      }\n      break;\n    case 'toolcall_end':\n      console.log(`\\nTool called: ${event.toolCall.name}`);\n      console.log(`Arguments: ${JSON.stringify(event.toolCall.arguments)}`);\n      break;\n    case 'done':\n      console.log(`\\nFinished: ${event.reason}`);\n      break;\n    case 'error':\n      console.error(`Error: ${event.error.errorMessage}`);\n      break;\n  }\n}\n\n// Get the final message after streaming, add it to the context\nconst finalMessage = await s.result();\ncontext.messages.push(finalMessage);\n\n// Handle tool calls if any\nconst toolCalls = finalMessage.content.filter(b => b.type === 'toolCall');\nfor (const call of toolCalls) {\n  const result = call.name === 'get_time'\n    ? new Date().toLocaleString('en-US', {\n        timeZone: call.arguments.timezone || 'UTC',\n        dateStyle: 'full',\n        timeStyle: 'long'\n      })\n    : 'Unknown tool';\n\n  // Add tool result to context (supports text and images)\n  context.messages.push({\n    role: 'toolResult',\n    toolCallId: call.id,\n    toolName: call.name,\n    content: [{ type: 'text', text: result }],\n    isError: false,\n    timestamp: Date.now()\n  });\n}\n\n// Continue if there were tool calls\nif (toolCalls.length > 0) {\n  const continuation = await models.complete(model, context);\n  context.messages.push(continuation);\n  console.log('After tool execution:', continuation.content);\n}\n\nconsole.log(`Total tokens: ${finalMessage.usage.input} in, ${finalMessage.usage.output} out`);\nconsole.log(`Cost: $${finalMessage.usage.cost.total.toFixed(4)}`);\n\n// Option 2: Get complete response without streaming\nconst response = await models.complete(model, context);\n\nfor (const block of response.content) {\n  if (block.type === 'text') {\n    console.log(block.text);\n  } else if (block.type === 'toolCall') {\n    console.log(`Tool: ${block.name}(${JSON.stringify(block.arguments)})`);\n  }\n}\n```\n\nSnippets in the rest of this README assume a `models` collection set up like this (with the relevant providers registered).\n\n## Providers and Models\n\nA **provider** is the runtime unit: it owns its model catalog, its auth (API key resolution, OAuth flows), and its stream behavior. A `Models` collection holds providers and routes every request to the provider that owns the model.\n\nProviders internally share **API implementations** (the wire protocols): Anthropic models use `anthropic-messages`, OpenAI uses `openai-responses`, while xAI, Groq, Cerebras, OpenRouter, and most others share `openai-completions`. Mixed-API providers (GitHub Copilot, OpenCode Zen) dispatch per model.\n\n### Provider Factories\n\nFor apps that only need specific providers, there is one factory per built-in provider, each a subpath import that pulls only that provider's catalog:\n\n```typescript\nimport { anthropicProvider } from '@aos-agent/ai/providers/anthropic';\nimport { openaiProvider } from '@aos-agent/ai/providers/openai';\nimport { openrouterProvider } from '@aos-agent/ai/providers/openrouter';\nimport { amazonBedrockProvider } from '@aos-agent/ai/providers/amazon-bedrock';\n// ...one module per provider in the Supported Providers list\n\nconst models = createModels();\nmodels.setProvider(anthropicProvider());\nmodels.setProvider(openrouterProvider());\n```\n\nProvider factories import their model catalog and a lazy API wrapper. They do not import other providers. With bundler code splitting, SDK implementations (`@anthropic-ai/sdk`, `openai`, `@google/genai`, etc.) stay in lazy chunks loaded on the first request to a model of that API.\n\n### All Built-in Providers\n\nFor apps that want everything (as in Quick Start):\n\n```typescript\nimport { builtinModels } from '@aos-agent/ai/providers/all';\n\nconst models = builtinModels(); // a Models collection with every built-in provider registered\n```\n\nThis imports all catalogs and every built-in provider factory. It is the heavy, explicit entrypoint. `builtinModels()` accepts the same options as `createModels()` (`credentials`, `authContext`); `builtinProviders()` returns the provider array if you want to register them on your own collection.\n\n### Querying Models\n\nReads are synchronous and return the last-known lists:\n\n```typescript\nconst providers = models.getProviders();           // registered Provider objects\nconst provider = models.getProvider('anthropic');  // one provider\n\nconst all = models.getModels();                    // every model across providers\nconst anthropicModels = models.getModels('anthropic');\nconst model = models.getModel('anthropic', 'claude-sonnet-4-5');\n\nfor (const m of anthropicModels) {\n  console.log(`${m.id}: ${m.name}`);\n  console.log(`  API: ${m.api}`);\n  console.log(`  Context: ${m.contextWindow} tokens`);\n  console.log(`  Vision: ${m.input.includes('image')}`);\n  console.log(`  Reasoning: ${m.reasoning}`);\n}\n```\n\nDynamically listed models are typed `Model<Api>`. Narrow with the `hasApi()` guard when you need API-specific option typing:\n\n```typescript\nimport { hasApi } from '@aos-agent/ai';\n\nconst m = models.getModel('anthropic', 'claude-sonnet-4-5');\nif (m && hasApi(m, 'anthropic-messages')) {\n  // m: Model<'anthropic-messages'> — stream options fully typed\n  models.stream(m, context, { thinkingEnabled: true, thinkingBudgetTokens: 2048 });\n}\n```\n\n### Static Catalog Reads\n\nFor tooling that wants the generated built-in catalog with full literal typing (provider and model IDs auto-complete), independent of any collection:\n\n```typescript\nimport { getBuiltinModel, getBuiltinModels, getBuiltinProviders } from '@aos-agent/ai/providers/all';\n\nconst model = getBuiltinModel('openai', 'gpt-4o-mini'); // typed Model<'openai-responses'>\nconst providers = getBuiltinProviders();\nconst anthropic = getBuiltinModels('anthropic');\n```\n\n### Dynamic Providers\n\nProviders may have dynamic model lists (a llama.cpp server, a live OpenRouter listing). Reads stay sync; fetching is an explicit async verb:\n\n```typescript\n// getModels() returns the last-known list (empty before the first refresh)\nawait models.refresh({ providers: ['llamacpp'] }); // refresh one provider\nawait models.refresh();                            // refresh all providers concurrently, best-effort\nconst fresh = models.getModel('llamacpp', 'qwen3-30b');\n```\n\nStatic built-in providers are no-ops for `refresh()`. See [createProvider()](#createprovider) for building a dynamic provider.\n\n## Auth\n\nEvery provider owns its auth: how API keys resolve (stored credentials, environment variables, ambient sources like AWS profiles or gcloud ADC) and, where supported, OAuth login/refresh flows.\n\n### How Auth Resolves\n\nWhen you call `models.stream()`, the collection resolves auth through the owning provider and merges it into the request. Explicit per-request values always win:\n\n```typescript\n// Resolved through the provider (env var, stored credential, OAuth token):\nawait models.complete(model, context);\n\n// Explicit key wins over anything the provider would resolve:\nawait models.complete(model, context, { apiKey: 'sk-explicit' });\n```\n\nYou can inspect resolution without making a request. Pass a provider ID for provider-scoped auth, or a model to include its static `model.headers`:\n\n```typescript\nconst providerAuth = await models.getAuth(model.provider);\nconst modelAuth = await models.getAuth(model);\n\nif (modelAuth) {\n  console.log(`configured via ${modelAuth.source}`); // e.g. \"ANTHROPIC_API_KEY\", \"OAuth\", \"stored credential\"\n  console.log(modelAuth.auth.headers);              // Provider auth headers + model.headers\n} else {\n  console.log('not configured');\n}\n```\n\nBoth overloads resolve credentials, refresh expired OAuth when necessary, and may return an auth-derived `apiKey`, `headers`, or `baseUrl`. `getAuth()` resolves `undefined` for unconfigured providers and rejects with `ModelsError` when something is actually broken (`\"oauth\"`: token refresh failed, credential preserved for re-login; `\"auth\"`: key resolution or credential store failure). Request paths surface the same failures as stream errors.\n\n`getAuth()`, `checkAuth()`, `getAvailable()`, login, and logout accept optional caller cancellation through their existing options or interaction objects and remain unbounded when no signal is supplied. Provider `login`, `ApiKeyAuth.check`, `ApiKeyAuth.resolve`, and `OAuthAuth.refresh` implementations always receive a concrete signal and must honor it for blocking work.\n\n### Transforming Request Headers\n\n`Models.stream()`, `complete()`, `streamSimple()`, and `completeSimple()` accept a Models-only `transformHeaders` option. It runs once after provider auth, `model.headers`, and explicit `options.headers` have been merged, but before provider dispatch:\n\n```typescript\nconst response = await models.completeSimple(model, context, {\n  headers: { \"X-Client\": \"my-app\" },\n  transformHeaders: async (headers) => ({\n    ...headers,\n    \"X-Request-ID\": crypto.randomUUID(),\n  }),\n});\n```\n\nThe ordering is:\n\n```text\nprovider auth headers -> model.headers -> explicit options.headers -> transformHeaders -> Provider.stream*()\n```\n\nHeader names are merged case-insensitively. Explicit headers override auth/model headers, and the transform has final control; returning `null` for a header suppresses lower-level defaults that support deletion.\n\n`transformHeaders` belongs to `Models`, not `Provider`. A `Models` implementation must consume it and remove it before calling `Provider.stream*()`. Provider implementations continue receiving ordinary `ApiStreamOptions` or `SimpleStreamOptions` and never handle the transform themselves. Use this option instead of calling `getAuth(model)` before `stream*()`, which would resolve request auth twice.\n\n### Credential Store\n\nStored credentials (API keys entered interactively, OAuth tokens) live in a `CredentialStore` — one type-tagged credential per provider. @aos-agent/ai ships an in-memory default; apps inject persistent storage:\n\n```typescript\nimport { createModels, type CredentialStore } from '@aos-agent/ai';\n\nconst models = createModels({ credentials: myFileBackedStore });\n// builtinModels() takes the same options:\n// const models = builtinModels({ credentials: myFileBackedStore });\n```\n\nThe contract is small: `read(providerId)`, `list()` for non-secret `{ providerId, type }` metadata, `modify(providerId, fn)` (the only write path — a serialized read-modify-write), and `delete(providerId)`. Each operation accepts optional cancellation options. Enumeration must not resolve secrets or execute configured key commands. OAuth token refresh runs inside `modify`, so concurrent requests and processes cannot double-refresh a rotated token. A stored credential *owns* its provider: environment variables are only consulted when nothing is stored, and a failed refresh never silently falls back to an env key.\n\nAPI-key credentials use the same discriminator as AOS Agent's `auth.json` and can carry provider-scoped env/config values:\n\n```typescript\nconst credential = {\n  type: 'api_key',\n  key: '...',\n  env: {\n    CLOUDFLARE_ACCOUNT_ID: 'account-id',\n    CLOUDFLARE_GATEWAY_ID: 'gateway-id'\n  }\n} as const;\n```\n\n### Environment Variables\n\nBuilt-in providers resolve these env vars (Node.js; in browsers pass `apiKey` explicitly):\n\n| Provider | Environment Variable(s) |\n|----------|------------------------|\n| OpenAI | `OPENAI_API_KEY` |\n| Ant Ling | `ANT_LING_API_KEY` |\n| Azure OpenAI | `AZURE_OPENAI_API_KEY` + `AZURE_OPENAI_BASE_URL` (e.g. `https://{resource}.ai.azure.com`) or `AZURE_OPENAI_RESOURCE_NAME`. Supports `*.openai.azure.com`, `*.cognitiveservices.azure.com` and `*.ai.azure.com`; root endpoints auto-normalize to `/openai/v1`. Optional: `AZURE_OPENAI_API_VERSION` (default `v1`), `AZURE_OPENAI_DEPLOYMENT_NAME_MAP`. |\n| Anthropic | `ANTHROPIC_API_KEY` or `ANTHROPIC_OAUTH_TOKEN` |\n| DeepSeek | `DEEPSEEK_API_KEY` |\n| NVIDIA NIM | `NVIDIA_API_KEY` |\n| Google | `GEMINI_API_KEY` |\n| Vertex AI | `GOOGLE_CLOUD_API_KEY` or `GOOGLE_CLOUD_PROJECT` (or `GCLOUD_PROJECT`) + `GOOGLE_CLOUD_LOCATION` + ADC |\n| Mistral | `MISTRAL_API_KEY` |\n| Groq | `GROQ_API_KEY` |\n| Cerebras | `CEREBRAS_API_KEY` |\n| Cloudflare AI Gateway | `CLOUDFLARE_API_KEY` + `CLOUDFLARE_ACCOUNT_ID` + `CLOUDFLARE_GATEWAY_ID` |\n| Cloudflare Workers AI | `CLOUDFLARE_API_KEY` + `CLOUDFLARE_ACCOUNT_ID` |\n| xAI | `XAI_API_KEY` |\n| Fireworks | `FIREWORKS_API_KEY` |\n| Together AI | `TOGETHER_API_KEY` |\n| Baseten | `BASETEN_API_KEY` |\n| OpenRouter | `OPENROUTER_API_KEY` |\n| Vercel AI Gateway | `AI_GATEWAY_API_KEY` |\n| ZAI Coding Plan (Global) | `ZAI_API_KEY` |\n| ZAI Coding Plan (China) | `ZAI_CODING_CN_API_KEY` |\n| MiniMax (Global) | `MINIMAX_API_KEY` |\n| MiniMax (China) | `MINIMAX_CN_API_KEY` |\n| Moonshot AI / Moonshot AI (China) | `MOONSHOT_API_KEY` |\n| Hugging Face | `HF_TOKEN` |\n| OpenCode Zen / OpenCode Go | `OPENCODE_API_KEY` |\n| Kimi For Coding | `KIMI_API_KEY` |\n| Qwen Token Plan (existing catalog) | `QWEN_TOKEN_PLAN_API_KEY` |\n| Qwen Token Plan (Individual) | `QWEN_TOKEN_PLAN_API_KEY` |\n| Qwen Token Plan (China) | `QWEN_TOKEN_PLAN_CN_API_KEY` |\n| Xiaomi MiMo (API billing) | `XIAOMI_API_KEY` |\n| Xiaomi MiMo Token Plan (China) | `XIAOMI_TOKEN_PLAN_CN_API_KEY` |\n| Xiaomi MiMo Token Plan (Amsterdam) | `XIAOMI_TOKEN_PLAN_AMS_API_KEY` |\n| Xiaomi MiMo Token Plan (Singapore) | `XIAOMI_TOKEN_PLAN_SGP_API_KEY` |\n| GitHub Copilot | `COPILOT_GITHUB_TOKEN` |\n\n`qwen-token-plan-individual` and `qwen-token-plan` share the international endpoint and\n`QWEN_TOKEN_PLAN_API_KEY`. The Individual provider exposes only the models documented for Individual\nsubscriptions, while the existing provider retains its broader catalog for backward compatibility.\nStored credentials remain provider-scoped, so save the key under the provider ID you register.\n\nAmazon Bedrock resolves ambient AWS credentials (`AWS_PROFILE`, access key pairs, `AWS_BEARER_TOKEN_BEDROCK`, ECS task roles, web identity tokens); its provider-owned login flow supports bearer tokens, AWS profiles, and the existing credential chain. Vertex AI resolves either an explicit key or gcloud Application Default Credentials plus project/location, with a provider-owned login flow for API keys, ADC, and service-account files.\n\n## Tools\n\nTools enable LLMs to interact with external systems. This library uses TypeBox schemas for type-safe tool definitions with automatic validation using TypeBox's built-in validator and value conversion utilities. TypeBox schemas can be serialized and deserialized as plain JSON, making them ideal for distributed systems.\n\n### Defining Tools\n\n```typescript\nimport { Type, type Tool, StringEnum } from '@aos-agent/ai';\n\n// Define tool parameters with TypeBox\nconst weatherTool: Tool = {\n  name: 'get_weather',\n  description: 'Get current weather for a location',\n  parameters: Type.Object({\n    location: Type.String({ description: 'City name or coordinates' }),\n    units: StringEnum(['celsius', 'fahrenheit'], { default: 'celsius' })\n  })\n};\n\n// Note: For Google API compatibility, use StringEnum helper instead of Type.Enum\n// Type.Enum generates anyOf/const patterns that Google doesn't support\n\nconst bookMeetingTool: Tool = {\n  name: 'book_meeting',\n  description: 'Schedule a meeting',\n  parameters: Type.Object({\n    title: Type.String({ minLength: 1 }),\n    startTime: Type.String({ format: 'date-time' }),\n    endTime: Type.String({ format: 'date-time' }),\n    attendees: Type.Array(Type.String({ format: 'email' }), { minItems: 1 })\n  })\n};\n```\n\n### Constrained Sampling for Tools\n\nTools can opt in to provider-side constrained sampling. For JSON-schema tools, `strict: 'prefer'` uses provider-side strict schema enforcement when supported and otherwise falls back to normal tool calling. `strict: 'require'` fails the request when the active provider/model cannot honor it. Set `constrainedSampling: false` to explicitly opt out; it behaves the same as omitting the field.\n\n```typescript\nconst strictTool: Tool = {\n  name: 'edit_file',\n  description: 'Edit a file',\n  parameters: Type.Object({\n    path: Type.String(),\n    content: Type.String()\n  }, { additionalProperties: false }),\n  constrainedSampling: { type: 'json_schema', strict: 'prefer' }\n};\n```\n\nStrict JSON-schema constrained sampling is supported for OpenAI, Anthropic, supported Amazon Bedrock Converse models, Mistral, and Gemini 3 tool calls through the Google Generative AI and Vertex adapters. Google uses `VALIDATED` function-calling mode (or `ANY` when explicitly requested); earlier Gemini versions fall back for `strict: 'prefer'` and reject `strict: 'require'` because they do not enforce required parameters. Bedrock strict-tool capability is generated from model structured-output metadata; custom Bedrock models can override `compat.supportsStrictMode`. OpenAI Responses and Chat Completions can also emit grammar-constrained custom tools with OpenAI Lark or regex grammar variants. If multiple OpenAI variants are supplied, Lark is preferred over regex. Grammar constraints are enforced when the active model supports grammar tools; otherwise the tool falls back to normal function/JSON-schema handling. Grammar tool capability is model metadata: the generated catalog sets `compat.supportsOpenAIGrammarTools` for GPT-5+ models on endpoints that pass OpenAI custom tools through (OpenAI, OpenAI Codex, Azure OpenAI Responses, GitHub Copilot, opencode, and Cloudflare AI Gateway). OpenAI rejects `type: \"custom\"` tools for pre-GPT-5 models, and gateways that normalize tool schemas (e.g. OpenRouter) mangle them, so the flag stays off elsewhere. Custom model definitions can opt in via `compat`. Grammar-capable models reject grammar configurations without a non-empty supported variant. Native grammar tools must have an object parameter schema with exactly one required string property:\n\n```typescript\nconst patchTool: Tool = {\n  name: 'apply_patch',\n  description: 'Apply a patch',\n  parameters: Type.Object({\n    input: Type.String()\n  }, { additionalProperties: false }),\n  constrainedSampling: {\n    type: 'grammar',\n    variants: {\n      openai_lark: 'start: /.+/s'\n    }\n  }\n};\n```\n\n### Handling Tool Calls\n\nTool results use content blocks and can include both text and images:\n\n```typescript\nimport { readFileSync } from 'fs';\n\nconst context: Context = {\n  messages: [{ role: 'user', content: 'What is the weather in London?', timestamp: Date.now() }],\n  tools: [weatherTool]\n};\n\nconst response = await models.complete(model, context);\n\n// Check for tool calls in the response\nfor (const block of response.content) {\n  if (block.type === 'toolCall') {\n    // Execute your tool with the arguments\n    // See \"Validating Tool Arguments\" section for validation\n    const result = await executeWeatherApi(block.arguments);\n\n    // Add tool result with text content\n    context.messages.push({\n      role: 'toolResult',\n      toolCallId: block.id,\n      toolName: block.name,\n      content: [{ type: 'text', text: JSON.stringify(result) }],\n      isError: false,\n      timestamp: Date.now()\n    });\n  }\n}\n\n// Tool results can also include images (for vision-capable models)\nconst imageBuffer = readFileSync('chart.png');\ncontext.messages.push({\n  role: 'toolResult',\n  toolCallId: 'tool_xyz',\n  toolName: 'generate_chart',\n  content: [\n    { type: 'text', text: 'Generated chart showing temperature trends' },\n    { type: 'image', data: imageBuffer.toString('base64'), mimeType: 'image/png' }\n  ],\n  isError: false,\n  timestamp: Date.now()\n});\n```\n\n### Streaming Tool Calls with Partial JSON\n\nDuring streaming, tool call arguments are progressively parsed as they arrive. This enables real-time UI updates before the complete arguments are available:\n\n```typescript\nconst s = models.stream(model, context);\n\nfor await (const event of s) {\n  if (event.type === 'toolcall_delta') {\n    const toolCall = event.partial.content[event.contentIndex];\n\n    // toolCall.arguments contains partially parsed JSON during streaming\n    // This allows for progressive UI updates\n    if (toolCall.type === 'toolCall' && toolCall.arguments) {\n      // BE DEFENSIVE: arguments may be incomplete\n      // Example: Show file path being written even before content is complete\n      if (toolCall.name === 'write_file' && toolCall.arguments.path) {\n        console.log(`Writing to: ${toolCall.arguments.path}`);\n\n        // Content might be partial or missing\n        if (toolCall.arguments.content) {\n          console.log(`Content preview: ${toolCall.arguments.content.substring(0, 100)}...`);\n        }\n      }\n    }\n  }\n\n  if (event.type === 'toolcall_end') {\n    // Here toolCall.arguments is complete (but not yet validated)\n    const toolCall = event.toolCall;\n    console.log(`Tool completed: ${toolCall.name}`, toolCall.arguments);\n  }\n}\n```\n\n**Important notes about partial tool arguments:**\n- During `toolcall_delta` events, `arguments` contains the best-effort parse of partial JSON\n- Fields may be missing or incomplete - always check for existence before use\n- String values may be truncated mid-word\n- Arrays may be incomplete\n- Nested objects may be partially populated\n- At minimum, `arguments` will be an empty object `{}`, never `undefined`\n- The Google provider does not support function call streaming. Instead, you will receive a single `toolcall_delta` event with the full arguments.\n\n### Validating Tool Arguments\n\nWhen implementing your own tool execution loop, use `validateToolCall` to validate arguments before passing them to your tools:\n\n```typescript\nimport { validateToolCall, type Tool } from '@aos-agent/ai';\n\nconst tools: Tool[] = [weatherTool, calculatorTool];\nconst s = models.stream(model, { messages, tools });\n\nfor await (const event of s) {\n  if (event.type === 'toolcall_end') {\n    const toolCall = event.toolCall;\n\n    try {\n      // Validate arguments against the tool's schema (throws on invalid args)\n      const validatedArgs = validateToolCall(tools, toolCall);\n      const result = await executeMyTool(toolCall.name, validatedArgs);\n      // ... add tool result to context\n    } catch (error) {\n      // Validation failed - return error as tool result so model can retry\n      context.messages.push({\n        role: 'toolResult',\n        toolCallId: toolCall.id,\n        toolName: toolCall.name,\n        content: [{ type: 'text', text: error.message }],\n        isError: true,\n        timestamp: Date.now()\n      });\n    }\n  }\n}\n```\n\n### Complete Event Reference\n\nAll streaming events emitted during assistant message generation:\n\n| Event Type | Description | Key Properties |\n|------------|-------------|----------------|\n| `start` | Stream begins | `partial`: Initial assistant message structure |\n| `text_start` | Text block starts | `contentIndex`: Position in content array |\n| `text_delta` | Text chunk received | `delta`: New text, `contentIndex`: Position |\n| `text_end` | Text block complete | `content`: Full text, `contentIndex`: Position |\n| `thinking_start` | Thinking block starts | `contentIndex`: Position in content array |\n| `thinking_delta` | Thinking chunk received | `delta`: New text, `contentIndex`: Position |\n| `thinking_end` | Thinking block complete | `content`: Full thinking, `contentIndex`: Position |\n| `toolcall_start` | Tool call begins | `contentIndex`: Position in content array |\n| `toolcall_delta` | Tool arguments streaming | `delta`: JSON chunk, `partial.content[contentIndex].arguments`: Partial parsed args |\n| `toolcall_end` | Tool call complete | `toolCall`: Complete validated tool call with `id`, `name`, `arguments` |\n| `done` | Stream complete | `reason`: Stop reason (\"stop\", \"length\", \"toolUse\"), `message`: Final assistant message |\n| `error` | Error occurred | `reason`: Error type (\"error\" or \"aborted\"), `error`: AssistantMessage with partial content |\n\nStreaming events for different content blocks are not guaranteed to be contiguous. Providers may emit deltas for text, thinking, and tool calls in the same upstream chunk, and AOS Agent may surface corresponding events interleaved, for example `text_start`, `text_delta`, `toolcall_start`, `text_delta`, `toolcall_delta`. Consumers must use `contentIndex` to associate each delta/end event with its block and must not assume that a block's `*_start`/`*_delta`/`*_end` sequence is uninterrupted by events for other blocks.\n\n## Image Input\n\nModels with vision capabilities can process images. You can check if a model supports images via the `input` property. If you pass images to a non-vision model, they are silently ignored.\n\n```typescript\nimport { readFileSync } from 'fs';\n\nconst model = models.getModel('openai', 'gpt-4o-mini')!;\n\n// Check if model supports images\nif (model.input.includes('image')) {\n  console.log('Model supports vision');\n}\n\nconst imageBuffer = readFileSync('image.png');\nconst base64Image = imageBuffer.toString('base64');\n\nconst response = await models.complete(model, {\n  messages: [{\n    role: 'user',\n    content: [\n      { type: 'text', text: 'What is in this image?' },\n      { type: 'image', data: base64Image, mimeType: 'image/png' }\n    ],\n    timestamp: Date.now()\n  }]\n});\n\n// Access the response\nfor (const block of response.content) {\n  if (block.type === 'text') {\n    console.log(block.text);\n  }\n}\n```\n\n## Image Generation\n\nImage generation uses a separate API surface from text/chat generation, mirroring the chat-side design: an `ImagesModels` collection holds `ImagesProvider`s, reads are sync, and auth resolves through the owning provider. Image generation is a one-shot API: `generateImages()` waits for the provider response and returns the final `AssistantImages` result — do not use the chat/stream APIs for it.\n\n### Basic Image Generation\n\n```typescript\nimport { builtinImagesModels } from '@aos-agent/ai/providers/all';\n\n// Every built-in image-generation provider; accepts the same options as createModels()\nconst imagesModels = builtinImagesModels();\n\nconst model = imagesModels.getModel('openrouter', 'google/gemini-2.5-flash-image')!;\n\n// Auth resolves through the provider (OPENROUTER_API_KEY here); explicit apiKey wins\nconst result = await imagesModels.generateImages(model, {\n  input: [{ type: 'text', text: 'Generate a red circle on a plain white background.' }]\n});\n\nfor (const block of result.output) {\n  if (block.type === 'text') {\n    console.log(block.text);\n  } else if (block.type === 'image') {\n    console.log(block.mimeType);\n    console.log(block.data.substring(0, 32));\n  }\n}\n```\n\nLike the chat side, you can build the collection from parts: `createImagesModels({ credentials?, authContext? })`, the `openrouterImagesProvider()` factory from `@aos-agent/ai/providers/openrouter-images`, and `createImagesProvider({ id, auth, models, refreshModels?, api })` for custom image providers (with `imagesModels.refresh(provider?)` for dynamic lists). Failures never reject — they return an `AssistantImages` with `stopReason: \"error\"`. The collection's provider-scoped `getAuth(providerId)` works exactly like the chat-side one.\n\nThe old global API (`getImageModel()` / `getImageModels()` / `getImageProviders()` / `generateImages()`) remains available on the [compat entrypoint](#migrating-from-the-old-global-api):\n\n```typescript\nimport { getImageModel, generateImages } from '@aos-agent/ai/compat';\n\nconst model = getImageModel('openrouter', 'google/gemini-2.5-flash-image');\nconst result = await generateImages(model, {\n  input: [{ type: 'text', text: 'Generate a red circle on a plain white background.' }]\n}, {\n  apiKey: process.env.OPENROUTER_API_KEY\n});\n```\n\nSome models also support image input:\n\n```typescript\nimport { readFileSync } from 'fs';\n\nconst imageBuffer = readFileSync('input.png');\nconst result = await imagesModels.generateImages(model, {\n  input: [\n    { type: 'text', text: 'Create a variation of this image with a blue background.' },\n    { type: 'image', data: imageBuffer.toString('base64'), mimeType: 'image/png' }\n  ]\n});\n```\n\nCheck capabilities on the model metadata:\n\n```typescript\nconsole.log(model.input);   // ['text', 'image']\nconsole.log(model.output);  // ['image'] or ['image', 'text']\n```\n\n### Notes and Limitations\n\n- Image models live in `ImagesModels` collections, chat models in `Models` collections; the two are separate surfaces.\n- Use `generateImages()`, not the chat/stream APIs.\n- Image-generation models do not participate in tool calling.\n- Outputs are returned in `AssistantImages.output` and can include both base64-encoded `ImageContent` blocks and `TextContent` blocks.\n- Some models return only images, others return images plus text. Check `model.output`.\n- Some models accept image input, others are text-to-image only. Check `model.input`.\n- Like the streaming APIs, image generation supports options such as `apiKey`, `signal`, `headers`, `onPayload`, and `onResponse`, and results may include `stopReason`, `responseId`, and `usage`.\n- If you want a model to analyze images in a conversation or call tools, use the regular chat APIs with a model that supports image input.\n- At the moment, image generation is available through only one provider, OpenRouter.\n\n## Thinking/Reasoning\n\nMany models support thinking/reasoning capabilities where they can show their internal thought process. You can check if a model supports reasoning via the `reasoning` property. If you pass reasoning options to a non-reasoning model, they are silently ignored.\n\n### Unified Interface (streamSimple/completeSimple)\n\n```typescript\n// Many models across providers support thinking/reasoning\nconst model = models.getModel('anthropic', 'claude-sonnet-4-5')!;\n// or models.getModel('openai', 'gpt-5-mini');\n// or models.getModel('google', 'gemini-2.5-flash');\n// or models.getModel('xai', 'grok-4.5');\n\n// Check if model supports reasoning\nif (model.reasoning) {\n  console.log('Model supports reasoning/thinking');\n}\n\n// Use the simplified reasoning option\nconst response = await models.completeSimple(model, {\n  messages: [{ role: 'user', content: 'Solve: 2x + 5 = 13', timestamp: Date.now() }]\n}, {\n  reasoning: 'medium'  // 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max'\n});\n\n// Access thinking and text blocks\nfor (const block of response.content) {\n  if (block.type === 'thinking') {\n    console.log('Thinking:', block.thinking);\n  } else if (block.type === 'text') {\n    console.log('Response:', block.text);\n  }\n}\n```\n\n`xhigh` and `max` are model-specific, opt-in levels. Use `getSupportedThinkingLevels(model)` to determine whether a concrete model exposes either level; models such as GPT-5.6 can expose both.\n\n### Provider-Specific Options (stream/complete)\n\n`models.stream()`/`complete()` accept the owning API's full option set. Use `hasApi()` to narrow a dynamically looked-up model to its API for full option typing:\n\n```typescript\nimport { hasApi } from '@aos-agent/ai';\n\n// OpenAI Reasoning (o1, o3, gpt-5)\nconst openaiModel = models.getModel('openai', 'gpt-5-mini')!;\nif (hasApi(openaiModel, 'openai-responses')) {\n  await models.complete(openaiModel, context, {\n    reasoningEffort: 'medium',\n    reasoningSummary: 'detailed'  // OpenAI Responses API only\n  });\n}\n\n// Anthropic Thinking\nconst anthropicModel = models.getModel('anthropic', 'claude-sonnet-4-5')!;\nif (hasApi(anthropicModel, 'anthropic-messages')) {\n  await models.complete(anthropicModel, context, {\n    thinkingEnabled: true,\n    thinkingBudgetTokens: 8192  // Optional token limit\n  });\n}\n\n// Google Gemini Thinking\nconst googleModel = models.getModel('google', 'gemini-2.5-flash')!;\nif (hasApi(googleModel, 'google-generative-ai')) {\n  await models.complete(googleModel, context, {\n    thinking: {\n      enabled: true,\n      budgetTokens: 8192  // -1 for dynamic, 0 to disable\n    }\n  });\n}\n```\n\n### Streaming Thinking Content\n\nWhen streaming, thinking content is delivered through specific events:\n\n```typescript\nconst s = models.streamSimple(model, context, { reasoning: 'high' });\n\nfor await (const event of s) {\n  switch (event.type) {\n    case 'thinking_start':\n      console.log('[Model started thinking]');\n      break;\n    case 'thinking_delta':\n      process.stdout.write(event.delta);  // Stream thinking content\n      break;\n    case 'thinking_end':\n      console.log('\\n[Thinking complete]');\n      break;\n  }\n}\n```\n\n## Stop Reasons\n\nEvery `AssistantMessage` includes a `stopReason` field that indicates how the generation ended:\n\n- `\"pending\"` - Only present in partial messages when we do not know what the stop reason will be\n- `\"stop\"` - This is the final message the model will produce this turn\n- `\"length\"` - Output hit the maximum token limit\n- `\"toolUse\"` - Model is calling tools and expects tool results\n- `\"error\"` - An error occurred during generation\n- `\"aborted\"` - Request was cancelled via abort signal\n\n`AssistantMessage` may also include `responseId`, a provider-specific upstream response or message identifier when the underlying API exposes one. Do not assume it is always present across providers.\n\n## Error Handling\n\nRequest failures never throw out of the stream functions: when a request ends with an error (including aborts and tool call validation errors), the streaming API emits an error event and the final message carries the details:\n\n```typescript\n// In streaming\nfor await (const event of s) {\n  if (event.type === 'error') {\n    // event.reason is either \"error\" or \"aborted\"\n    // event.error is the AssistantMessage with partial content\n    console.error(`Error (${event.reason}):`, event.error.errorMessage);\n    console.log('Partial content:', event.error.content);\n  }\n}\n\n// The final message will have the error details\nconst message = await s.result();\nif (message.stopReason === 'error' || message.stopReason === 'aborted') {\n  console.error('Request failed:', message.errorMessage);\n  // message.content contains any partial content received before the error\n  // message.usage contains partial token counts and costs\n}\n```\n\nAuth failures (no key configured, OAuth refresh failed, unknown provider) surface the same way: as a stream error with `stopReason: \"error\"`.\n\n### Aborting Requests\n\nThe abort signal allows you to cancel in-progress requests. Aborted requests have `stopReason === 'aborted'`:\n\n```typescript\nconst controller = new AbortController();\n\n// Abort after 2 seconds\nsetTimeout(() => controller.abort(), 2000);\n\nconst s = models.stream(model, {\n  messages: [{ role: 'user', content: 'Write a long story', timestamp: Date.now() }]\n}, {\n  signal: controller.signal\n});\n\nfor await (const event of s) {\n  if (event.type === 'text_delta') {\n    process.stdout.write(event.delta);\n  } else if (event.type === 'error') {\n    // event.reason tells you if it was \"error\" or \"aborted\"\n    console.log(`${event.reason === 'aborted' ? 'Aborted' : 'Error'}:`, event.error.errorMessage);\n  }\n}\n\n// Get results (may be partial if aborted)\nconst response = await s.result();\nif (response.stopReason === 'aborted') {\n  console.log('Request was aborted:', response.errorMessage);\n  console.log('Partial content received:', response.content);\n  console.log('Tokens used:', response.usage);\n}\n```\n\n### Continuing After Abort\n\nAborted messages can be added to the conversation context and continued in subsequent requests:\n\n```typescript\nconst context = {\n  messages: [\n    { role: 'user', content: 'Explain quantum computing in detail', timestamp: Date.now() }\n  ]\n};\n\n// First request gets aborted after 2 seconds\nconst controller1 = new AbortController();\nsetTimeout(() => controller1.abort(), 2000);\n\nconst partial = await models.complete(model, context, { signal: controller1.signal });\n\n// Add the partial response to context\ncontext.messages.push(partial);\ncontext.messages.push({ role: 'user', content: 'Please continue', timestamp: Date.now() });\n\n// Continue the conversation\nconst continuation = await models.complete(model, context);\n```\n\n### Debugging Provider Payloads\n\nUse the `onPayload` callback to inspect the request payload sent to the provider. This is useful for debugging request formatting issues or provider validation errors.\n\n```typescript\nconst response = await models.complete(model, context, {\n  onPayload: (payload) => {\n    console.log('Provider payload:', JSON.stringify(payload, null, 2));\n  }\n});\n```\n\nThe callback is supported by `stream`, `complete`, `streamSimple`, and `completeSimple`.\n\n## Custom Providers\n\n### createProvider()\n\n`createProvider()` builds a provider from parts: identity, auth, a model list, and an API implementation. Use it for local inference servers, proxies, or any OpenAI/Anthropic-compatible endpoint:\n\n```typescript\nimport { createModels, createProvider, envApiKeyAuth, type Model } from '@aos-agent/ai';\nimport { openAICompletionsApi } from '@aos-agent/ai/api/openai-completions.lazy';\n\nconst ollamaModel: Model<'openai-completions'> = {\n  id: 'llama-3.1-8b',\n  name: 'Llama 3.1 8B (Ollama)',\n  api: 'openai-completions',\n  provider: 'ollama',\n  baseUrl: 'http://localhost:11434/v1',\n  reasoning: false,\n  input: ['text'],\n  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },\n  contextWindow: 128000,\n  maxTokens: 32000\n};\n\nconst ollama = createProvider({\n  id: 'ollama',\n  name: 'Ollama',\n  baseUrl: 'http://localhost:11434/v1',\n  // Every provider declares auth; keyless local servers resolve as configured with no key.\n  auth: { apiKey: { name: 'Ollama', resolve: async () => ({ auth: {} }) } },\n  models: [ollamaModel],\n  api: openAICompletionsApi(),\n});\n\nconst models = createModels();\nmodels.setProvider(ollama);\n\nawait models.complete(models.getModel('ollama', 'llama-3.1-8b')!, context);\n```\n\nFor providers with real keys, `envApiKeyAuth(displayName, envVars)` gives the standard behavior (stored credential wins, then the first set env var):\n\n```typescript\nconst proxy = createProvider({\n  id: 'my-proxy',\n  auth: { apiKey: envApiKeyAuth('My proxy API key', ['MY_PROXY_API_KEY']) },\n  models: [/* ... */],\n  api: openAICompletionsApi(),\n});\n```\n\nMixed-API providers pass a map keyed by `model.api`; each model dispatches to its API's implementation:\n\n```typescript\nimport { anthropicMessagesApi } from '@aos-agent/ai/api/anthropic-messages.lazy';\nimport { openAIResponsesApi } from '@aos-agent/ai/api/openai-responses.lazy';\n\nconst gateway = createProvider({\n  id: 'my-gateway',\n  auth: { apiKey: envApiKeyAuth('Gateway key', ['GATEWAY_API_KEY']) },\n  models: [/* models with api: 'anthropic-messages' or 'openai-responses' */],\n  api: {\n    'anthropic-messages': anthropicMessagesApi(),\n    'openai-responses': openAIResponsesApi(),\n  },\n});\n```\n\nProvider-wide endpoint or request transformations belong in the provider's API implementation: wrap the `ProviderStreams` you pass as `api` so every request goes through the transformation before dispatch. The Cloudflare providers do this to materialize account/gateway endpoint placeholders from the resolved provider env:\n\n```typescript\nfunction tenantStreams(streams: ProviderStreams): ProviderStreams {\n  const withTenant = (model: Model<Api>) => ({ ...model, baseUrl: model.baseUrl.replace('{tenant}', tenantId) });\n  return {\n    stream: (model, context, options) => streams.stream(withTenant(model), context, options),\n    streamSimple: (model, context, options) => streams.streamSimple(withTenant(model), context, options),\n  };\n}\n\nconst tenantGateway = createProvider({\n  id: 'tenant-gateway',\n  auth: { apiKey: envApiKeyAuth('Gateway key', ['GATEWAY_API_KEY']) },\n  models: [/* ... */],\n  api: tenantStreams(openAICompletionsApi()),\n});\n```\n\nDynamic model lists use `fetchModels`. `Models.refresh()` refreshes every configured dynamic provider, passing its effective API-key or refreshed OAuth credential. A `ModelsStore` persists dynamic catalogs; both stores default to in-memory implementations. Its `read`, `write`, and `delete` operations accept optional cancellation, and `Models` binds those waits to the provider refresh signal.\n\n```typescript\nconst models = createModels({ credentials, modelsStore });\nconst llamacpp = createProvider({\n  id: 'llamacpp',\n  auth: { apiKey: { name: 'llama.cpp', resolve: async () => ({ auth: {} }) } },\n  models: [],\n  fetchModels: async ({ signal }) => fetchModelsFromServer('http://localhost:8080', signal),\n  api: openAICompletionsApi(),\n});\n\nmodels.setProvider(llamacpp);\nconst result = await models.refresh({ signal });\nif (result.aborted) console.log('refresh cancelled');\nfor (const [provider, error] of result.errors) console.error(provider, error);\n```\n\n`Models.refresh()` is unbounded when its optional signal is omitted. Providers always receive a concrete `RefreshModelsContext.signal` and must honor it for network requests and other blocking work. When a caller supplies a signal, `Models.refresh()` returns promptly with `aborted: true` after cancellation even if a custom provider fails to cooperate; the provider must still honor the signal to stop its underlying work.\n\nUse `models.refresh({ providers: ['openrouter'] })` to restrict work to selected providers, `models.refresh({ allowNetwork: false })` to restore persisted catalogs without network access, or `models.refresh({ force: true })` to bypass provider freshness checks. Model reads stay synchronous and return the last restored or refreshed list.\n\n`createProvider()` handles dynamic publication and persistence automatically. Handwritten `Provider.refreshModels()` implementations receive the read-only `context.stored` snapshot and publish through `context.publish({ persist?, update? })`. Omit `persist` to leave storage unchanged, pass a `ModelsStoreEntry` to write it, or pass `persist: null` to delete it. Publication is generation-checked; put synchronous in-memory catalog changes in `update` rather than mutating state before publication.\n\nCustom models can carry `headers` (e.g. proxies behind bot detection) and `compat` flags. `Models.getAuth(model)` includes those model headers, and stream methods merge them before explicit request headers and `transformHeaders`. See [OpenAI Compatibility Settings](#openai-compatibility-settings).\n\nSome OpenAI-compatible servers do not understand the `developer` role used for reasoning-capable models. For those providers, set `compat.supportsDeveloperRole` to `false` so the system prompt is sent as a `system` message instead. If the server also does not support `reasoning_effort`, set `compat.supportsReasoningEffort` to `false` too. This commonly applies to Ollama, vLLM, SGLang, and similar OpenAI-compatible servers.\n\nUse model-level `thinkingLevelMap` to describe model-specific thinking controls. Keys are AOS Agent thinking levels (`off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`). Missing standard levels through `high` use provider defaults; `xhigh` and `max` are opt-in and require a non-null map entry. String values are sent to the provider, `null` marks a level unsupported, and maps may skip levels.\n\n```typescript\nconst ollamaReasoningModel: Model<'openai-completions'> = {\n  id: 'gpt-oss:20b',\n  name: 'GPT-OSS 20B (Ollama)',\n  api: 'openai-completions',\n  provider: 'ollama',\n  baseUrl: 'http://localhost:11434/v1',\n  reasoning: true,\n  input: ['text'],\n  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },\n  contextWindow: 131072,\n  maxTokens: 32000,\n  thinkingLevelMap: {\n    minimal: null,\n    low: null,\n    medium: null,\n    high: 'high',\n    xhigh: null,\n  },\n  compat: {\n    supportsDeveloperRole: false,\n    supportsReasoningEffort: false,\n  }\n};\n```\n\n### Calling API Implementations Directly\n\nThe API implementations are importable on their own. Each module exports exactly `stream` and `streamSimple` with that API's full option typing. Direct calls bypass provider auth — pass `apiKey` explicitly:\n\n```typescript\nimport { stream } from '@aos-agent/ai/api/anthropic-messages';\n\nconst s = stream(claudeModel, context, {\n  apiKey: process.env.ANTHROPIC_API_KEY,\n  thinkingEnabled: true,\n  thinkingBudgetTokens: 2048,\n});\n```\n\nBuilt-in API implementations live under `./api/<api-id>`:\n\n| API id | Options type |\n|--------|--------------|\n| `anthropic-messages` | `AnthropicOptions` |\n| `openai-completions` | `OpenAICompletionsOptions` |\n| `openai-responses` | `OpenAIResponsesOptions` |\n| `openai-codex-responses` | `OpenAICodexResponsesOptions` |\n| `azure-openai-responses` | `AzureOpenAIResponsesOptions` |\n| `google-generative-ai` | `GoogleOptions` |\n| `google-vertex` | `GoogleVertexOptions` |\n| `mistral-conversations` | `MistralOptions` |\n| `bedrock-converse-stream` | `BedrockOptions` |\n\nImporting an implementation module loads its SDK. The `./api/<id>.lazy` wrappers (used by the provider factories) defer that load to the first request when the runtime or bundler supports dynamic import chunking. Legacy raw API subpaths from older releases (`./anthropic`, `./google`, `./mistral`, `./openai-completions`, ...) were removed; use `@aos-agent/ai/api/<api-id>`.\n\n### OpenAI Compatibility Settings\n\nThe `openai-completions` API is implemented by many providers with minor differences. By default, the library auto-detects compatibility settings based on `baseUrl` for a small set of known OpenAI-compatible providers (Cerebras, xAI, Chutes, DeepSeek, NVIDIA NIM, Together AI, zAi, OpenCode, Cloudflare Workers AI, etc.). For custom proxies or unknown endpoints, you can override these settings via the `compat` field. For `openai-responses` models, the compat field supports Responses-specific flags.\n\n```typescript\ninterface OpenAICompletionsCompat {\n  supportsStore?: boolean;           // Whether provider supports the `store` field (default: true)\n  supportsDeveloperRole?: boolean;   // Whether provider supports `developer` role vs `system` (default: true)\n  supportsReasoningEffort?: boolean; // Whether provider supports `reasoning_effort` (default: true)\n  supportsUsageInStreaming?: boolean; // Whether provider supports `stream_options: { include_usage: true }` (default: true)\n  supportsStrictMode?: boolean;      // Whether provider supports `strict` in tool definitions (default: true)\n  supportsOpenAIGrammarTools?: boolean; // Whether to emit OpenAI custom Lark/regex grammar tools; false falls back to normal function tools (default: false; the generated catalog enables it for capable models)\n  sendSessionAffinityHeaders?: boolean; // Send session-affinity data from `sessionId` (default: false)\n  sessionAffinityFormat?: 'openai' | 'openai-nosession' | 'openrouter'; // Format for session affinity: 'openai' uses `prompt_cache_key`, `session_id`, `x-client-request-id`, and `x-session-affinity`; 'openai-nosession' uses `prompt_cache_key`, `x-client-request-id`, and `x-session-affinity`; 'openrouter' uses `x-session-id` (default: auto-detected)\n  maxTokensField?: 'max_completion_tokens' | 'max_tokens';  // Which field name to use (default: max_completion_tokens)\n  requiresToolResultName?: boolean;  // Whether tool results require the `name` field (default: false)\n  requiresAssistantAfterToolResult?: boolean; // Whether tool results must be followed by an assistant message (default: false)\n  requiresThinkingAsText?: boolean;  // Whether thinking blocks must be converted to text (default: false)\n  requiresReasoningContentOnAssistantMessages?: boolean; // Whether all replayed assistant messages must include empty reasoning_content when reasoning is enabled (default: auto-detected for DeepSeek)\n  thinkingFormat?: 'openai' | 'openrouter' | 'deepseek' | 'together' | 'baseten' | 'zai' | 'qwen' | 'chat-template' | 'qwen-chat-template' | 'string-thinking' | 'ant-ling'; // Format for reasoning param: 'openai' uses reasoning_effort, 'openrouter' uses reasoning: { effort }, 'deepseek' uses thinking: { type } plus reasoning_effort when supported, 'together' uses reasoning: { enabled } plus reasoning_effort when supported, 'baseten' uses configurable chat_template_args plus reasoning_effort when supported, 'zai' uses thinking: { type }, 'qwen' uses enable_thinking, 'chat-template' uses configurable chat_template_kwargs, 'qwen-chat-template' uses chat_template_kwargs.enable_thinking and preserve_thinking, 'string-thinking' uses top-level thinking, 'ant-ling' uses reasoning: { effort } only for mapped efforts (default: openai)\n  chatTemplateKwargs?: Record<string, string | number | boolean | null | { '$var': 'thinking.enabled' | 'thinking.effort'; omitWhenOff?: boolean }>; // chat_template_kwargs values; use $var for AOS Agent-controlled thinking values\n  chatTemplateArgs?: Record<string, string | number | boolean | null | { '$var': 'thinking.enabled' | 'thinking.effort'; omitWhenOff?: boolean }>; // chat_template_args values for thinkingFormat: 'baseten'; use $var for AOS Agent-controlled thinking values\n  cacheControlFormat?: 'anthropic';  // Anthropic-style cache_control on system prompt, last tool, and last user/assistant text content\n  openRouterRouting?: OpenRouterRouting; // OpenRouter routing preferences (default: {})\n  vercelGatewayRouting?: VercelGatewayRouting; // Vercel AI Gateway routing preferences (default: {})\n}\n\ninterface OpenAIResponsesCompat {\n  supportsDeveloperRole?: boolean;   // Whether provider supports `developer` role vs `system` (default: true)\n  sessionAffinityFormat?: 'openai' | 'openai-nosession' | 'openrouter'; // Session-affinity header format: 'openai' sends `session_id` and `x-client-request-id`; 'openai-nosession' sends `x-client-request-id`; 'openrouter' sends `x-session-id`. Does not affect the `prompt_cache_key` body param (default: auto-detected)\n  supportsLongCacheRetention?: boolean; // Whether provider supports `prompt_cache_retention: \"24h\"` (default: true)\n  supportsStrictMode?: boolean;      // Whether provider supports strict JSON-schema function tools (default: false; enabled in metadata for built-in OpenAI models)\n  supportsOpenAIGrammarTools?: boolean; // Whether to emit OpenAI custom Lark/regex grammar tools; false falls back to normal function tools (default: false; the generated catalog enables it for capable models)\n}\n```\n\nIf `compat` is not set, the library falls back to URL-based detection. If `compat` is partially set, unspecified fields use the detected defaults. This is useful for:\n\n- **LiteLLM proxies**: May not support `store` field\n- **Custom inference servers**: May use non-standard field names\n- **Self-hosted endpoints**: May have different feature support\n\n## Fake Provider for Tests\n\n`fakeProvider()` builds an in-memory provider with scripted responses for tests and demos:\n\n```typescript\nimport {\n  createModels,\n  fakeAssistantMessage,\n  fakeProvider,\n  fakeText,\n  fakeThinking,\n  fakeToolCall,\n} from '@aos-agent/ai';\n\nconst fake = fakeProvider({\n  tokensPerSecond: 50 // optional\n});\n\nconst models = createModels();\nmodels.setProvider(fake.provider);\n\nconst model = fake.getModel();\nconst context = {\n  messages: [{ role: 'user', content: 'Summarize package.json and then call echo', timestamp: Date.now() }]\n};\n\nfake.setResponses([\n  fakeAssistantMessage([\n    fakeThinking('Need to inspect package metadata first.'),\n    fakeToolCall('echo', { text: 'package.json' })\n  ], { stopReason: 'toolUse' })\n]);\n\nconst first = await models.complete(model, context, {\n  sessionId: 'session-1',\n  cacheRetention: 'short'\n});\ncontext.messages.push(first);\n\ncontext.messages.push({\n  role: 'toolResult',\n  toolCallId: first.content.find((block) => block.type === 'toolCall')!.id,\n  toolName: 'echo',\n  content: [{ type: 'text', text: 'package.json contents here' }],\n  isError: false,\n  timestamp: Date.now()\n});\n\nfake.setResponses([\n  fakeAssistantMessage([\n    fakeThinking('Now I can summarize the tool output.'),\n    fakeText('Here is the summary.')\n  ])\n]);\n\nconst s = models.stream(model, context);\nfor await (const event of s) {\n  console.log(event.type);\n}\n\n// Optional: multiple fake models for model-switching tests\nconst multiModel = fakeProvider({\n  provider: 'fake-multi',\n  models: [\n    { id: 'fake-fast', reasoning: false },\n    { id: 'fake-thinker', reasoning: true }\n  ]\n});\nmodels.setProvider(multiModel.provider);\nconst thinker = multiModel.getModel('fake-thinker');\n\nconsole.log(thinker?.reasoning);\nconsole.log(fake.getPendingResponseCount());\nconsole.log(fake.state.callCount);\n```\n\nNotes:\n- Responses are consumed from a queue in request start order.\n- If the queue is empty, the fake provider returns an assistant error message with `errorMessage: \"No more fake responses queued\"`.\n- Use `fake.setResponses([...])` to replace the remaining queue and `fake.appendResponses([...])` to add more responses.\n- `fake.models` exposes all fake models. `fake.getModel()` returns the first one, and `fake.getModel(id)` returns a specific one.\n- Use `fakeAssistantMessage(...)` for scripted assistant replies. Use `fakeText(...)`, `fakeThinking(...)`, and `fakeToolCall(...)` to build content blocks without filling in low-level fields manually.\n- Usage is estimated at roughly 1 token per 4 characters. When `sessionId` is present and `cacheRetention` is not `\"none\"`, prompt cache reads and writes are simulated automatically.\n- Tool call arguments stream incrementally via `toolcall_delta` chunks.\n- By default, each streamed chunk is emitted on its own microtask. Set `tokensPerSecond` to pace chunk delivery in real time.\n- The intended use is one deterministic scripted flow per handle. If you need independent concurrent flows, create separate fake providers with distinct `provider` ids.\n\n## Cross-Provider Handoffs\n\nThe library supports seamless handoffs between different LLM providers within the same conversation. This allows you to switch models mid-conversation while preserving context, including thinking blocks, tool calls, and tool results.\n\nWhen messages from one provider are sent to a different provider, the library automatically transforms them for compatibility:\n\n- **User and tool result messages** are passed through unchanged\n- **Assistant messages from the same provider/API** are preserved as-is\n- **Assistant messages from different providers** have their thinking blocks converted to text with `<thinking>` tags\n- **Tool calls and regular text** are preserved unchanged\n\n```typescript\nimport { createModels, type Context } from '@aos-agent/ai';\nimport { anthropicProvider } from '@aos-agent/ai/providers/anthropic';\nimport { openaiProvider } from '@aos-agent/ai/providers/openai';\nimport { googleProvider } from '@aos-agent/ai/providers/google';\n\nconst models = createModels();\nmodels.setProvider(anthropicProvider());\nmodels.setProvider(openaiProvider());\nmodels.setProvider(googleProvider());\n\nconst context: Context = { messages: [] };\n\n// Start with Claude\nconst claude = models.getModel('anthropic', 'claude-sonnet-4-5')!;\ncontext.messages.push({ role: 'user', content: 'What is 25 * 18?', timestamp: Date.now() });\ncontext.messages.push(await models.completeSimple(claude, context, { reasoning: 'medium' }));\n\n// Switch to GPT-5 - it will see Claude's thinking as <thinking> tagged text\nconst gpt5 = models.getModel('openai', 'gpt-5-mini')!;\ncontext.messages.push({ role: 'user', content: 'Is that calculation correct?', timestamp: Date.now() });\ncontext.messages.push(await models.complete(gpt5, context));\n\n// Switch to Gemini\nconst gemini = models.getModel('google', 'gemini-2.5-flash')!;\ncontext.messages.push({ role: 'user', content: 'What was the original question?', timestamp: Date.now() });\nconst geminiResponse = await models.complete(gemini, context);\n```\n\nAll providers can handle messages from other providers — text, tool calls and results (including images), thinking blocks (transformed to tagged text), and aborted messages with partial content. This enables flexible workflows: start with a fast model, switch to a more capable one for complex reasoning, or maintain continuity across provider outages.\n\n## Context Serialization\n\nThe `Context` object can be easily serialized and deserialized using standard JSON methods, making it simple to persist conversations, implement chat history, or transfer contexts between services:\n\n```typescript\nconst context: Context = {\n  systemPrompt: 'You are a helpful assistant.',\n  messages: [\n    { role: 'user', content: 'What is TypeScript?', timestamp: Date.now() }\n  ]\n};\n\nconst model = models.getModel('openai', 'gpt-4o-mini')!;\nconst response = await models.complete(model, context);\ncontext.messages.push(response);\n\n// Serialize the entire context\nconst serialized = JSON.stringify(context);\n\n// Save to database, localStorage, file, etc.\nlocalStorage.setItem('conversation', serialized);\n\n// Later: deserialize and continue the conversation\nconst restored: Context = JSON.parse(localStorage.getItem('conversation')!);\nrestored.messages.push({ role: 'user', content: 'Tell me more about its type system', timestamp: Date.now() });\n\n// Continue with any model\nconst newModel = models.getModel('anthropic', 'claude-3-5-haiku-20241022')!;\nconst continuation = await models.complete(newModel, restored);\n```\n\nModels are plain serializable data too — no functions or implementations attached — so persisting \"which model was this conversation using\" is a `JSON.stringify` away.\n\n> **Note**: If the context contains images (encoded as base64 as shown in the Image Input section), those will also be serialized.\n\n## Browser Usage\n\nThe library supports browser environments. The core entrypoint and provider factories are side-effect free and bundle cleanly. Environment variables are not available in browsers, so pass API keys explicitly — or inject a `CredentialStore` (e.g. localStorage-backed) and let provider auth resolve from stored credentials:\n\n```typescript\nimport { createModels } from '@aos-agent/ai';\nimport { anthropicProvider } from '@aos-agent/ai/providers/anthropic';\n\nconst models = createModels();\nmodels.setProvider(anthropicProvider());\n\nconst model = models.getModel('anthropic', 'claude-3-5-haiku-20241022')!;\nconst response = await models.complete(model, {\n  messages: [{ role: 'user', content: 'Hello!', timestamp: Date.now() }]\n}, {\n  apiKey: 'your-api-key'\n});\n```\n\n> **Security Warning**: Exposing API keys in frontend code is dangerous. Anyone can extract and abuse your keys. Only use this approach for internal tools or demos. For production applications, use a backend proxy that keeps your API keys secure.\n\nBrowser compatibility notes:\n\n- Amazon Bedrock (`bedrock-converse-stream`) is not supported in browser environments. It can still appear in model lists; calls fail at runtime.\n- OAuth login flows are Node-only. They are lazy-loaded behind bundler-opaque imports, so registering an OAuth-capable provider does not pull Node-only code into a browser bundle — only actually logging in would.\n- Use a server-side proxy or backend service if you need Bedrock or OAuth-based auth from a web app.\n\n## Bundling and Tree Shaking\n\nFor small bundles, import only the providers you need:\n\n```typescript\nimport { createModels } from '@aos-agent/ai';\nimport { openaiProvider } from '@aos-agent/ai/providers/openai';\n\nconst mo","readmeFilename":"README.md"}