{"_id":"@alexplusplus/llm-fallback-chain","_rev":"3-07c285dbe9c2d6b896e891c92236ba82","name":"@alexplusplus/llm-fallback-chain","dist-tags":{"latest":"0.2.0"},"versions":{"0.1.0":{"name":"@alexplusplus/llm-fallback-chain","version":"0.1.0","keywords":["llm","fallback","structured-output","zod","gemini","openai","openrouter","json-schema"],"license":"MIT","_id":"@alexplusplus/llm-fallback-chain@0.1.0","maintainers":[{"name":"alexplusplus","email":"alexanderekb@gmail.com"}],"dist":{"shasum":"03da45cc445253fa60dcd7ead35e39b9bf42febd","tarball":"https://registry.npmjs.org/@alexplusplus/llm-fallback-chain/-/llm-fallback-chain-0.1.0.tgz","fileCount":6,"integrity":"sha512-J9INaeCvNXAIHISG/qcLpLOhM3ziX2E9RTzEULJfASS88UwPFWvIp42SARLaOT2LgK02EUvjOhNn0HovsvPmnA==","signatures":[{"sig":"MEQCIEJdfE1hjeFPUGFI+qXSzmgXBMajDFpGQ0SnK4/2P36MAiBDEnnOld8qC9/ao2EEQ9RjtoUA6b7rZJG4n3A5guzdPg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":102027},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"644ac11e20fa5f0736d7c8a72bcb8f6a985ccf08","scripts":{"test":"vitest run","build":"tsup","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"npm run typecheck && npm run test && npm run build"},"_npmUser":{"name":"alexplusplus","email":"alexanderekb@gmail.com"},"_npmVersion":"11.7.0","description":"Structured LLM output through a configurable provider fallback chain: free tiers first, paid floor last, with pluggable cooldown storage.","directories":{},"sideEffects":false,"_nodeVersion":"24.12.0","dependencies":{"openai":"^6.45.0","@google/genai":"^2.10.0"},"_hasShrinkwrap":false,"devDependencies":{"zod":"^4.4.3","tsup":"^8.5.1","vitest":"^4.1.9","typescript":"^6.0.3","@types/node":"^26.1.0"},"peerDependencies":{"zod":"^4.0.0"},"_npmOperationalInternal":{"tmp":"tmp/llm-fallback-chain_0.1.0_1783012757459_0.4516054536819618","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@alexplusplus/llm-fallback-chain","version":"0.1.1","keywords":["llm","fallback","structured-output","zod","gemini","openai","openrouter","json-schema"],"license":"MIT","_id":"@alexplusplus/llm-fallback-chain@0.1.1","maintainers":[{"name":"alexplusplus","email":"alexanderekb@gmail.com"}],"dist":{"shasum":"6a9f588425f9019b650c0f25745c9ee3b242f8bc","tarball":"https://registry.npmjs.org/@alexplusplus/llm-fallback-chain/-/llm-fallback-chain-0.1.1.tgz","fileCount":7,"integrity":"sha512-PP89VQ+jewSEcCtF52ozdBiKmHIIuANGQweopIHCexSkB0q3AbCRt8FqSND8SuqqwueXLwmwqF4zFHdXURYavg==","signatures":[{"sig":"MEYCIQDyPn5TzDtRDn6ks/J4O07QG/Q/kh0ixkTbeENqEmtx2gIhANyvZYthKzEqZj6tqKs/ucOaWvdGp0fUhiRYJv2/FVB/","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":112035},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","default":"./dist/index.js"}},"gitHead":"0e364c16bd3cc7ffaee25a32034dc04c79bcb41f","scripts":{"test":"vitest run","build":"tsup","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"npm run typecheck && npm run test && npm run build"},"_npmUser":{"name":"alexplusplus","email":"alexanderekb@gmail.com"},"_npmVersion":"11.7.0","description":"Structured LLM output through a configurable provider fallback chain: free tiers first, paid floor last, with pluggable cooldown storage.","directories":{},"sideEffects":false,"_nodeVersion":"24.12.0","dependencies":{"openai":"^6.45.0","@google/genai":"^2.10.0"},"_hasShrinkwrap":false,"devDependencies":{"zod":"^4.4.3","tsup":"^8.5.1","vitest":"^4.1.9","typescript":"^6.0.3","@types/node":"^26.1.0"},"peerDependencies":{"zod":"^4.0.0"},"_npmOperationalInternal":{"tmp":"tmp/llm-fallback-chain_0.1.1_1783061013871_0.1666827846060137","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@alexplusplus/llm-fallback-chain","version":"0.2.0","description":"Structured or plain-text LLM output through a configurable provider fallback chain: free tiers first, paid floor last, with pluggable cooldown storage and unified reasoning effort.","keywords":["llm","fallback","structured-output","zod","gemini","openai","openrouter","json-schema"],"license":"MIT","type":"module","main":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","default":"./dist/index.js"}},"sideEffects":false,"scripts":{"build":"tsup","typecheck":"tsc --noEmit","test":"vitest run","test:watch":"vitest","prepublishOnly":"npm run typecheck && npm run test && npm run build"},"engines":{"node":">=20"},"dependencies":{"@google/genai":"^2.10.0","openai":"^6.45.0"},"devDependencies":{"@types/node":"^26.1.0","tsup":"^8.5.1","typescript":"^6.0.3","vitest":"^4.1.9","zod":"^4.4.3"},"peerDependencies":{"zod":"^4.0.0"},"gitHead":"4e7b181a0ca58ab8f329d1bb45e8c090f63edb0d","_id":"@alexplusplus/llm-fallback-chain@0.2.0","_nodeVersion":"24.12.0","_npmVersion":"11.7.0","dist":{"integrity":"sha512-ZYn9IAz5FsBj39lZ0c/lvXdVDNF8ycSfjJYApw317HSUiT9bXCvKmweMyvkJlFEopO4GjzgbDtZdkSDSJUZfBQ==","shasum":"5024457132460f8a687f03cf72d0aee8bedd0608","tarball":"https://registry.npmjs.org/@alexplusplus/llm-fallback-chain/-/llm-fallback-chain-0.2.0.tgz","fileCount":7,"unpackedSize":133138,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIGnwt2jCrf8xdRUuJORyDSi7oWWuTbdV7ixTP+Rrf2rPAiAPX8r5tJyDGHAEez5jCqrOCV5Kpx3aTWrOaDV+PCoq4g=="}]},"_npmUser":{"name":"alexplusplus","email":"alexanderekb@gmail.com"},"directories":{},"maintainers":[{"name":"alexplusplus","email":"alexanderekb@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/llm-fallback-chain_0.2.0_1783075459122_0.5596436321616352"},"_hasShrinkwrap":false}},"time":{"created":"2026-07-02T17:19:17.219Z","modified":"2026-07-03T10:44:19.401Z","0.1.0":"2026-07-02T17:19:17.618Z","0.1.1":"2026-07-03T06:43:34.022Z","0.2.0":"2026-07-03T10:44:19.284Z"},"license":"MIT","keywords":["llm","fallback","structured-output","zod","gemini","openai","openrouter","json-schema"],"description":"Structured or plain-text LLM output through a configurable provider fallback chain: free tiers first, paid floor last, with pluggable cooldown storage and unified reasoning effort.","maintainers":[{"name":"alexplusplus","email":"alexanderekb@gmail.com"}],"readme":"# @alexplusplus/llm-fallback-chain\r\n\r\nStructured or plain-text LLM output through a configurable provider\r\n**fallback chain**: free tiers first, a paid floor last, with pluggable\r\ncooldown storage.\r\n\r\nPass a prompt and a [zod](https://zod.dev) schema for parsed,\r\nTypeScript-typed data — or just a prompt for free-form text; the chain walks\r\nyour entries top-down and returns the result with serving-entry metadata, or\r\na classified error. Quota-exhausted or flaky entries go on **cooldown** and\r\nare skipped until they recover, so a free tier running dry falls through to\r\nthe next entry instead of failing your request.\r\n\r\n```\r\nGemini (free) ──quota──▶ OpenRouter (cheap) ──5xx──▶ OpenAI (floor) ──▶ ✓ typed data\r\n     │                        │\r\n  cooldown until          cooldown 60s\r\n  daily reset\r\n```\r\n\r\n## Install\r\n\r\n```sh\r\nnpm install @alexplusplus/llm-fallback-chain zod\r\n```\r\n\r\n`zod` (v4) is a peer dependency.\r\n\r\nThe package is ESM. From CommonJS projects, `require()` works on Node ≥ 20.19\r\n/ ≥ 22.12 (native `require(esm)`); on older Node use dynamic `import()`.\r\n\r\n## Quickstart\r\n\r\n```ts\r\nimport { z } from \"zod\";\r\nimport {\r\n  createFallbackChain,\r\n  GeminiAdapter,\r\n  OpenAiAdapter,\r\n  OpenRouterAdapter,\r\n} from \"@alexplusplus/llm-fallback-chain\";\r\n\r\nconst chain = createFallbackChain({\r\n  entries: [\r\n    {\r\n      key: \"gemini-flash\",\r\n      adapter: new GeminiAdapter({ apiKey: process.env.GEMINI_API_KEY! }),\r\n      modelId: \"gemini-2.5-flash\",\r\n    },\r\n    {\r\n      key: \"openrouter-llama\",\r\n      adapter: new OpenRouterAdapter({ apiKey: process.env.OPENROUTER_API_KEY! }),\r\n      modelId: \"meta-llama/llama-3.3-70b-instruct\",\r\n    },\r\n    {\r\n      key: \"openai-mini\",\r\n      adapter: new OpenAiAdapter({ apiKey: process.env.OPENAI_API_KEY! }),\r\n      modelId: \"gpt-4o-mini\",\r\n    },\r\n  ],\r\n});\r\n\r\nconst WordSet = z.object({\r\n  paragraphs: z.array(z.string()).describe(\"Three short example texts\"),\r\n  definitions: z.string(),\r\n  word_forms: z.array(z.object({ word: z.string(), forms: z.array(z.string()) })),\r\n});\r\n\r\nconst { data, entry } = await chain.generate({\r\n  prompt: \"Generate study material for: bank, spring, light\",\r\n  schema: WordSet,\r\n  schemaName: \"word_set\",\r\n});\r\n\r\ndata.paragraphs; // string[] — fully typed via z.infer\r\nconsole.log(`served by ${entry.key} (${entry.providerId}/${entry.modelId})`);\r\n```\r\n\r\nLog `entry` on every request: it tells you which chain position served it,\r\nwhich is how you notice free-tier utilization dropping (cost drift) early.\r\n\r\n## Plain-text quickstart\r\n\r\nOmit `schema` and the same chain generates free-form text — full fallback and\r\ncooldown machinery, no JSON parsing or validation:\r\n\r\n```ts\r\nconst { text, entry, failures } = await chain.generate({\r\n  prompt: \"Explain the difference between 'bank' and 'shore' in one paragraph.\",\r\n});\r\n\r\ntext; // string — the response verbatim (not trimmed)\r\n```\r\n\r\nThe portable schema subset and the native-schema-enforcement requirement\r\n([ADR 0002](docs/adr/0002-native-schema-only.md)) apply to **structured mode\r\nonly**. In plain-text mode, chain entries don't need schema-capable models —\r\nOpenRouter `:free` variants that are disqualified for structured use are\r\neligible in plain-text chains.\r\n\r\nA whitespace-only response is treated like failed schema validation:\r\nshort cooldown, fall through to the next entry (`reason: \"invalid-output\"`\r\nin `failures`).\r\n\r\n## Reasoning effort\r\n\r\nBoth modes accept an optional `reasoningEffort` — one provider-agnostic value\r\n(OpenRouter's effort vocabulary) that each adapter converts to its provider's\r\ndialect via a hardcoded correspondence\r\n([ADR 0003](docs/adr/0003-unified-reasoning-effort.md)):\r\n\r\n```ts\r\nimport { REASONING_EFFORTS, type ReasoningEffort } from \"@alexplusplus/llm-fallback-chain\";\r\n\r\nawait chain.generate({ prompt, reasoningEffort: \"low\" });\r\n```\r\n\r\n| `reasoningEffort` | OpenRouter `reasoning.effort` | OpenAI `reasoning_effort` | Gemini `thinkingConfig.thinkingBudget` |\r\n| --- | --- | --- | --- |\r\n| `minimal` | `minimal` | `minimal` | `0` |\r\n| `low` | `low` | `low` | `1024` |\r\n| `medium` | `medium` | `medium` | `8192` |\r\n| `high` | `high` | `high` | `16384` |\r\n| `xhigh` | `xhigh` | `xhigh` | `24576` |\r\n\r\n- Omitted → no reasoning-related field is sent to any provider at all.\r\n- A value outside the dictionary (possible from untyped callers or config\r\n  strings — validate yours against the exported `REASONING_EFFORTS` array)\r\n  throws `InvalidRequestError` before any provider is called.\r\n- The conversion is deterministic from the effort value and the serving\r\n  `entry.providerId`, so you can persist \"effort actually sent\" from the\r\n  metadata you already log — no extra result fields.\r\n- Gemini budgets are sized to fit every 2.5-family model range (Flash caps\r\n  at 24576), so one request-level value survives every entry it walks.\r\n\r\n> **⚠️ A model that rejects the reasoning field aborts the whole call.**\r\n> The chain does not pre-filter or retry without the field: a provider\r\n> rejection classifies as `InvalidRequestError`, which fails the entire call\r\n> immediately — **no fallthrough to later entries**. Known case: Gemini\r\n> models that cannot disable thinking (e.g. 2.5 Pro, floor 128) reject\r\n> `minimal` (budget `0`). Choosing reasoning-capable models for every entry\r\n> of a chain that receives `reasoningEffort` is your responsibility.\r\n\r\n## How the chain walks\r\n\r\nFor each entry, top-down:\r\n\r\n| Outcome | Classification | Effect |\r\n| --- | --- | --- |\r\n| Success | — | Structured: response validated against your zod schema. Plain: text returned verbatim. Both carry serving-entry metadata |\r\n| Quota exhausted (429/402) | `QuotaError` | Long cooldown — provider's retry hint, else next UTC midnight — then falls through |\r\n| Transient failure (5xx, timeout, network) | `TransientError` | Short cooldown (default 60 s), falls through |\r\n| Output fails JSON/schema validation (structured) or is whitespace-only (plain) | — | Treated like transient: short cooldown, falls through |\r\n| Bad request (4xx: schema, prompt, API key) | `InvalidRequestError` | **Whole call fails immediately.** No cooldown, no fallthrough — the same bug would fail on every entry, silently burning paid quota |\r\n| Every entry skipped/failed | `ChainExhaustedError` | Carries a per-entry failure list for diagnostics |\r\n\r\nEntries already on cooldown are skipped without a provider call. Cooldowns\r\nare keyed by chain entry and shared across both output modes on the same\r\nchain instance — they represent provider/model quota state, which is\r\nmode-independent.\r\n\r\n## The portable schema subset (structured mode)\r\n\r\nIn structured mode, all entries must be able to enforce your schema natively\r\n(see [ADR 0002](docs/adr/0002-native-schema-only.md)), so schemas are limited to\r\nthe intersection of the Gemini `responseSchema`, OpenAI strict `json_schema`,\r\nand OpenRouter `response_format` dialects:\r\n\r\n- objects — **all fields required**; use `.nullable()`, not `.optional()`\r\n- `z.string()`, `z.number()`, `z.number().int()`, `z.boolean()`\r\n- `z.enum([...])` and string literals\r\n- `z.array(...)` of any of the above\r\n- `.nullable()` on any of the above\r\n- `.describe()` descriptions are forwarded to the provider\r\n\r\nAnything else (unions, records, tuples, dates, recursion) throws\r\n`InvalidRequestError` **before any provider is called**, naming the offending\r\npath. Your zod refinements still run on the response — the subset only limits\r\nwhat is sent to providers, not what you validate.\r\n\r\n## Cooldown stores\r\n\r\nCooldowns are recorded through a two-method interface:\r\n\r\n```ts\r\ninterface CooldownStore {\r\n  mark(entryKey: string, retryAt: Date): Promise<void>;\r\n  check(entryKey: string): Promise<Date | null>; // null = not on cooldown\r\n}\r\n```\r\n\r\nThe default `InMemoryCooldownStore` works out of the box for long-lived\r\nprocesses. On serverless hosts, inject a durable store so a quota discovery\r\non one instance benefits all instances:\r\n\r\n```ts\r\ncreateFallbackChain({ entries, cooldownStore: new FirestoreCooldownStore(db) });\r\n```\r\n\r\n### Entry keys are arbitrary strings — encode them\r\n\r\nChain Entry keys routinely contain `/`, `:` and `.` (e.g. an OpenRouter entry\r\nkeyed `openrouter:meta-llama/llama-3.3-70b-instruct`). Many document stores\r\nforbid these characters in document IDs or treat `/` as a path separator, so\r\na store that uses the raw key as an ID will fail — or worse, fail silently —\r\nin production. Encode the key when deriving the ID. A minimal Firestore\r\nimplementation:\r\n\r\n```ts\r\nimport type { CooldownStore } from \"@alexplusplus/llm-fallback-chain\";\r\nimport type { Firestore } from \"firebase-admin/firestore\";\r\n\r\nexport class FirestoreCooldownStore implements CooldownStore {\r\n  constructor(\r\n    private readonly db: Firestore,\r\n    private readonly collection = \"llmCooldowns\",\r\n  ) {}\r\n\r\n  // Firestore doc IDs cannot contain \"/\" — encode the entry key.\r\n  private doc(entryKey: string) {\r\n    return this.db.collection(this.collection).doc(encodeURIComponent(entryKey));\r\n  }\r\n\r\n  async mark(entryKey: string, retryAt: Date): Promise<void> {\r\n    await this.doc(entryKey).set({ retryAt });\r\n  }\r\n\r\n  async check(entryKey: string): Promise<Date | null> {\r\n    const snap = await this.doc(entryKey).get();\r\n    if (!snap.exists) return null;\r\n    const retryAt: Date = snap.get(\"retryAt\").toDate();\r\n    if (retryAt.getTime() <= Date.now()) {\r\n      await this.doc(entryKey).delete(); // prune expired cooldowns\r\n      return null;\r\n    }\r\n    return retryAt;\r\n  }\r\n}\r\n```\r\n\r\n### Verifying a store implementation\r\n\r\nVerify any implementation against the behavioral contract (framework-agnostic,\r\nworks in any test runner). Since v0.1.1 the contract includes a\r\nslash-containing key, so it catches the document-ID class of bug above:\r\n\r\n```ts\r\nimport { verifyCooldownStoreContract } from \"@alexplusplus/llm-fallback-chain\";\r\n\r\nawait verifyCooldownStoreContract(() => new FirestoreCooldownStore(db));\r\n```\r\n\r\n**Persistent stores need a throwaway collection per run.** The contract suite\r\nuses fixed key names, so state left behind by a previous run violates the\r\n\"unmarked key returns `null`\" assertion. Point each run at a fresh,\r\ndisposable collection (and delete it afterwards, or let a TTL policy expire\r\nit):\r\n\r\n```ts\r\nconst collection = `cooldown-contract-${Date.now()}`;\r\nawait verifyCooldownStoreContract(() => new FirestoreCooldownStore(db, collection));\r\n```\r\n\r\nThe factory is called several times per run; sharing one throwaway collection\r\nacross those instances is fine — the contract's key names don't collide with\r\neach other, only with earlier runs.\r\n\r\n## Deploying on Netlify: environment variables\r\n\r\nA typical serverless deployment (the setup this section describes was proven\r\non Netlify with a Nuxt/Nitro app) needs two groups of environment variables:\r\nprovider API keys for the chain, and Firebase service-account credentials for\r\na Firestore-backed cooldown store. All of them are **server-side secrets** —\r\nnone may ever be exposed to the client bundle (no `NUXT_PUBLIC_` / `VITE_` /\r\n`NEXT_PUBLIC_` prefixes).\r\n\r\n### 1. Provider API keys\r\n\r\nOne per provider that appears in your chain:\r\n\r\n| Variable | Used by | Where to get it |\r\n| --- | --- | --- |\r\n| `GEMINI_API_KEY` | `GeminiAdapter` | [Google AI Studio](https://aistudio.google.com/apikey) |\r\n| `OPENROUTER_API_KEY` | `OpenRouterAdapter` | [OpenRouter → Keys](https://openrouter.ai/keys) |\r\n| `OPENAI_API_KEY` | `OpenAiAdapter` | [OpenAI platform → API keys](https://platform.openai.com/api-keys) |\r\n\r\nConsider skipping chain entries whose key is missing (with a startup warning)\r\ninstead of failing: the app then keeps working on whatever providers are\r\nconfigured, and a partially configured preview deploy still serves requests.\r\n\r\n### 2. Firebase credentials for the cooldown store\r\n\r\nNetlify functions have no Google Cloud identity, so firebase-admin's\r\n`applicationDefault()` cannot work there — you must pass an explicit service\r\naccount:\r\n\r\n1. In the [Firebase console](https://console.firebase.google.com/), open your\r\n   project → ⚙ **Project settings** → **Service accounts** → **Generate new\r\n   private key**. This downloads a JSON file.\r\n2. From that JSON you need three values: `project_id`, `client_email`, and\r\n   `private_key`. Do **not** commit the file or ship it in the repo.\r\n3. Set them as `FIREBASE_PROJECT_ID`, `FIREBASE_CLIENT_EMAIL`, and\r\n   `FIREBASE_PRIVATE_KEY`.\r\n\r\n**The private-key newline gotcha.** `private_key` is a multi-line PEM block.\r\nDepending on how you set the variable (UI paste vs. CLI vs. copying the JSON\r\nvalue with its `\\n` escape sequences intact), the value that reaches your\r\nfunction may contain literal backslash-n instead of real newlines — and\r\nfirebase-admin then fails with `Invalid PEM formatted message`. Normalize in\r\ncode; the `replace` is a no-op when the newlines are already real:\r\n\r\n```ts\r\nimport { cert, getApps, initializeApp } from \"firebase-admin/app\";\r\nimport { getFirestore } from \"firebase-admin/firestore\";\r\n\r\n// A named app avoids colliding with any firebase-admin app your framework\r\n// integration (e.g. nuxt-vuefire) registers in the same process.\r\nconst APP_NAME = \"llm-chain\";\r\n\r\nexport function getAdminFirestore() {\r\n  const existing = getApps().find((a) => a.name === APP_NAME);\r\n  const app =\r\n    existing ??\r\n    initializeApp(\r\n      {\r\n        credential: cert({\r\n          projectId: process.env.FIREBASE_PROJECT_ID!,\r\n          clientEmail: process.env.FIREBASE_CLIENT_EMAIL!,\r\n          privateKey: process.env.FIREBASE_PRIVATE_KEY!.replace(/\\\\n/g, \"\\n\"),\r\n        }),\r\n      },\r\n      APP_NAME,\r\n    );\r\n  return getFirestore(app);\r\n}\r\n```\r\n\r\n### 3. Setting the variables in Netlify\r\n\r\n- **UI:** Project configuration → **Environment variables** → *Add a\r\n  variable*. Paste the PEM value as-is (the multi-line textarea preserves\r\n  newlines). Mark each of these as **secret** so they're masked in logs and\r\n  the UI, and restrict the scope to **Functions** — neither the build nor\r\n  post-processing needs provider keys or Firebase credentials.\r\n- **CLI:** `netlify env:set GEMINI_API_KEY \"…\" --secret`. For the private\r\n  key it's usually easier to paste the JSON's `private_key` string (with its\r\n  `\\n` escapes) and rely on the `replace()` above.\r\n- Environment variables are baked into functions **at deploy time** — after\r\n  adding or changing one, trigger a redeploy or it won't be picked up.\r\n- **Size limit:** Netlify functions run on AWS Lambda, which caps the total\r\n  environment at 4 KB. A Firebase private key alone is ~1.7 KB, so keep\r\n  unrelated variables scoped away from Functions if you get close.\r\n- **Local dev:** `netlify dev` injects the same variables locally; without\r\n  it, put the values in your framework's `.env` (git-ignored).\r\n\r\n### 4. Wire it together\r\n\r\n```ts\r\nconst chain = createFallbackChain({\r\n  entries,\r\n  cooldownStore: new FirestoreCooldownStore(getAdminFirestore()),\r\n});\r\n```\r\n\r\nConsider wrapping the store fail-open (catch and log store errors, treat\r\n`check` as \"not on cooldown\") so Firestore trouble can degrade cooldown\r\npersistence instead of blocking generation.\r\n\r\n## Writing an adapter\r\n\r\nA Provider is one class implementing two members ([ADR 0001](docs/adr/0001-raw-provider-sdks.md)):\r\n\r\n```ts\r\nimport {\r\n  type ProviderAdapter, type AdapterRequest,\r\n  toStrictJsonSchema, // or toGeminiSchema / toJsonSchemaResponseFormat\r\n  QuotaError, TransientError, InvalidRequestError,\r\n} from \"@alexplusplus/llm-fallback-chain\";\r\n\r\nclass MyAdapter implements ProviderAdapter {\r\n  readonly providerId = \"my-provider\";\r\n\r\n  async generate(request: AdapterRequest): Promise<string> {\r\n    // 1. If request.schema is present (structured mode), compile it (portable\r\n    //    form) to your provider's dialect; request.schemaName is set alongside\r\n    //    it. If absent (plain-text mode), omit your provider's\r\n    //    schema-enforcement field from the request entirely.\r\n    // 2. If request.reasoningEffort is present, convert it to your provider's\r\n    //    dialect (hardcoded map, see ADR 0003); if absent, send no\r\n    //    reasoning-related field. The chain has already validated the value.\r\n    // 3. Call the provider with request.modelId and request.prompt.\r\n    // 4. Return the raw response text — in structured mode the chain parses\r\n    //    and validates it; in plain mode it is returned verbatim.\r\n    // 5. Map every failure to QuotaError (with a retryAt hint when the\r\n    //    provider gives one), TransientError, or InvalidRequestError.\r\n  }\r\n}\r\n```\r\n\r\nAdapters never see the chain: ordering, cooldowns, and fallthrough live\r\nentirely in the chain walker.\r\n\r\n> Since v0.2.0, `request.schema` / `request.schemaName` are optional\r\n> (absent = plain-text mode) and `request.reasoningEffort` was added. Custom\r\n> adapters written against v0.1.x assumed `schema` was always present —\r\n> handle its absence when upgrading.\r\n\r\n## Configuration reference\r\n\r\n```ts\r\ncreateFallbackChain({\r\n  entries,                              // required, tried top-down\r\n  cooldownStore,                        // default: new InMemoryCooldownStore()\r\n  transientCooldownMs: 60_000,          // short cooldown length\r\n  quotaRetryFallback: nextUtcMidnight,  // long cooldown when provider gives no hint\r\n  now: () => new Date(),                // injectable clock (tests)\r\n});\r\n```\r\n\r\n## License\r\n\r\n[MIT](LICENSE)\r\n","readmeFilename":"README.md"}