{"_id":"@divinci-ai/langextract-ts","_rev":"2-9399809ff837b150f9ca88e52f9b94b8","name":"@divinci-ai/langextract-ts","dist-tags":{"latest":"0.2.0"},"versions":{"0.1.0":{"name":"@divinci-ai/langextract-ts","version":"0.1.0","keywords":["langextract","extraction","nlp","llm","rag","chunking","tokenizer","structured-extraction","source-grounding","gemini","cloudflare-workers-ai"],"author":{"name":"Divinci AI"},"license":"Apache-2.0","_id":"@divinci-ai/langextract-ts@0.1.0","maintainers":[{"name":"mike-divinci","email":"mike@divinci.ai"}],"homepage":"https://github.com/Divinci-AI/langextract-ts#readme","bugs":{"url":"https://github.com/Divinci-AI/langextract-ts/issues"},"dist":{"shasum":"e7ef21bf759f3478a582745e75c1209d5bc425ef","tarball":"https://registry.npmjs.org/@divinci-ai/langextract-ts/-/langextract-ts-0.1.0.tgz","fileCount":39,"integrity":"sha512-uNYQEZkDwtWhxMmnJBUMB2yf8S43qzZ27NI+pttdNZOMnVQcEhUr4rq0AfSIoQ7C8QUEDhYEhyAbBnzOB1Ee+g==","signatures":[{"sig":"MEYCIQC58m/KhR87giXO+6EqGBf5MTtd28Lfe8/Bvv0zSftrBgIhALGzgMSHfwELntVDNiBNBPpjbzvcwFrxezWFlj+FLYmq","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1183709},"main":"./dist/index.cjs","type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=18.0.0"},"exports":{".":{"import":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"require":{"types":"./dist/index.d.cts","default":"./dist/index.cjs"}},"./providers/gemini":{"import":{"types":"./dist/providers/gemini.d.ts","default":"./dist/providers/gemini.js"},"require":{"types":"./dist/providers/gemini.d.cts","default":"./dist/providers/gemini.cjs"}},"./providers/cloudflare":{"import":{"types":"./dist/providers/cloudflare.d.ts","default":"./dist/providers/cloudflare.js"},"require":{"types":"./dist/providers/cloudflare.d.cts","default":"./dist/providers/cloudflare.cjs"}}},"gitHead":"5522d54e52be8db900c6dc91427e8fba95457d53","scripts":{"dev":"tsup --watch","lint":"eslint src/ tests/","test":"vitest run","build":"tsup","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"npm run build"},"_npmUser":{"name":"mike-divinci","email":"mike@divinci.ai"},"repository":{"url":"git+https://github.com/Divinci-AI/langextract-ts.git","type":"git"},"_npmVersion":"11.6.2","description":"TypeScript port of Google's LangExtract — structured information extraction from text using LLMs with source grounding","directories":{},"_nodeVersion":"25.6.1","dependencies":{"zod":"^3.23.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.0.0","eslint":"^9.0.0","vitest":"^3.0.0","typescript":"^5.7.0","@google/genai":"^1.0.0","@types/js-yaml":"^4.0.0"},"peerDependencies":{"@google/genai":">=1.0.0"},"optionalDependencies":{"js-yaml":"^4.1.0"},"peerDependenciesMeta":{"@google/genai":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/langextract-ts_0.1.0_1771029421215_0.20360860449705398","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@divinci-ai/langextract-ts","version":"0.2.0","description":"TypeScript port of Google's LangExtract — structured information extraction from text using LLMs with source grounding","license":"Apache-2.0","type":"module","main":"./dist/index.cjs","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"import":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"require":{"types":"./dist/index.d.cts","default":"./dist/index.cjs"}},"./providers/gemini":{"import":{"types":"./dist/providers/gemini.d.ts","default":"./dist/providers/gemini.js"},"require":{"types":"./dist/providers/gemini.d.cts","default":"./dist/providers/gemini.cjs"}},"./providers/cloudflare":{"import":{"types":"./dist/providers/cloudflare.d.ts","default":"./dist/providers/cloudflare.js"},"require":{"types":"./dist/providers/cloudflare.d.cts","default":"./dist/providers/cloudflare.cjs"}},"./providers/openai":{"import":{"types":"./dist/providers/openai.d.ts","default":"./dist/providers/openai.js"},"require":{"types":"./dist/providers/openai.d.cts","default":"./dist/providers/openai.cjs"}},"./providers/anthropic":{"import":{"types":"./dist/providers/anthropic.d.ts","default":"./dist/providers/anthropic.js"},"require":{"types":"./dist/providers/anthropic.d.cts","default":"./dist/providers/anthropic.cjs"}}},"engines":{"node":">=18.0.0"},"scripts":{"build":"tsup","dev":"tsup --watch","test":"vitest run","test:watch":"vitest","lint":"eslint src/ tests/","typecheck":"tsc --noEmit","prepublishOnly":"npm run build"},"dependencies":{"zod":"^3.23.0"},"optionalDependencies":{"js-yaml":"^4.1.0"},"peerDependencies":{"@google/genai":">=1.0.0"},"peerDependenciesMeta":{"@google/genai":{"optional":true}},"devDependencies":{"@google/genai":"^1.0.0","@types/js-yaml":"^4.0.0","eslint":"^9.0.0","tsup":"^8.0.0","typescript":"^5.7.0","vitest":"^3.0.0"},"repository":{"type":"git","url":"git+https://github.com/Divinci-AI/langextract-ts.git"},"keywords":["langextract","extraction","nlp","llm","rag","chunking","tokenizer","structured-extraction","source-grounding","gemini","cloudflare-workers-ai","openai","anthropic"],"author":{"name":"Divinci AI"},"homepage":"https://github.com/Divinci-AI/langextract-ts#readme","gitHead":"5522d54e52be8db900c6dc91427e8fba95457d53","_id":"@divinci-ai/langextract-ts@0.2.0","bugs":{"url":"https://github.com/Divinci-AI/langextract-ts/issues"},"_nodeVersion":"25.6.1","_npmVersion":"11.6.2","dist":{"integrity":"sha512-n1Pt19peOMhRSJBL2hbpP+s317F+xWKyHxMEn20q3/nvNzOVf7Wg5F4NATj8QHYhFeMVT3ptg5XNhAyXgf3hdw==","shasum":"4dc6748835268599b364a92b90a82daf9d3266fe","tarball":"https://registry.npmjs.org/@divinci-ai/langextract-ts/-/langextract-ts-0.2.0.tgz","fileCount":59,"unpackedSize":1287400,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCID2OieMJuBUBPAw4WmzDPZUMxdxOBjgtk7PhvrJvA0I0AiAe0UbQUMl0mwzIgfQCb2+B7YNwVKUVvrwqVhojm/2gKg=="}]},"_npmUser":{"name":"mike-divinci","email":"mike@divinci.ai"},"directories":{},"maintainers":[{"name":"mike-divinci","email":"mike@divinci.ai"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/langextract-ts_0.2.0_1771056279448_0.11208085172944338"},"_hasShrinkwrap":false}},"time":{"created":"2026-02-14T00:37:01.145Z","modified":"2026-02-14T08:04:39.787Z","0.1.0":"2026-02-14T00:37:01.469Z","0.2.0":"2026-02-14T08:04:39.633Z"},"bugs":{"url":"https://github.com/Divinci-AI/langextract-ts/issues"},"author":{"name":"Divinci AI"},"license":"Apache-2.0","homepage":"https://github.com/Divinci-AI/langextract-ts#readme","keywords":["langextract","extraction","nlp","llm","rag","chunking","tokenizer","structured-extraction","source-grounding","gemini","cloudflare-workers-ai","openai","anthropic"],"repository":{"type":"git","url":"git+https://github.com/Divinci-AI/langextract-ts.git"},"description":"TypeScript port of Google's LangExtract — structured information extraction from text using LLMs with source grounding","maintainers":[{"name":"mike-divinci","email":"mike@divinci.ai"}],"readme":"# LangExtract-TS\n\nTypeScript port of Google's [LangExtract](https://github.com/google/langextract) — structured information extraction from text using LLMs with precise character-level source grounding.\n\n> Based on LangExtract v1.1.1 by Google. Ported to TypeScript with Gemini and Cloudflare Workers AI support.\n\n## Features\n\n- **Source grounding** — every extraction maps back to exact character positions in the original text\n- **Sentence-aware chunking** — three-strategy chunker that respects sentence boundaries\n- **Two-phase alignment** — exact token matching + fuzzy fallback for robust source mapping\n- **Universal runtime** — runs on Node.js 18+, Cloudflare Workers, Deno, and Bun\n- **Minimal dependencies** — only `zod` required; provider SDKs are optional\n- **Interactive visualization** — self-contained HTML with playback controls\n- **Provider plugins** — built-in Gemini + Cloudflare, extensible for custom providers\n\n## Installation\n\n```bash\nnpm install langextract-ts\n# or\npnpm add langextract-ts\n```\n\nFor Gemini support (optional):\n```bash\nnpm install @google/genai\n```\n\n## Quick Start\n\n```typescript\nimport { extract } from \"langextract-ts\";\n\nconst result = await extract(\n  \"The patient takes Aspirin 81mg daily for heart health.\",\n  {\n    promptDescription: \"Extract all medications with their dosage and frequency.\",\n    examples: [{\n      text: \"She takes Lisinopril 10mg once daily.\",\n      extractions: [{\n        extractionClass: \"medication\",\n        text: \"Lisinopril\",\n        attributes: { dosage: \"10mg\", frequency: \"once daily\" },\n      }],\n    }],\n    modelId: \"gemini-2.0-flash\",\n    apiKey: process.env.GOOGLE_API_KEY,\n  },\n);\n\n// result.extractions[0]:\n// {\n//   extractionClass: \"medication\",\n//   text: \"Aspirin\",\n//   charInterval: { startPos: 18, endPos: 25 },\n//   alignmentStatus: \"exact\",\n//   attributes: { dosage: \"81mg\", frequency: \"daily\" },\n// }\n```\n\n### With Cloudflare Workers AI\n\n```typescript\nimport { extract } from \"langextract-ts\";\n\nconst result = await extract(\n  \"Romeo professes his love for Juliet in the famous balcony scene.\",\n  {\n    promptDescription: \"Extract all characters mentioned.\",\n    examples: [{\n      text: \"Hamlet speaks to Horatio.\",\n      extractions: [{\n        extractionClass: \"character\",\n        text: \"Hamlet\",\n      }],\n    }],\n    modelId: \"@cf/meta/llama-3.3-70b-instruct-fp8-fast\",\n    apiKey: process.env.CF_API_TOKEN,\n    accountId: process.env.CF_ACCOUNT_ID,\n  },\n);\n```\n\n## API\n\n### `extract(input, options)`\n\nMain entry point. Accepts strings, URLs, or `Document[]`.\n\n| Option | Default | Description |\n|---|---|---|\n| `promptDescription` | required | Task instructions for the LLM |\n| `examples` | required | Few-shot examples |\n| `modelId` | `\"gemini-2.0-flash\"` | Model identifier |\n| `apiKey` | env var | Provider API key |\n| `maxCharBuffer` | `1000` | Max characters per chunk |\n| `batchLength` | `10` | Chunks per inference batch |\n| `maxWorkers` | `10` | Concurrent requests |\n| `extractionPasses` | `1` | Number of extraction passes |\n| `contextWindowChars` | `0` | Cross-chunk context window |\n| `formatType` | `\"json\"` | Output format (`\"json\"` or `\"yaml\"`) |\n\n### Chunking\n\n```typescript\nimport { chunkDocument, createDocument } from \"langextract-ts\";\n\nconst doc = createDocument(\"Your long text here...\");\nfor (const chunk of chunkDocument(doc, { maxCharBuffer: 500 })) {\n  console.log(chunk.text, chunk.charInterval);\n}\n```\n\n### Tokenization\n\n```typescript\nimport { RegexTokenizer, UnicodeTokenizer } from \"langextract-ts\";\n\nconst tokenizer = new RegexTokenizer();\nconst { tokens } = tokenizer.tokenize(\"Hello world!\");\n// tokens: [{ text: \"Hello\", tokenType: \"word\", charInterval: { startPos: 0, endPos: 5 } }, ...]\n\n// For CJK/international text:\nconst unicode = new UnicodeTokenizer();\nconst { tokens: cjkTokens } = unicode.tokenize(\"Hello 世界\");\n```\n\n### Visualization\n\n```typescript\nimport { visualize } from \"langextract-ts\";\n\nconst html = visualize(annotatedDocument, {\n  title: \"Medication Extraction\",\n  animationSpeed: 1500,\n});\n// Save `html` to a file and open in browser\n```\n\n### Custom Providers\n\n```typescript\nimport { BaseLanguageModel, registerProvider } from \"langextract-ts\";\n\nclass MyProvider extends BaseLanguageModel {\n  async *infer(prompts) {\n    for (const prompt of prompts) {\n      const response = await myApi.call(prompt);\n      yield [{ output: response, score: 1.0 }];\n    }\n  }\n}\n\nregisterProvider([/^my-model/], () => MyProvider, 20);\n```\n\n## Architecture\n\n```\nInput Text/URL\n  -> Tokenization (RegexTokenizer or UnicodeTokenizer)\n  -> Sentence-aware Chunking (3 strategies)\n  -> Few-shot Prompt Construction\n  -> Batched LLM Inference (concurrent with Semaphore)\n  -> JSON Parsing + Extraction\n  -> Two-phase Alignment (exact + fuzzy via SequenceMatcher)\n  -> AnnotatedDocument with CharInterval positions\n```\n\n## Runtime Compatibility\n\n| Runtime | Supported | Notes |\n|---|---|---|\n| Node.js 18+ | Yes | Full support |\n| Cloudflare Workers | Yes | Web APIs only |\n| Deno | Yes | V8-based |\n| Bun | Yes | JavaScriptCore |\n\n## License\n\nApache-2.0\n\nThis project is a derivative work of [Google's LangExtract](https://github.com/google/langextract), originally licensed under Apache-2.0.\n","readmeFilename":"README.md"}