{"_id":"@circulo-ai/file-parsers","_rev":"5-3a47201978b4ffa49e0daf77eeb41c83","name":"@circulo-ai/file-parsers","dist-tags":{"latest":"0.2.0"},"versions":{"0.1.0":{"name":"@circulo-ai/file-parsers","version":"0.1.0","license":"Apache-2.0","_id":"@circulo-ai/file-parsers@0.1.0","maintainers":[{"name":"monobit","email":"1839491@gmail.com"}],"dist":{"shasum":"79f9b66ce5a729e99d40de6f467ef28376656a0b","tarball":"https://registry.npmjs.org/@circulo-ai/file-parsers/-/file-parsers-0.1.0.tgz","fileCount":46,"integrity":"sha512-QUpn8S6KFLvZ4IGmXAa5wKht4YjzZ+cBzrZzWfo0T+WcBqfPJw/EsHZxyRdvZMDBgAhhKU6L42iYDwwsl7GOHg==","signatures":[{"sig":"MEUCIQD2RqGdf/d+2OkrPFZQC9npbc1px11tlzMSuoJS+n9utQIgV9ygSYKcRe02NlXb9TeuArCs5EKtr08mUK8XARoBSLE=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":40548},"main":"./dist/index.js","type":"module","_from":"file:circulo-ai-file-parsers-0.1.0.tgz","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"private":false,"scripts":{"dev":"tsc -p tsconfig.json --watch","build":"tsc -p tsconfig.json","type-check":"tsc --noEmit -p tsconfig.json"},"_npmUser":{"name":"monobit","email":"1839491@gmail.com"},"_resolved":"/tmp/9739a02b3e3cb5c4e85233ca23c7a6dd/circulo-ai-file-parsers-0.1.0.tgz","_integrity":"sha512-QUpn8S6KFLvZ4IGmXAa5wKht4YjzZ+cBzrZzWfo0T+WcBqfPJw/EsHZxyRdvZMDBgAhhKU6L42iYDwwsl7GOHg==","_npmVersion":"10.8.2","directories":{},"_nodeVersion":"20.19.6","dependencies":{"xlsx":"^0.18.5","cheerio":"^1.1.2","js-yaml":"^4.1.1","mammoth":"^1.11.0","csv-parse":"^6.1.0","pdf-parse":"^2.4.5","officeparser":"^5.2.2"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.3","@types/node":"^20.19.25","@types/js-yaml":"^4.0.9"},"_npmOperationalInternal":{"tmp":"tmp/file-parsers_0.1.0_1764958831155_0.19877398477072017","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@circulo-ai/file-parsers","version":"0.1.1","license":"Apache-2.0","_id":"@circulo-ai/file-parsers@0.1.1","maintainers":[{"name":"monobit","email":"1839491@gmail.com"}],"dist":{"shasum":"6b0478a79d1ec54f36268ff13796323ea9e2e11f","tarball":"https://registry.npmjs.org/@circulo-ai/file-parsers/-/file-parsers-0.1.1.tgz","fileCount":46,"integrity":"sha512-p2KiChCVrSezoGePnYTjTWhXK8ZHpBtY3KUTXVYfIjzF3JrvwrYkIocq/7ZrUsZs1fk8UjG9c5UPMBjPBX71QQ==","signatures":[{"sig":"MEQCIH2RtgwvEmVhrylredeBPOB9Z1z7fKo8sYkjQ4MQwRMsAiBaJdocFLFTLhDw/8+sxGinnLzoK1E/oLZcjcunUBXrpA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":40607},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"5ea8c5a2f3f846884ff384236baacfe2800309e6","private":false,"scripts":{"dev":"tsc -p tsconfig.json --watch","build":"tsc -p tsconfig.json","prepack":"bun run build","typecheck":"tsc --noEmit -p tsconfig.json"},"_npmUser":{"name":"monobit","email":"1839491@gmail.com"},"_npmVersion":"10.8.2","directories":{},"_nodeVersion":"20.19.6","dependencies":{"xlsx":"^0.18.5","cheerio":"^1.1.2","js-yaml":"^4.1.1","mammoth":"^1.11.0","csv-parse":"^6.1.0","pdf-parse":"^2.4.5","officeparser":"^5.2.2"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.3","@types/node":"^20.19.25","@types/js-yaml":"^4.0.9"},"_npmOperationalInternal":{"tmp":"tmp/file-parsers_0.1.1_1765377797398_0.43432144080893176","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@circulo-ai/file-parsers","version":"0.1.2","license":"Apache-2.0","_id":"@circulo-ai/file-parsers@0.1.2","maintainers":[{"name":"monobit","email":"1839491@gmail.com"}],"dist":{"shasum":"f9d4b4991871c2a90ff058f72094cf0a19b9326d","tarball":"https://registry.npmjs.org/@circulo-ai/file-parsers/-/file-parsers-0.1.2.tgz","fileCount":47,"integrity":"sha512-dD5aQLXrtzyKWZp+OC/avc67ylqVETOFzAd71T9kxkTMqm6M6hhVebGcTcjNvtoxppmobH1PTJ35u7qjNYNdxg==","signatures":[{"sig":"MEQCIDvsgsI/QZIBkWLPsGfbGLvwr8b9QtdSP9cYw4UbOL16AiBG5DBmySDWfIxEx8qjkBsyNxYCOCr7tyOUa2a2zWYdQg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":45642},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"b024191cff85eaad2b81be6448c37634414016c0","private":false,"scripts":{"dev":"tsc -p tsconfig.json --watch","build":"tsc -p tsconfig.json","prepack":"bun run build","typecheck":"tsc --noEmit -p tsconfig.json"},"_npmUser":{"name":"monobit","email":"1839491@gmail.com"},"_npmVersion":"10.8.2","description":"Lightweight, promise-based parsers for common document types. The package detects the file type by extension, extracts UTF-8-safe text, and returns structured metadata so downstream pipelines can reason about the content (row counts, sheet names, page cou","directories":{},"_nodeVersion":"20.20.0","dependencies":{"xlsx":"^0.18.5","cheerio":"^1.1.2","js-yaml":"^4.1.1","mammoth":"^1.11.0","csv-parse":"^6.1.0","pdf-parse":"^2.4.5","officeparser":"^5.2.2"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.3","@types/node":"^20.19.25","@types/js-yaml":"^4.0.9"},"_npmOperationalInternal":{"tmp":"tmp/file-parsers_0.1.2_1770571056763_0.6752599306689371","host":"s3://npm-registry-packages-npm-production"}},"0.1.3":{"name":"@circulo-ai/file-parsers","version":"0.1.3","license":"Apache-2.0","_id":"@circulo-ai/file-parsers@0.1.3","maintainers":[{"name":"monobit","email":"1839491@gmail.com"}],"homepage":"https://github.com/circulo-ai/circulo#readme","bugs":{"url":"https://github.com/circulo-ai/circulo/issues"},"dist":{"shasum":"a21b8e3ca1401d60d5915e3d0044a2d977d3a970","tarball":"https://registry.npmjs.org/@circulo-ai/file-parsers/-/file-parsers-0.1.3.tgz","fileCount":48,"integrity":"sha512-yYj/BeYY7LuacelDa0oRILkiLDpeUvz6hdlMirWwq6/uB+ohBqkRplt1Egp8LPbXAh27ZGiPGro2fGzwbyxHgQ==","signatures":[{"sig":"MEUCID+DiD3Iwc4ms1+NUoYIX0diqYEcEaD3dfb3UjrXLwMnAiEAmzLrXmyWwrCOjqwW3Gn6JrijgHjsDqEESv/L7irVeC8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":55877},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"a5808a0f18a84c47044ae8ffcaab2568c558c25c","private":false,"scripts":{"dev":"bun ./node_modules/typescript/bin/tsc -p tsconfig.json --watch","build":"bun ./node_modules/typescript/bin/tsc -p tsconfig.json","prepack":"bun run build","typecheck":"bun ./node_modules/typescript/bin/tsc --noEmit -p tsconfig.json"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:3b75df59-3f0f-4987-88e0-17fdb358fc24"}},"repository":{"url":"git+https://github.com/circulo-ai/circulo.git","type":"git"},"_npmVersion":"11.17.0","description":"Lightweight, promise-based parsers for common document types. The package detects the file type by extension, extracts UTF-8-safe text, and returns structured metadata so downstream pipelines can reason about the content (row counts, sheet names, page cou","directories":{},"sideEffects":false,"_nodeVersion":"24.19.0","dependencies":{"xlsx":"^0.18.5","cheerio":"^1.1.2","js-yaml":"^4.1.1","mammoth":"^1.11.0","csv-parse":"^6.1.0","pdf-parse":"^2.4.5","officeparser":"^5.2.2"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.9.3","@types/node":"^20.19.25","@types/js-yaml":"^4.0.9"},"_npmOperationalInternal":{"tmp":"tmp/file-parsers_0.1.3_1786994283381_0.08858665882111527","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@circulo-ai/file-parsers","version":"0.2.0","private":false,"type":"module","sideEffects":false,"repository":{"type":"git","url":"git+https://github.com/circulo-ai/circulo.git"},"license":"Apache-2.0","main":"./dist/index.js","types":"./dist/index.d.ts","publishConfig":{"access":"public"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"scripts":{"dev":"bun ./node_modules/typescript/bin/tsc -p tsconfig.json --watch","build":"bun ./node_modules/typescript/bin/tsc -p tsconfig.json","typecheck":"bun ./node_modules/typescript/bin/tsc --noEmit -p tsconfig.json","prepack":"bun run build"},"dependencies":{"cheerio":"^1.1.2","csv-parse":"^6.1.0","js-yaml":"^4.1.1","mammoth":"^1.11.0","officeparser":"^5.2.2","pdf-parse":"^2.4.5","xlsx":"^0.18.5"},"devDependencies":{"@types/js-yaml":"^4.0.9","@types/node":"^20.19.25","typescript":"^5.9.3"},"gitHead":"6aa82631a97e7dae0ef6efbbe85d1c3574b5416e","_id":"@circulo-ai/file-parsers@0.2.0","description":"Lightweight, promise-based parsers for common document types. The package detects the file type by extension, extracts UTF-8-safe text, and returns structured metadata so downstream pipelines can reason about the content (row counts, sheet names, page cou","bugs":{"url":"https://github.com/circulo-ai/circulo/issues"},"homepage":"https://github.com/circulo-ai/circulo#readme","_nodeVersion":"24.19.0","_npmVersion":"11.17.0","dist":{"integrity":"sha512-fcgNBb2LzpoqD5/8uPnmWi5nuUFpMob3RtwrWu+gGMgemLrB3yj+13T9mYfOQ0dm4BSEM+MXPHZV62e1d5HxMQ==","shasum":"d06083b26770cd6a85eac0755083f27ac25a6045","tarball":"https://registry.npmjs.org/@circulo-ai/file-parsers/-/file-parsers-0.2.0.tgz","fileCount":48,"unpackedSize":55877,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIGT/+wBatBTtmxjYiJXnPl+endNQ/TBQoTvjoUP4VNACAiAlCRPidUeEX5Z46OpH7ROtJkmHeWkyPM/7nwsZ8ENg6Q=="}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:3b75df59-3f0f-4987-88e0-17fdb358fc24"}},"directories":{},"maintainers":[{"name":"monobit","email":"1839491@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/file-parsers_0.2.0_1787587260385_0.9338260446168265"},"_hasShrinkwrap":false}},"time":{"created":"2025-12-05T18:20:31.091Z","modified":"2026-08-24T16:01:00.709Z","0.1.0":"2025-12-05T18:20:31.326Z","0.1.1":"2025-12-10T14:43:17.532Z","0.1.2":"2026-02-08T17:17:36.904Z","0.1.3":"2026-08-17T19:18:03.549Z","0.2.0":"2026-08-24T16:01:00.546Z"},"bugs":{"url":"https://github.com/circulo-ai/circulo/issues"},"license":"Apache-2.0","homepage":"https://github.com/circulo-ai/circulo#readme","repository":{"type":"git","url":"git+https://github.com/circulo-ai/circulo.git"},"description":"Lightweight, promise-based parsers for common document types. The package detects the file type by extension, extracts UTF-8-safe text, and returns structured metadata so downstream pipelines can reason about the content (row counts, sheet names, page cou","maintainers":[{"name":"monobit","email":"1839491@gmail.com"}],"readme":"# @circulo-ai/file-parsers\n\nLightweight, promise-based parsers for common document types. The package detects the file type by extension, extracts UTF-8-safe text, and returns structured metadata so downstream pipelines can reason about the content (row counts, sheet names, page counts, headings, parsed JSON/YAML, etc.).\n\n## Features at a glance\n\n| Feature                  | Description                                                                                         |\n| ------------------------ | --------------------------------------------------------------------------------------------------- |\n| Unified API              | `parseFile`, `parseBuffer`, and `isSupportedFileType` route to the right parser based on extension. |\n| Broad format coverage    | pdf, csv, doc/docx, txt, md, xlsx/xls, html/htm, ppt/pptx, json, yaml/yml.                          |\n| UTF-8 sanitization       | Control characters and invalid surrogates are removed before returning content.                     |\n| Streaming for large CSVs | Reads in chunks with row sampling, error throttling, and preview truncation.                        |\n| Structured metadata      | Each parser returns useful context (counts, headings, parsed objects, sheet names, etc.).           |\n| Pluggable logging        | Pass any logger with `info`, `warn`, and `error` to trace parsing steps and errors.                 |\n| Tree-shakable ESM        | Parsers are dynamically imported so dependencies only load when needed.                             |\n\n## Supported formats\n\n| Extension  | Notes                                                                                    |\n| ---------- | ---------------------------------------------------------------------------------------- |\n| pdf        | Uses `pdf-parse` (function or class export); estimates page count when missing.          |\n| csv        | Streams rows, samples first 100, previews first 1,000, and aborts after too many errors. |\n| doc / docx | DOC via `officeparser`; DOCX via `mammoth`.                                              |\n| txt / md   | Plain UTF-8 text passthrough with sanitization.                                          |\n| xlsx / xls | Uses `xlsx`, dumps each sheet with CSV-like rows.                                        |\n| ppt / pptx | Via `officeparser` text extraction.                                                      |\n| html / htm | Via `cheerio`, returns full text plus title/headings metadata.                           |\n| json       | Returns raw text plus parsed object.                                                     |\n| yaml / yml | Returns raw text plus parsed object via `js-yaml`.                                       |\n\n## Install\n\n```bash\nnpm install @circulo-ai/file-parsers\n# or\npnpm add @circulo-ai/file-parsers\n# or\nbun add @circulo-ai/file-parsers\n```\n\n## Quick start\n\n```typescript\nimport {\n  parseFile,\n  parseBuffer,\n  isSupportedFileType,\n} from \"@circulo-ai/file-parsers\";\n\n// Parse by path (extension auto-detected)\nconst pdfResult = await parseFile(\"docs/report.pdf\");\nconsole.log(pdfResult.content.slice(0, 200));\nconsole.log(pdfResult.metadata);\n\n// Parse a buffer (you must provide the extension)\nconst csvBuffer = await fs.promises.readFile(\"data/example.csv\");\nconst csvResult = await parseBuffer(csvBuffer, \"csv\");\n\n// Check support before parsing\nif (!(await isSupportedFileType(\"pptx\"))) {\n  throw new Error(\"Unsupported\");\n}\n```\n\n## Using a custom logger\n\n```typescript\nimport { createFileParser } from \"@circulo-ai/file-parsers\";\nimport pino from \"pino\";\n\nconst logger = pino({ level: \"info\" });\nconst parser = createFileParser({\n  logger: {\n    info: (...args) => logger.info(args),\n    warn: (...args) => logger.warn(args),\n    error: (...args) => logger.error(args),\n  },\n});\n\nconst result = await parser.parseFile(\"slides/talk.pptx\");\n```\n\n## Direct parser classes (advanced)\n\n```typescript\nimport { CsvParser, PdfParser } from \"@circulo-ai/file-parsers\";\n\nconst csv = new CsvParser();\nconst csvResult = await csv.parseFile(\"data/huge.csv\");\n\nconst pdf = new PdfParser();\nconst pdfResult = await pdf.parseBuffer(myPdfBuffer);\n```\n\n## Behavior notes\n\n- CSV streaming: chunk size 16KB; skips malformed rows; logs first few errors; truncates preview after 1,000 rows while keeping counts.\n- Sanitization: sanitizeTextForUTF8 strips control chars, null bytes, replacement chars, and surrogate pairs to keep DB writes safe.\n- Extension handling: extensions are lowercased; parseBuffer expects values like \"pdf\" not \".pdf\".\n- Error handling: missing files, empty buffers, and unsupported extensions throw descriptive errors; isSupportedFileType returns false on loader failures.\n- Approximate token count: many parsers set tokenCount = Math.floor(characterCount / 4) as a quick LLM sizing heuristic.\n","readmeFilename":"README.md"}