{"_id":"@docmost/pdf-inspector","_rev":"4-7bb7a13673d435a7fc306158da8607fd","name":"@docmost/pdf-inspector","dist-tags":{"latest":"1.9.6"},"versions":{"1.9.3":{"name":"@docmost/pdf-inspector","version":"1.9.3","keywords":["pdf","pdf-extraction","pdf-parser","text-extraction","ocr","pdf-classification","napi","rust","firecrawl"],"license":"MIT","_id":"@docmost/pdf-inspector@1.9.3","maintainers":[{"name":"philipinho","email":"phil@neuronlabs.co.uk"}],"homepage":"https://github.com/docmost/pdf-inspector","bugs":{"url":"https://github.com/docmost/pdf-inspector/issues"},"dist":{"shasum":"6f81fc05dca2b929c0b032e3d466ec36cc0f4ed3","tarball":"https://registry.npmjs.org/@docmost/pdf-inspector/-/pdf-inspector-1.9.3.tgz","fileCount":8,"integrity":"sha512-s1B+vgyvfSMb/GBnfq48iLVY59oBl2deso1byIwoL3iJHUuxrhwJkUzSTvLt1/OwHn/i14jyCe17HYEWpnZgbg==","signatures":[{"sig":"MEYCIQD527m7zz7YqNiEAMZsdP3xd4OJPZMD19CeIn9sjDVHMgIhALm43pwr5M2yW/Wt6Uwx5v9iepS7VbdWae2IVKyEk7Rw","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@docmost%2fpdf-inspector@1.9.3","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":20657498},"main":"index.js","napi":{"package":{"name":"@firecrawl/pdf-inspector-js"},"targets":["x86_64-unknown-linux-gnu","aarch64-unknown-linux-gnu","aarch64-apple-darwin","x86_64-pc-windows-msvc"],"binaryName":"pdf-inspector"},"types":"index.d.ts","scripts":{"build":"napi build --platform --release","build:debug":"napi build --platform"},"_npmUser":{"name":"philipinho","email":"phil@neuronlabs.co.uk"},"repository":{"url":"git+https://github.com/docmost/pdf-inspector.git","type":"git"},"_npmVersion":"10.8.2","description":"Fast PDF classification, text extraction, and image extraction. Native Rust performance via napi-rs.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"@napi-rs/cli":"^3.4.1"},"_npmOperationalInternal":{"tmp":"tmp/pdf-inspector_1.9.3_1777567704154_0.814641575077425","host":"s3://npm-registry-packages-npm-production"}},"1.9.4":{"name":"@docmost/pdf-inspector","version":"1.9.4","keywords":["pdf","pdf-extraction","pdf-parser","text-extraction","ocr","pdf-classification","napi","rust","firecrawl"],"license":"MIT","_id":"@docmost/pdf-inspector@1.9.4","maintainers":[{"name":"philipinho","email":"phil@neuronlabs.co.uk"}],"homepage":"https://github.com/docmost/pdf-inspector","bugs":{"url":"https://github.com/docmost/pdf-inspector/issues"},"dist":{"shasum":"d37b7e2a1bd6c0da5ca900a3e56924fc6c10980a","tarball":"https://registry.npmjs.org/@docmost/pdf-inspector/-/pdf-inspector-1.9.4.tgz","fileCount":8,"integrity":"sha512-G5DNyDtLNxybTXWakqi7PuOEuSb/A2ZjDlv2WCkOkiHszPeILdrC+G0a4e4UP10yxvzuLfb23pJ5jy8fUSYZPw==","signatures":[{"sig":"MEUCIBQy7GhM9D4sYAPmdqyqLoaK30Vl/1LHfd70nKXvYblwAiEApo7KlDqgzT5Xs4VWaWDDZt+FxIkJz2sq0jXqNA7D7MQ=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@docmost%2fpdf-inspector@1.9.4","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":20666298},"main":"index.js","napi":{"package":{"name":"@firecrawl/pdf-inspector-js"},"targets":["x86_64-unknown-linux-gnu","aarch64-unknown-linux-gnu","aarch64-apple-darwin","x86_64-pc-windows-msvc"],"binaryName":"pdf-inspector"},"types":"index.d.ts","scripts":{"build":"napi build --platform --release","build:debug":"napi build --platform"},"_npmUser":{"name":"philipinho","email":"phil@neuronlabs.co.uk"},"repository":{"url":"git+https://github.com/docmost/pdf-inspector.git","type":"git"},"_npmVersion":"10.8.2","description":"Fast PDF classification, text extraction, and image extraction. Native Rust performance via napi-rs.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"@napi-rs/cli":"^3.4.1"},"_npmOperationalInternal":{"tmp":"tmp/pdf-inspector_1.9.4_1777643542481_0.778300734476302","host":"s3://npm-registry-packages-npm-production"}},"1.9.5":{"name":"@docmost/pdf-inspector","version":"1.9.5","keywords":["pdf","pdf-extraction","pdf-parser","text-extraction","ocr","pdf-classification","napi","rust","firecrawl"],"license":"MIT","_id":"@docmost/pdf-inspector@1.9.5","maintainers":[{"name":"philipinho","email":"phil@neuronlabs.co.uk"}],"homepage":"https://github.com/docmost/pdf-inspector","bugs":{"url":"https://github.com/docmost/pdf-inspector/issues"},"dist":{"shasum":"4e74d53ecea4cd72ca8a9066d3179288c0ff51e8","tarball":"https://registry.npmjs.org/@docmost/pdf-inspector/-/pdf-inspector-1.9.5.tgz","fileCount":8,"integrity":"sha512-G8bfKq94sZ9reMpsaasM8I1K952oyfQXk+/0vXxaRS1Aujl+gQzmEuibszLnqfFM1+Y79GYVpjRPK4TcKYlqMQ==","signatures":[{"sig":"MEUCIHxv4gdgTbKg17J2H+pG80oZONQMJdSEK/RUl55yAF0yAiEAkrwJCnYL6sP3WTIr/gguNE6RIiOBIkHMXVsDNqyrRgQ=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@docmost%2fpdf-inspector@1.9.5","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":20689434},"main":"index.js","napi":{"package":{"name":"@firecrawl/pdf-inspector-js"},"targets":["x86_64-unknown-linux-gnu","aarch64-unknown-linux-gnu","aarch64-apple-darwin","x86_64-pc-windows-msvc"],"binaryName":"pdf-inspector"},"types":"index.d.ts","scripts":{"build":"napi build --platform --release","build:debug":"napi build --platform"},"_npmUser":{"name":"philipinho","email":"phil@neuronlabs.co.uk"},"repository":{"url":"git+https://github.com/docmost/pdf-inspector.git","type":"git"},"_npmVersion":"10.8.2","description":"Fast PDF classification, text extraction, and image extraction. Native Rust performance via napi-rs.","directories":{},"_nodeVersion":"20.20.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"@napi-rs/cli":"^3.4.1"},"_npmOperationalInternal":{"tmp":"tmp/pdf-inspector_1.9.5_1778366025366_0.21609330676168237","host":"s3://npm-registry-packages-npm-production"}},"1.9.6":{"name":"@docmost/pdf-inspector","version":"1.9.6","description":"Fast PDF classification, text extraction, and image extraction. Native Rust performance via napi-rs.","main":"index.js","types":"index.d.ts","license":"MIT","keywords":["pdf","pdf-extraction","pdf-parser","text-extraction","ocr","pdf-classification","napi","rust","firecrawl"],"repository":{"type":"git","url":"git+https://github.com/docmost/pdf-inspector.git"},"homepage":"https://github.com/docmost/pdf-inspector","publishConfig":{"access":"public"},"napi":{"binaryName":"pdf-inspector","targets":["x86_64-unknown-linux-gnu","aarch64-unknown-linux-gnu","aarch64-apple-darwin","x86_64-pc-windows-msvc"],"package":{"name":"@firecrawl/pdf-inspector-js"}},"scripts":{"build":"napi build --platform --release","build:debug":"napi build --platform"},"devDependencies":{"@napi-rs/cli":"^3.4.1"},"_id":"@docmost/pdf-inspector@1.9.6","bugs":{"url":"https://github.com/docmost/pdf-inspector/issues"},"_nodeVersion":"20.20.2","_npmVersion":"10.8.2","dist":{"integrity":"sha512-8k8N8Mwu9xbpRC1jLcz4sFv88ev2oBnW56a/2WLbrOBkfXzyZV2Tml5PikUwEWT4cUXfYfk2dGnJpWQYgCESCQ==","shasum":"9bf6b6bb3ec80b67561247a59b2c056841bfc875","tarball":"https://registry.npmjs.org/@docmost/pdf-inspector/-/pdf-inspector-1.9.6.tgz","fileCount":8,"unpackedSize":20810898,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@docmost%2fpdf-inspector@1.9.6","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDX1I4kJCyuRbMO/oGmn47lMUKFuAezZItWkfuxjG+MuAIgQWCZ4Wba5sE3V6E8ngNsQik1g/Hs7Ptrj1c9eS5efcg="}]},"_npmUser":{"name":"philipinho","email":"phil@neuronlabs.co.uk"},"directories":{},"maintainers":[{"name":"philipinho","email":"phil@neuronlabs.co.uk"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/pdf-inspector_1.9.6_1778794509408_0.42851032385201404"},"_hasShrinkwrap":false}},"time":{"created":"2026-04-30T16:48:24.094Z","modified":"2026-05-14T21:35:10.091Z","1.9.3":"2026-04-30T16:48:24.567Z","1.9.4":"2026-05-01T13:52:22.923Z","1.9.5":"2026-05-09T22:33:45.749Z","1.9.6":"2026-05-14T21:35:09.787Z"},"bugs":{"url":"https://github.com/docmost/pdf-inspector/issues"},"license":"MIT","homepage":"https://github.com/docmost/pdf-inspector","keywords":["pdf","pdf-extraction","pdf-parser","text-extraction","ocr","pdf-classification","napi","rust","firecrawl"],"repository":{"type":"git","url":"git+https://github.com/docmost/pdf-inspector.git"},"description":"Fast PDF classification, text extraction, and image extraction. Native Rust performance via napi-rs.","maintainers":[{"name":"philipinho","email":"phil@neuronlabs.co.uk"}],"readme":"# PDF Inspector\n\nFast PDF classification, text extraction, and image extraction for Node.js/Bun. Native Rust performance via [napi-rs](https://napi.rs).\n\nBuilt by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract text from PDF structure where possible, fall back to OCR only when needed.\n\n## Install\n\n```bash\nnpm install firecrawl-pdf-inspector\n# or\nbun add firecrawl-pdf-inspector\n```\n\nPrebuilt binaries included for **linux-x64**, **linux-arm64**, **macOS ARM64**, and **windows-x64**. No Rust toolchain needed.\n\n## API\n\n### `processPdf(buffer: Buffer, pages?: number[]): PdfResult`\n\nFull PDF processing — classify, extract text, and convert to Markdown in one call.\n\n```typescript\nimport { processPdf } from 'firecrawl-pdf-inspector'\nimport { readFileSync } from 'fs'\n\nconst pdf = readFileSync('document.pdf')\nconst result = processPdf(pdf)\n\nconsole.log(result.pdfType)    // \"TextBased\" | \"Scanned\" | \"Mixed\" | \"ImageBased\"\nconsole.log(result.markdown)   // Markdown string or null\nconsole.log(result.pageCount)  // 42\n```\n\n### `classifyPdf(buffer: Buffer): PdfClassification`\n\nClassify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.\n\n```typescript\nimport { classifyPdf } from 'firecrawl-pdf-inspector'\n\nconst result = classifyPdf(readFileSync('document.pdf'))\n\nconsole.log(result.pdfType)         // \"TextBased\" | \"Scanned\" | \"Mixed\" | \"ImageBased\"\nconsole.log(result.pageCount)       // 42\nconsole.log(result.pagesNeedingOcr) // [5, 12, 15] (0-indexed)\nconsole.log(result.confidence)      // 0.875\n```\n\n### `processPdfWithImages(buffer: Buffer, pages?: number[]): PdfResultWithImages`\n\nProcess a PDF and extract both markdown and images in one call. The markdown contains `![image](pdf-image://N)` placeholders where `N` is the index into the returned `images` array.\n\n```typescript\nimport { processPdfWithImages } from 'firecrawl-pdf-inspector'\nimport { readFileSync } from 'fs'\n\nconst result = processPdfWithImages(readFileSync('document.pdf'))\n\nconsole.log(result.images.length)  // 110\nconsole.log(result.markdown)       // \"# Title\\n\\n![image](pdf-image://0)\\n\\nSome text...\"\n\n// Replace placeholders with real URLs after uploading\nlet content = result.markdown\nfor (let i = 0; i < result.images.length; i++) {\n  const img = result.images[i]\n  const ext = img.format === 'Jpeg' ? 'jpg' : 'png'\n  const url = await uploadImage(img.data, `image-${i}.${ext}`)\n  content = content.replace(`pdf-image://${i}`, url)\n}\n```\n\n### `extractImages(buffer: Buffer): ExtractedImage[]`\n\nExtract embedded images from a PDF as raw bytes. Supports JPEG (DCTDecode) and PNG (FlateDecode) images. Returns image data with page position and dimensions.\n\nImage extraction is a separate function from text extraction — calling `processPdf` or `extractText` does **not** pay the cost of image decompression/encoding.\n\n```typescript\nimport { extractImages } from 'firecrawl-pdf-inspector'\nimport { writeFileSync } from 'fs'\n\nconst images = extractImages(readFileSync('document.pdf'))\n\nfor (const img of images) {\n  const ext = img.format === 'Jpeg' ? 'jpg' : 'png'\n  writeFileSync(`page${img.page}_${img.width}x${img.height}.${ext}`, img.data)\n  console.log(`Page ${img.page}: ${img.width}x${img.height} ${img.format}`)\n}\n```\n\n### `extractTextInRegions(buffer: Buffer, pageRegions: PageRegions[]): PageRegionTexts[]`\n\nExtract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR.\n\nEach region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues).\n\n```typescript\nimport { extractTextInRegions } from 'firecrawl-pdf-inspector'\n\nconst result = extractTextInRegions(pdf, [\n  {\n    page: 0, // 0-indexed\n    regions: [\n      [0, 0, 300, 400],    // [x1, y1, x2, y2] in PDF points, top-left origin\n      [300, 0, 612, 400],\n    ]\n  }\n])\n\nfor (const region of result[0].regions) {\n  if (region.needsOcr) {\n    // Unreliable text — send this region to OCR instead\n  } else {\n    console.log(region.text) // Extracted text in reading order\n  }\n}\n```\n\n### `extractText(buffer: Buffer): string`\n\nExtract plain text from a PDF.\n\n```typescript\nimport { extractText } from 'firecrawl-pdf-inspector'\n\nconst text = extractText(readFileSync('document.pdf'))\n```\n\n## Types\n\n```typescript\ninterface PdfResult {\n  pdfType: string          // \"TextBased\" | \"Scanned\" | \"Mixed\" | \"ImageBased\"\n  markdown: string | null  // Markdown output\n  pageCount: number\n  processingTimeMs: number\n  pagesNeedingOcr: number[] // 1-indexed page numbers\n  title: string | null\n  confidence: number        // 0.0 - 1.0\n  isComplexLayout: boolean\n  pagesWithTables: number[]\n  pagesWithColumns: number[]\n  hasEncodingIssues: boolean\n}\n\n// Extends PdfResult with extracted images\ninterface PdfResultWithImages extends PdfResult {\n  images: ExtractedImage[]  // Indices match pdf-image://N placeholders in markdown\n}\n\ninterface PdfClassification {\n  pdfType: string          // \"TextBased\" | \"Scanned\" | \"Mixed\" | \"ImageBased\"\n  pageCount: number\n  pagesNeedingOcr: number[] // 0-indexed page numbers\n  confidence: number        // 0.0 - 1.0\n}\n\ninterface ExtractedImage {\n  page: number              // 1-indexed page number\n  x: number                 // X position on page\n  y: number                 // Y position on page\n  width: number             // Pixel width\n  height: number            // Pixel height\n  format: string            // \"Jpeg\" | \"Png\"\n  data: Buffer              // Raw image bytes (valid JPEG or PNG file)\n}\n\ninterface PageRegions {\n  page: number              // 0-indexed\n  regions: number[][]       // [[x1, y1, x2, y2], ...] in PDF points, top-left origin\n}\n\ninterface PageRegionTexts {\n  page: number\n  regions: RegionText[]\n}\n\ninterface RegionText {\n  text: string\n  needsOcr: boolean         // true when text is unreliable\n}\n```\n\n## Supported image formats\n\n| PDF Filter    | Output Format | Status    |\n|---------------|---------------|-----------|\n| DCTDecode     | JPEG          | Supported |\n| FlateDecode   | PNG           | Supported |\n| JPXDecode     | JPEG2000      | Planned   |\n| CCITTFaxDecode| TIFF          | Planned   |\n\n**Color spaces:** DeviceRGB, DeviceGray, DeviceCMYK (converted to RGB), Indexed, ICCBased, CalRGB, CalGray.\n\n**Note:** Vector graphics (drawn with PDF path operators) are not raster images and cannot be extracted — they would need to be rendered.\n\n## Performance\n\nText extraction and image extraction are independent paths. `processPdf` and `extractText` skip image processing entirely, so there is zero overhead when you only need text.\n\n## Platforms\n\n| Platform | Architecture | Supported |\n|----------|--------------|-----------|\n| Linux    | x64          | Yes       |\n| Linux    | ARM64        | Yes       |\n| macOS    | ARM64        | Yes       |\n| Windows  | x64          | Yes       |\n\n## License\n\nMIT\n","readmeFilename":"README.md"}