{"_id":"@cat_tom/file-content-extractor","_rev":"7-46e2d33d01db2db499f71e87c277bff4","name":"@cat_tom/file-content-extractor","dist-tags":{"latest":"0.1.6"},"versions":{"0.1.0":{"name":"@cat_tom/file-content-extractor","version":"0.1.0","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","_id":"@cat_tom/file-content-extractor@0.1.0","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"dist":{"shasum":"b25c7d00ecda30ead5f940f4b66ffd37526e068a","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.0.tgz","fileCount":18,"integrity":"sha512-wURf3ZL18udLXhS1dHdSEnk3x5j1fkcvIy5/wrecn9JBHMs9V6mGxj+LepR1aP0XaS3ByFUogd4OdBAdweq11A==","signatures":[{"sig":"MEQCIHEYYrXVwD3TbdCGAtgM+bfCSlLxsXcpFf4N1VMRrSWXAiAeX/nb/ymyTayR7kj0kO6caOmLVe4XcmaF/O95ADG7og==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":61723},"main":"./dist/file-content-extractor.cjs","type":"module","types":"./src/index.ts","module":"./dist/file-content-extractor.js","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","preview":"vite preview","build:lib":"vue-tsc --noEmit && vite build --mode lib","type-check":"vue-tsc --noEmit"},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"_npmVersion":"11.12.1","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","directories":{},"sideEffects":false,"_nodeVersion":"24.15.0","_hasShrinkwrap":false,"devDependencies":{"vue":"^3.5.13","vite":"^6.2.1","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","vue-tsc":"^2.2.8","pdfjs-dist":"^3.11.174","typescript":"~5.8.0","@types/node":"^22.13.9","tesseract.js":"^6.0.0","@vitejs/plugin-vue":"^5.2.1"},"peerDependencies":{"vue":"^3.5.0","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/file-content-extractor_0.1.0_1786188313956_0.8307976619516244","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@cat_tom/file-content-extractor","version":"0.1.1","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","_id":"@cat_tom/file-content-extractor@0.1.1","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"homepage":"https://github.com/abc273/file-content-extractor#readme","bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"dist":{"shasum":"4fe1f9e6122523424b57745e5f1e717489e202bf","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.1.tgz","fileCount":18,"integrity":"sha512-YLmPYgkHsXiAViuNfgd3sn9qL9BRiqgUe7CTuFtGE0Q8BjMZMlZf+g6J36OZGiYzy8zTg8eW3rpnXpuCqu1VAw==","signatures":[{"sig":"MEYCIQDsh8rHpNqLCgBMRhUuvepwex6tkh7VKylVg43tkVbDYgIhAOI3PJX5sPw5AZ/8u5rcc3cA01LMHeBuC4DyXAZMoUJe","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":61994},"main":"./dist/file-content-extractor.cjs","type":"module","types":"./src/index.ts","module":"./dist/file-content-extractor.js","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"gitHead":"a203a3d05133fefb11fd38825eae5faf93d0b363","scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","preview":"vite preview","build:lib":"vue-tsc --noEmit && vite build --mode lib","type-check":"vue-tsc --noEmit"},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"repository":{"url":"git+https://github.com/abc273/file-content-extractor.git","type":"git"},"_npmVersion":"11.12.1","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","directories":{},"sideEffects":false,"_nodeVersion":"24.15.0","_hasShrinkwrap":false,"devDependencies":{"vue":"^3.5.13","vite":"^6.2.1","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","vue-tsc":"^2.2.8","pdfjs-dist":"^3.11.174","typescript":"~5.8.0","@types/node":"^22.13.9","tesseract.js":"^6.0.0","@vitejs/plugin-vue":"^5.2.1"},"peerDependencies":{"vue":"^3.5.0","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/file-content-extractor_0.1.1_1786189364910_0.02137925253630657","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@cat_tom/file-content-extractor","version":"0.1.2","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","_id":"@cat_tom/file-content-extractor@0.1.2","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"homepage":"https://github.com/abc273/file-content-extractor#readme","bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"dist":{"shasum":"5fa6edd5bdd01a1ae26af2ff6f4cc727d0d1aae9","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.2.tgz","fileCount":18,"integrity":"sha512-Qz5XLlN6kIZG+w++CxjGzBtaVvH2ZT3Qv59uxS6h6lJhRund8dyJknTXlHUbulO4mbkeG8hgqziiOqPd3hpNtw==","signatures":[{"sig":"MEUCIAWRat8ekh1+4bM1n4RRPsTClBPHTmaGNLMNDScGhcvOAiEAvoWu0GwoGNr7K6tWU9PDqtqBzMN6ad7plWLo4ro9Zm4=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":62983},"main":"./dist/file-content-extractor.cjs","type":"module","types":"./src/index.ts","module":"./dist/file-content-extractor.js","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"gitHead":"128296fae04c9185594e167d6d862282ec0632af","scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","preview":"vite preview","build:lib":"vue-tsc --noEmit && vite build --mode lib","type-check":"vue-tsc --noEmit"},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"repository":{"url":"git+https://github.com/abc273/file-content-extractor.git","type":"git"},"_npmVersion":"11.12.1","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","directories":{},"sideEffects":false,"_nodeVersion":"24.15.0","_hasShrinkwrap":false,"devDependencies":{"vue":"^3.5.13","vite":"^6.2.1","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","vue-tsc":"^2.2.8","pdfjs-dist":"^3.11.174","typescript":"~5.8.0","@types/node":"^22.13.9","tesseract.js":"^6.0.0","@vitejs/plugin-vue":"^5.2.1"},"peerDependencies":{"vue":"^3.5.0","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/file-content-extractor_0.1.2_1786190356734_0.904778398914768","host":"s3://npm-registry-packages-npm-production"}},"0.1.3":{"name":"@cat_tom/file-content-extractor","version":"0.1.3","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","_id":"@cat_tom/file-content-extractor@0.1.3","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"homepage":"https://github.com/abc273/file-content-extractor#readme","bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"dist":{"shasum":"c584882b862bee79a32311e153ff73a06d81e80d","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.3.tgz","fileCount":18,"integrity":"sha512-hxqum2W0b+UrXqxE84gJ1QLz9O8MlhWM/is7VdacnDc18PEKM7u5tdEirN2nIQfRk0Y9FCW05m7NGjuOshyO2g==","signatures":[{"sig":"MEQCIAWV2/cwcYjsxX23L1qldu+RquIWqfNcPrIgCbogapACAiB7f8zkPdk7tTw3nZMcMOZhPjWvuhW8OP83/+PG+9IFXQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":61588},"main":"./dist/file-content-extractor.cjs","type":"module","types":"./src/index.ts","module":"./dist/file-content-extractor.js","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"gitHead":"52a18e7d15ee48187d4210d1d9c5dc1b0a0e2322","scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","preview":"vite preview","build:lib":"vue-tsc --noEmit && vite build --mode lib","type-check":"vue-tsc --noEmit"},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"repository":{"url":"git+https://github.com/abc273/file-content-extractor.git","type":"git"},"_npmVersion":"11.12.1","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","directories":{},"sideEffects":false,"_nodeVersion":"24.15.0","_hasShrinkwrap":false,"devDependencies":{"vue":"^3.5.13","vite":"^6.2.1","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","vue-tsc":"^2.2.8","pdfjs-dist":"^3.11.174","typescript":"~5.8.0","@types/node":"^22.13.9","tesseract.js":"^6.0.0","@vitejs/plugin-vue":"^5.2.1"},"peerDependencies":{"vue":"^3.5.0","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/file-content-extractor_0.1.3_1786190744838_0.7478616775528963","host":"s3://npm-registry-packages-npm-production"}},"0.1.4":{"name":"@cat_tom/file-content-extractor","version":"0.1.4","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","_id":"@cat_tom/file-content-extractor@0.1.4","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"homepage":"https://github.com/abc273/file-content-extractor#readme","bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"dist":{"shasum":"34d8b93314df1cc29ea628404e041d0ab73867fb","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.4.tgz","fileCount":18,"integrity":"sha512-UnfhHYoV+FdNyqKmvTX7tqsraM5Jk2nMgxEres4wmQjvxToZTZg1U7PGY2vH7g4SuaY2tXb6SzjVY3JP/D/rYA==","signatures":[{"sig":"MEYCIQCez+QE/OXLNf7zjg3TBanu4EBXWgJ5ZAsiWuzCTCUWlgIhALmFWpiX52iaBhTfQjZLBZM7Ff9nfSN4+kJRyhYeMtcO","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":65519},"main":"./dist/file-content-extractor.cjs","type":"module","types":"./src/index.ts","module":"./dist/file-content-extractor.js","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"gitHead":"f0fa1889da41ea32f333fd60a01e8ba7d6929c19","scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","preview":"vite preview","build:lib":"vue-tsc --noEmit && vite build --mode lib","type-check":"vue-tsc --noEmit"},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"repository":{"url":"git+https://github.com/abc273/file-content-extractor.git","type":"git"},"_npmVersion":"11.12.1","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","directories":{},"sideEffects":false,"_nodeVersion":"24.15.0","_hasShrinkwrap":false,"devDependencies":{"vue":"^3.5.13","vite":"^6.2.1","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","vue-tsc":"^2.2.8","pdfjs-dist":"^3.11.174","typescript":"~5.8.0","@types/node":"^22.13.9","tesseract.js":"^6.0.0","@vitejs/plugin-vue":"^5.2.1"},"peerDependencies":{"vue":"^3.5.0","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/file-content-extractor_0.1.4_1786192003755_0.5522128017824182","host":"s3://npm-registry-packages-npm-production"}},"0.1.5":{"name":"@cat_tom/file-content-extractor","version":"0.1.5","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","_id":"@cat_tom/file-content-extractor@0.1.5","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"homepage":"https://github.com/abc273/file-content-extractor#readme","bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"dist":{"shasum":"4993d5e605d5fc4cd7c92c4e7d982477f489a0c4","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.5.tgz","fileCount":18,"integrity":"sha512-CVLNy28LktbK22io1D0yiOjI+xAKWIgmZtRilF/wMqOio7SgSRqzZhmihP/UDZRvE956PKd2NTEIMeqzu1Hmnw==","signatures":[{"sig":"MEUCIQDSC4GJ/25/0/mC75R3Y6p8HWK78wTlY0mGW8+c96i+JQIgdTbpkAzNJ8ZtgsU3uHXxDZmS0XcOmHU46/pCL1ez1+A=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":69711},"main":"./dist/file-content-extractor.cjs","type":"module","types":"./src/index.ts","module":"./dist/file-content-extractor.js","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"gitHead":"72a54bbcb3ac406cb54e600dde0d1ec33169adb1","scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","preview":"vite preview","build:lib":"vue-tsc --noEmit && vite build --mode lib","type-check":"vue-tsc --noEmit"},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"repository":{"url":"git+https://github.com/abc273/file-content-extractor.git","type":"git"},"_npmVersion":"11.12.1","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","directories":{},"sideEffects":false,"_nodeVersion":"24.15.0","_hasShrinkwrap":false,"devDependencies":{"vue":"^3.5.13","vite":"^6.2.1","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","vue-tsc":"^2.2.8","pdfjs-dist":"^3.11.174","typescript":"~5.8.0","@types/node":"^22.13.9","tesseract.js":"^6.0.0","@vitejs/plugin-vue":"^5.2.1"},"peerDependencies":{"vue":"^3.5.0","xlsx":"^0.18.5","marked":"^15.0.7","mammoth":"^1.6.0","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/file-content-extractor_0.1.5_1786192038510_0.012955784945462456","host":"s3://npm-registry-packages-npm-production"}},"0.1.6":{"name":"@cat_tom/file-content-extractor","version":"0.1.6","description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","type":"module","main":"./dist/file-content-extractor.cjs","module":"./dist/file-content-extractor.js","types":"./src/index.ts","exports":{".":{"types":"./src/index.ts","import":"./dist/file-content-extractor.js","require":"./dist/file-content-extractor.cjs"}},"sideEffects":false,"scripts":{"dev":"vite","build":"vue-tsc --noEmit && vite build","build:lib":"vue-tsc --noEmit && vite build --mode lib","preview":"vite preview","type-check":"vue-tsc --noEmit"},"keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"author":{"name":"cat_tom"},"license":"MIT","repository":{"type":"git","url":"git+https://github.com/abc273/file-content-extractor.git"},"bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"homepage":"https://github.com/abc273/file-content-extractor#readme","devDependencies":{"@types/node":"^22.13.9","@vitejs/plugin-vue":"^5.2.1","mammoth":"^1.6.0","marked":"^15.0.7","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0","typescript":"~5.8.0","vite":"^6.2.1","vue":"^3.5.13","vue-tsc":"^2.2.8","xlsx":"^0.18.5"},"peerDependencies":{"mammoth":"^1.6.0","marked":"^15.0.7","pdfjs-dist":"^3.11.174","tesseract.js":"^6.0.0","xlsx":"^0.18.5","vue":"^3.5.0"},"peerDependenciesMeta":{"vue":{"optional":true}},"gitHead":"c63bfe1057a45ad60503a8edee6348ceaad66aeb","_id":"@cat_tom/file-content-extractor@0.1.6","_nodeVersion":"24.15.0","_npmVersion":"11.12.1","dist":{"integrity":"sha512-Ylv2tfMxYCBupVid9csbo4Csn8COXnrUzbWxpqSstip/+G7WAGQdIrXxRYTxSh4SQE6ICuo0yAZNeKUGHUCSBA==","shasum":"84d15195cd38d4ba92cff9a1be57e3c4f19625f3","tarball":"https://registry.npmjs.org/@cat_tom/file-content-extractor/-/file-content-extractor-0.1.6.tgz","fileCount":18,"unpackedSize":70638,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQCT97uzlEZKQjJv7SnULmksoZ/U2BWE/pgG01lgvo8GRgIhAJLy3v8FOuMOe5cZ0SBWRmaJX4chlwQ7UU1gHgda7rZc"}]},"_npmUser":{"name":"cat_tom","email":"3100728114@qq.com"},"directories":{},"maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/file-content-extractor_0.1.6_1786192676232_0.16356119287282"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-08T11:25:13.791Z","modified":"2026-08-08T12:37:56.539Z","0.1.0":"2026-08-08T11:25:14.155Z","0.1.1":"2026-08-08T11:42:45.068Z","0.1.2":"2026-08-08T11:59:16.874Z","0.1.3":"2026-08-08T12:05:45.004Z","0.1.4":"2026-08-08T12:26:43.891Z","0.1.5":"2026-08-08T12:27:18.667Z","0.1.6":"2026-08-08T12:37:56.380Z"},"bugs":{"url":"https://github.com/abc273/file-content-extractor/issues"},"author":{"name":"cat_tom"},"license":"MIT","homepage":"https://github.com/abc273/file-content-extractor#readme","keywords":["file","extract","txt","docx","pdf","markdown","ocr","encoding","no-garbled"],"repository":{"type":"git","url":"git+https://github.com/abc273/file-content-extractor.git"},"description":"从 txt / docx / xlsx / pdf / md / 图片 中提取纯文本，内置多编码乱码处理。前端通用 TS 库，Vue/React/原生 JS 都能直接用。","maintainers":[{"name":"cat_tom","email":"3100728114@qq.com"}],"readme":"# file-content-extractor\n\n从常见文档里抽纯文本的前端库：**txt / doc / docx / xls / xlsx / pdf / md / 图片（OCR）**。\n继承自 [sensitive_word_management_system](../sensitive_word_management_system) 里跑通的那套提取算法，\n重点保留「多编码自动探测 + 乱码检测」那一块。\n\n> 适用场景：用户上传文件 → 后端要拿纯文本做关键词匹配 / 分类 / 摘要 / 搜索索引。\n> 在前端先把内容洗干净，能省掉后端一大堆 PDF/Office 解析依赖。\n\n## 支持的格式\n\n| 类别 | 后缀 | 引擎 | 备注 |\n|------|------|------|------|\n| `text` | `.txt` `.csv` `.log` 无后缀 | 内置多编码探测 | UTF-8 / GB18030 / GBK / Big5 / UTF-16 / Shift-JIS / EUC-KR / ISO-8859-1 / KOI8-R 等 |\n| `docx` | `.docx` | mammoth | 旧版 `.doc` 用文本片段扫描兜底（建议转 docx） |\n| `excel` | `.xlsx` `.xls` | xlsx | 过滤纯数字 / 序号列 |\n| `pdf` | `.pdf` | pdfjs-dist | 默认按页提取文本（扫描件请走 OCR） |\n| `markdown` | `.md` `.markdown` | marked | parse → 去 HTML 标签 |\n| `image` | `.png` `.jpg` `.jpeg` `.gif` `.bmp` `.webp` | tesseract.js | 默认 `chi_sim+eng` |\n\n## 安装\n\n> ⚠️ 包名是 **`@cat_tom/file-content-extractor`**（带 scope），不是裸名。\n\n库的解析引擎全部以 `peerDependencies` 声明，**npm 7+ 会自动装上**。按你要处理的格式挑：\n\n### 只处理 txt / md（最轻量）\n\n```bash\nnpm install @cat_tom/file-content-extractor\n```\n\n只装本库和 `marked`，可处理 `.txt` `.csv` `.md` `.markdown` 无后缀文件。\n\n### 加 Word / Excel / PDF / OCR\n\n按需要加装引擎：\n\n```bash\n# Word 文档（.docx）\nnpm install @cat_tom/file-content-extractor mammoth\n\n# Excel（.xlsx / .xls）\nnpm install @cat_tom/file-content-extractor xlsx\n\n# PDF\nnpm install @cat_tom/file-content-extractor pdfjs-dist\n\n# 图片 OCR\nnpm install @cat_tom/file-content-extractor tesseract.js\n```\n\n### 全格式\n\n```bash\nnpm install @cat_tom/file-content-extractor \\\n  mammoth \\\n  xlsx \\\n  pdfjs-dist \\\n  tesseract.js\n```\n\n### 验证装好了\n\n```bash\nnode -e \"import('@cat_tom/file-content-extractor').then(m => console.log(typeof m.extractFileContent))\"\n# 期望输出: function\n```\n\n> 用 pnpm / yarn 也一样，`pnpm add` / `yarn add` 同样能自动装 peer。\n\n## 最小用法（任意 JS 框架）\n\n```ts\nimport { extractFileContent } from '@cat_tom/file-content-extractor'\n\n// 在 <input type=\"file\"> 的 change 事件里：\nconst file = e.target.files[0]\nconst result = await extractFileContent(file)\n\nif (result.success) {\n  console.log(result.content)   // 纯文本\n  console.log(result.category)  // 'pdf' | 'docx' | ...\n} else {\n  console.error(result.error)\n}\n```\n\n## Vue 3 用法\n\n```vue\n<script setup lang=\"ts\">\nimport { ref } from 'vue'\nimport FileContentExtractor, { extractFileContent } from '@cat_tom/file-content-extractor'\nimport type { ExtractResult } from '@cat_tom/file-content-extractor'\n\nconst text = ref('')\nconst onResult = (r: ExtractResult) => {\n  if (r.success) text.value = r.content\n}\n</script>\n\n<template>\n  <FileContentExtractor :multiple=\"true\" @result=\"onResult\" />\n  <pre>{{ text }}</pre>\n</template>\n```\n\n## 批量\n\n```ts\nimport { extractFilesContent } from '@cat_tom/file-content-extractor'\n\nconst results = await extractFilesContent(files, { /* options */ }, 4)\n// 并发 4 个，结果按完成顺序追加\n```\n\n## 选项\n\n```ts\ninterface ExtractorOptions {\n  /** OCR 识别语言，默认 'chi_sim+eng'；只要英文可以传 'eng' 省一半下载量 */\n  ocrLanguages?: string\n  /** OCR 进度回调 */\n  onOcrProgress?: (info: { status: string; progress: number }) => void\n  /**\n   * PDF.js worker 源。\n   * 推荐用 cdnjs：`https://cdnjs.cloudflare.com/ajax/libs/pdf.js/3.11.174/pdf.worker.min.js`\n   * 不传也能跑，但会卡主线程。\n   */\n  pdfWorkerSrc?: string\n  /** 文本最大返回字符数，防止大文件把页面卡死。默认 5MB，0 表示不限制。 */\n  maxContentLength?: number\n}\n```\n\n## PDF.js worker 配置\n\nVite 项目里最简单的做法 —— 在 `index.html` 加：\n\n```html\n<script type=\"module\">\n  import { GlobalWorkerOptions } from 'pdfjs-dist'\n  GlobalWorkerOptions.workerSrc = 'https://cdnjs.cloudflare.com/ajax/libs/pdf.js/3.11.174/pdf.worker.min.js'\n</script>\n```\n\n或者在调用 `extractFileContent` 时传 `pdfWorkerSrc`：\n\n```ts\nawait extractFileContent(file, {\n  pdfWorkerSrc: 'https://cdnjs.cloudflare.com/ajax/libs/pdf.js/3.11.174/pdf.worker.min.js',\n})\n```\n\n## 关于\"无乱码\"\n\n`extractFromText`（在 `src/utils/encoding.ts`）做了三件事保证中文不乱码：\n\n1. **多编码顺序尝试** —— UTF-8 → GB18030 → GBK → Big5 → UTF-16 ...\n2. **乱码检测** —— 检测控制字符、U+FFFD 替换符、BOM、连续控制序列、非 ASCII 比例\n3. **评分挑最佳** —— 候选编码解码后按\"有效字符比例 + 中文加权 + 有意义模式数\"打分\n\n跑过的格式：UTF-8 / GB18030 / GBK / Big5 / UTF-16 LE/BE / Shift-JIS / EUC-JP / EUC-KR / ISO-8859-1 / windows-1252 / KOI8-R / ISO-8859-5 / windows-1251。\n\n## 调试\n\n```ts\nconst r = await extractFileContent(file)\nconsole.log({\n  category: r.category,    // 识别到的文件类别\n  encoding: r.encoding,    // 文本实际用的编码\n  size: r.size,\n  success: r.success,\n})\n```\n\n## 本地跑 demo\n\n```bash\nnpm install\nnpm run dev\n# 打开 http://localhost:5174/demo/\n```\n\n## 打包\n\n```bash\nnpm run build:lib\n# 产物在 dist/：file-content-extractor.js / .cjs / .umd + .d.ts\n```\n\n## 项目结构\n\n```\nsrc/\n├── index.ts                     # 主入口，dispatcher\n├── FileContentExtractor.vue     # 可选 Vue 3 组件\n├── types.ts                     # TS 类型\n├── utils/\n│   ├── fileType.ts              # 文件后缀识别\n│   └── encoding.ts              # 多编码探测、评分、乱码检测\n└── extractors/\n    ├── textExtractor.ts         # txt\n    ├── binaryExtractor.ts       # 兜底（无后缀、.doc 等）\n    ├── docxExtractor.ts         # .docx (mammoth)\n    ├── docExtractor.ts          # .doc 尽力而为\n    ├── excelExtractor.ts        # .xlsx/.xls (xlsx)\n    ├── pdfExtractor.ts          # .pdf (pdfjs-dist)\n    ├── markdownExtractor.ts     # .md (marked)\n    └── imageExtractor.ts        # 图片 OCR (tesseract.js)\n```\n\n## License\n\nMIT\n","readmeFilename":"README.md"}