{"_id":"expo-pdf-text-extract","_rev":"3-a66e1b0d8502da68c70134375348da48","name":"expo-pdf-text-extract","dist-tags":{"latest":"1.1.0"},"versions":{"1.0.0":{"name":"expo-pdf-text-extract","version":"1.0.0","keywords":["react-native","expo","pdf","pdf-extraction","pdf-text","text-extraction","native-module","expo-module","pdfkit","pdfbox","document-processing","ocr-alternative","ios","android","expo-module","pdf-parser","pdf-reader","document-scanner"],"author":{"name":"Pathik Gandhi","email":"pathikgandhi@gmail.com"},"license":"MIT","_id":"expo-pdf-text-extract@1.0.0","maintainers":[{"name":"gr8pathik","email":"gr8pathik@gmail.com"}],"homepage":"https://github.com/gr8pathik/expo-pdf-text-extract#readme","bugs":{"url":"https://github.com/gr8pathik/expo-pdf-text-extract/issues"},"dist":{"shasum":"c5cbb8187f1b2c9b6cf662a95bd2f7a1757c1f5c","tarball":"https://registry.npmjs.org/expo-pdf-text-extract/-/expo-pdf-text-extract-1.0.0.tgz","fileCount":16,"integrity":"sha512-7ZZzRb/ZWjsgzy9bZg7xQbysUx+fkHZBLThI11/f8pni6jY0TaRlmrras5HmS14IC9lEmlUNmu44Eb4p7kdNRA==","signatures":[{"sig":"MEQCIBAPuqMMxGzNXzOh/Tk8/MAbMMO2MsNV5uxRB99hn7G5AiBkJdv3nmp8S7MYpqWDUUyxgvCA9rXTAay/72NsIXxDmg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":68591},"main":"dist/index.js","types":"dist/index.d.ts","module":"dist/index.js","source":"src/index.ts","engines":{"node":">=16.0.0"},"gitHead":"c9b6f7610389ff090d02c18a7173de067a608bfc","scripts":{"build":"tsc","clean":"rm -rf dist","prepublishOnly":"npm run build"},"_npmUser":{"name":"gr8pathik","email":"gr8pathik@gmail.com"},"repository":{"url":"git+https://github.com/gr8pathik/expo-pdf-text-extract.git","type":"git"},"_npmVersion":"10.8.1","description":"Native PDF text extraction for React Native and Expo. Extract text content from PDF files using platform-native APIs (PDFKit on iOS, PDFBox on Android). Works with Expo development builds.","directories":{},"_nodeVersion":"22.3.0","_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.0.0","expo-modules-core":"~2.4.1"},"peerDependencies":{"expo":">=49.0.0","react":">=18.0.0","react-native":">=0.72.0"},"_npmOperationalInternal":{"tmp":"tmp/expo-pdf-text-extract_1.0.0_1768412078532_0.9519988203573937","host":"s3://npm-registry-packages-npm-production"}},"1.0.1":{"name":"expo-pdf-text-extract","version":"1.0.1","keywords":["react-native","expo","pdf","pdf-extraction","pdf-text","text-extraction","native-module","expo-module","pdfkit","pdfbox","document-processing","ocr-alternative","ios","android","expo-module","pdf-parser","pdf-reader","document-scanner"],"author":{"name":"Pathik Gandhi","email":"pathikgandhi@gmail.com"},"license":"MIT","_id":"expo-pdf-text-extract@1.0.1","maintainers":[{"name":"gr8pathik","email":"gr8pathik@gmail.com"}],"homepage":"https://github.com/gr8pathik/expo-pdf-text-extract#readme","bugs":{"url":"https://github.com/gr8pathik/expo-pdf-text-extract/issues"},"dist":{"shasum":"747b8cb8bdc64d994290d10e08d97c4262d7a3bd","tarball":"https://registry.npmjs.org/expo-pdf-text-extract/-/expo-pdf-text-extract-1.0.1.tgz","fileCount":16,"integrity":"sha512-5ytTRDZIL+exa9eqKw+kbntqknxFpN+WqH88le84RnEXJBGH7WbY1ubNnyuARrM2KCxLty2tJZ2UyLrQjLb5vA==","signatures":[{"sig":"MEUCIQCitruMnWVeCOuxlCA4wFKAYBhk/zcutNxU2iwLRaYm3AIgMFbaXimr+HdP47UCTRsdUu2HsxwQv1UCYKbylRU8Ybk=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":68630},"main":"dist/index.js","types":"dist/index.d.ts","module":"dist/index.js","source":"src/index.ts","engines":{"node":">=16.0.0"},"gitHead":"a85a3a003ff91b8bfb8c0996271a9b677d1ece8d","scripts":{"build":"tsc","clean":"rm -rf dist","prepublishOnly":"npm run build"},"_npmUser":{"name":"gr8pathik","email":"gr8pathik@gmail.com"},"repository":{"url":"git+https://github.com/gr8pathik/expo-pdf-text-extract.git","type":"git"},"_npmVersion":"10.8.1","description":"Native PDF text extraction for React Native and Expo. Extract text content from PDF files using platform-native APIs (PDFKit on iOS, PDFBox on Android). Works with Expo development builds.","directories":{},"_nodeVersion":"22.3.0","_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.0.0","expo-modules-core":"~2.4.1"},"peerDependencies":{"expo":">=49.0.0","react":">=18.0.0","react-native":">=0.72.0"},"_npmOperationalInternal":{"tmp":"tmp/expo-pdf-text-extract_1.0.1_1771736775022_0.2568715481598147","host":"s3://npm-registry-packages-npm-production"}},"1.1.0":{"name":"expo-pdf-text-extract","version":"1.1.0","description":"Native PDF text extraction for React Native and Expo. Extract text content from PDF files using platform-native APIs (PDFKit on iOS, PDFBox on Android). Works with Expo development builds.","main":"dist/index.js","module":"dist/index.js","types":"dist/index.d.ts","source":"src/index.ts","scripts":{"build":"tsc","prepublishOnly":"npm run build","clean":"rm -rf dist","test":"jest"},"keywords":["react-native","expo","pdf","pdf-extraction","pdf-text","text-extraction","native-module","expo-module","pdfkit","pdfbox","document-processing","ocr-alternative","ios","android","expo-module","pdf-parser","pdf-reader","document-scanner"],"author":{"name":"Pathik Gandhi","email":"pathikgandhi@gmail.com"},"license":"MIT","repository":{"type":"git","url":"git+https://github.com/gr8pathik/expo-pdf-text-extract.git"},"homepage":"https://github.com/gr8pathik/expo-pdf-text-extract#readme","bugs":{"url":"https://github.com/gr8pathik/expo-pdf-text-extract/issues"},"engines":{"node":">=16.0.0"},"peerDependencies":{"expo":">=49.0.0","react":">=18.0.0","react-native":">=0.72.0"},"devDependencies":{"@types/jest":"^29.5.12","expo-modules-core":"~2.4.1","jest":"^29.7.0","ts-jest":"^29.1.2","typescript":"^5.0.0"},"jest":{"preset":"ts-jest","testEnvironment":"node","testMatch":["<rootDir>/__tests__/**/*.test.ts"]},"gitHead":"8ed071e120158dcfa91a18e8378c4113ae9dd55c","_id":"expo-pdf-text-extract@1.1.0","_nodeVersion":"24.15.0","_npmVersion":"11.12.1","dist":{"integrity":"sha512-LH0qNUZBTdIZ4EJUOHF+akqP0IgrlnLFM3skVlZxvyZFHJeU7LFGnr6eOSmRZ2PpTFUYsbs+aZSKmUdWPeSCYw==","shasum":"c1dff5afe8532aeec7723d94bcde88a665f39eed","tarball":"https://registry.npmjs.org/expo-pdf-text-extract/-/expo-pdf-text-extract-1.1.0.tgz","fileCount":16,"unpackedSize":91710,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIEOeoNQpUmI7ePQCCBtueSI+dFyqW23vBJv08JWQxlVCAiApLSRHhvYwwTDAR1OpUTpMJIYFCshNNnj6lk0BNGxR7w=="}]},"_npmUser":{"name":"gr8pathik","email":"gr8pathik@gmail.com"},"directories":{},"maintainers":[{"name":"gr8pathik","email":"gr8pathik@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/expo-pdf-text-extract_1.1.0_1779269735502_0.04192876668308676"},"_hasShrinkwrap":false}},"time":{"created":"2026-01-14T17:34:38.531Z","modified":"2026-05-20T09:35:35.747Z","1.0.0":"2026-01-14T17:34:38.709Z","1.0.1":"2026-02-22T05:06:15.184Z","1.1.0":"2026-05-20T09:35:35.658Z"},"bugs":{"url":"https://github.com/gr8pathik/expo-pdf-text-extract/issues"},"author":{"name":"Pathik Gandhi","email":"pathikgandhi@gmail.com"},"license":"MIT","homepage":"https://github.com/gr8pathik/expo-pdf-text-extract#readme","keywords":["react-native","expo","pdf","pdf-extraction","pdf-text","text-extraction","native-module","expo-module","pdfkit","pdfbox","document-processing","ocr-alternative","ios","android","expo-module","pdf-parser","pdf-reader","document-scanner"],"repository":{"type":"git","url":"git+https://github.com/gr8pathik/expo-pdf-text-extract.git"},"description":"Native PDF text extraction for React Native and Expo. Extract text content from PDF files using platform-native APIs (PDFKit on iOS, PDFBox on Android). Works with Expo development builds.","maintainers":[{"name":"gr8pathik","email":"gr8pathik@gmail.com"}],"readme":"# expo-pdf-text-extract\n\nNative PDF text extraction for React Native and Expo. Extract text content from PDF files using platform-native APIs - no OCR needed for digital PDFs.\n\n[![npm version](https://img.shields.io/npm/v/expo-pdf-text-extract.svg)](https://www.npmjs.com/package/expo-pdf-text-extract)\n[![license](https://img.shields.io/npm/l/expo-pdf-text-extract.svg)](https://github.com/gr8pathik/expo-pdf-text-extract/blob/main/LICENSE)\n[![platforms](https://img.shields.io/badge/platforms-iOS%20%7C%20Android-lightgrey.svg)](https://reactnative.dev/)\n\n## Features\n\n- **Native Performance** - Uses PDFKit (iOS) and PDFBox (Android) for fast, reliable extraction\n- **No OCR Required** - Extracts embedded text directly from digital PDFs\n- **Password-Protected PDFs** - First-class support for encrypted PDFs on both platforms\n- **Expo Compatible** - Works with Expo development builds (SDK 49+)\n- **TypeScript Support** - Full type definitions included\n- **Simple API** - Just one function to extract text\n- **Page-level Control** - Extract from specific pages or get page count\n- **Multiple Path Formats** - Supports `file://`, `content://`, and absolute paths\n\n## When to Use This\n\n| Scenario | This Package | Alternative |\n|----------|-------------|-------------|\n| Digital PDFs (from email, downloads) | Yes | - |\n| Scanned PDFs (images of paper) | No | Use OCR library |\n| Need text content only | Yes | - |\n| Need to render/view PDF | No | Use react-native-pdf |\n| Expo Go | No | Requires dev build |\n\n## Requirements\n\n- **Expo SDK**: 49.0.0 or higher\n- **React Native**: 0.72.0 or higher\n- **iOS**: 15.1 or higher\n- **Android**: API 21 (Lollipop) or higher\n\n> **Important**: This package requires an Expo development build. It will not work in Expo Go.\n\n## Installation\n\n### Using Expo\n\n```bash\nnpx expo install expo-pdf-text-extract\n```\n\n### Using npm/yarn\n\n```bash\nnpm install expo-pdf-text-extract\n# or\nyarn add expo-pdf-text-extract\n```\n\n### Create Development Build\n\nSince this is a native module, you need to create a development build:\n\n```bash\n# For iOS\nnpx expo run:ios\n\n# For Android\nnpx expo run:android\n\n# Or create a development build\neas build --profile development --platform all\n```\n\n## Quick Start\n\n```typescript\nimport { extractText, isAvailable } from 'expo-pdf-text-extract';\n\n// Check if native module is available\nif (isAvailable()) {\n  // Extract text from a PDF file\n  const text = await extractText('/path/to/document.pdf');\n  console.log(text);\n}\n```\n\n## API Reference\n\n### `isAvailable()`\n\nCheck if the native PDF extractor is available.\n\n```typescript\nfunction isAvailable(): boolean\n```\n\nReturns `false` when:\n- Running in Expo Go\n- Native module failed to load\n- Platform not supported\n\n**Example:**\n```typescript\nimport { isAvailable } from 'expo-pdf-text-extract';\n\nif (isAvailable()) {\n  // Show PDF upload option\n} else {\n  // Show message: \"PDF extraction requires a development build\"\n}\n```\n\n### `extractText(filePath, password?)`\n\nExtract all text from a PDF file.\n\n```typescript\nfunction extractText(filePath: string, password?: string): Promise<string>\n```\n\n**Parameters:**\n- `filePath` - Path to the PDF file. Supports:\n  - `file:///path/to/file.pdf` - File URI\n  - `/absolute/path/to/file.pdf` - Absolute path\n  - `content://...` - Content URI (Android document picker)\n- `password` *(optional)* - Password for encrypted PDFs. Omit for clear PDFs.\n\n**Returns:** Promise resolving to extracted text\n\n**Throws:** an `Error` with a stable `.code`:\n- `'PASSWORD_REQUIRED'` - PDF is encrypted and no password was supplied\n- `'INCORRECT_PASSWORD'` - The supplied password does not unlock the PDF\n- generic error if file not found or PDF is invalid / corrupted\n- error if native module not available (Expo Go)\n\n**Example:**\n```typescript\nimport { extractText } from 'expo-pdf-text-extract';\nimport * as DocumentPicker from 'expo-document-picker';\n\n// Pick a PDF file\nconst result = await DocumentPicker.getDocumentAsync({\n  type: 'application/pdf',\n});\n\nif (!result.canceled) {\n  const text = await extractText(result.assets[0].uri);\n  console.log('Extracted text:', text);\n}\n```\n\n### `getPageCount(filePath, password?)`\n\nGet the number of pages in a PDF.\n\n```typescript\nfunction getPageCount(filePath: string, password?: string): Promise<number>\n```\n\nThrows `PASSWORD_REQUIRED` / `INCORRECT_PASSWORD` for encrypted PDFs without a\nvalid password (same error semantics as `extractText`).\n\n**Example:**\n```typescript\nimport { getPageCount } from 'expo-pdf-text-extract';\n\nconst pages = await getPageCount('/path/to/document.pdf');\nconsole.log(`PDF has ${pages} pages`);\n```\n\n### `extractTextFromPage(filePath, pageNumber, password?)`\n\nExtract text from a specific page.\n\n```typescript\nfunction extractTextFromPage(\n  filePath: string,\n  pageNumber: number,\n  password?: string,\n): Promise<string>\n```\n\n**Parameters:**\n- `filePath` - Path to the PDF file\n- `pageNumber` - Page number (1-indexed, first page is 1)\n- `password` *(optional)* - Password for encrypted PDFs\n\n**Example:**\n```typescript\nimport { extractTextFromPage, getPageCount } from 'expo-pdf-text-extract';\n\n// Extract text from first page only\nconst firstPageText = await extractTextFromPage('/path/to/document.pdf', 1);\n\n// Extract text from each page separately\nconst pageCount = await getPageCount('/path/to/document.pdf');\nfor (let i = 1; i <= pageCount; i++) {\n  const pageText = await extractTextFromPage('/path/to/document.pdf', i);\n  console.log(`Page ${i}:`, pageText);\n}\n```\n\n### `extractTextWithInfo(filePath, password?)`\n\nExtract text with additional metadata. This is the **non-throwing variant** —\npassword failures and other errors are returned as data rather than thrown.\n\n```typescript\nfunction extractTextWithInfo(\n  filePath: string,\n  password?: string,\n): Promise<{\n  text: string;\n  pageCount: number;\n  success: boolean;\n  isEncrypted: boolean;          // true if the PDF declared encryption\n  passwordRequired?: boolean;    // true if the call failed because of password\n  error?: string;\n  errorCode?:\n    | 'PASSWORD_REQUIRED'\n    | 'INCORRECT_PASSWORD'\n    | 'FILE_NOT_FOUND'\n    | 'CORRUPT_PDF'\n    | 'UNKNOWN';\n}>\n```\n\n**Example:**\n```typescript\nimport { extractTextWithInfo } from 'expo-pdf-text-extract';\n\nconst result = await extractTextWithInfo('/path/to/document.pdf');\n\nif (result.success) {\n  console.log(`Extracted ${result.text.length} chars from ${result.pageCount} pages`);\n} else if (result.passwordRequired) {\n  // prompt user for a password and retry with extractTextWithInfo(uri, pwd)\n} else {\n  console.error('Extraction failed:', result.error, result.errorCode);\n}\n```\n\n### `isPasswordProtected(filePath)`\n\nDetect whether a PDF actually requires a password to read.\n\n```typescript\nfunction isPasswordProtected(filePath: string): Promise<boolean>\n```\n\nReturns `true` only if the PDF cannot be opened without a password. PDFs that\ndeclare encryption but unlock with an empty password return `false` — they can\nbe read by `extractText` without supplying a password.\n\n## Password-Protected PDFs\n\n`extractText`, `getPageCount`, and `extractTextFromPage` accept an optional\n`password` parameter. If the PDF is encrypted and no password (or the wrong\npassword) is provided, the call throws an `Error` with a stable `.code`:\n\n| `.code`              | Meaning                                            |\n| -------------------- | -------------------------------------------------- |\n| `PASSWORD_REQUIRED`  | PDF is encrypted and no password was supplied      |\n| `INCORRECT_PASSWORD` | Supplied password does not unlock the PDF          |\n\nUse `isPasswordProtected(filePath)` for a fast detection check before\nprompting the user for a password.\n\n### Example\n\n```typescript\nimport {\n  isPasswordProtected,\n  extractText,\n} from 'expo-pdf-text-extract';\n\nasync function readPdf(uri: string) {\n  if (await isPasswordProtected(uri)) {\n    const password = await promptUserForPassword();\n    try {\n      return await extractText(uri, password);\n    } catch (e: any) {\n      if (e.code === 'INCORRECT_PASSWORD') {\n        // ask the user again\n        return readPdf(uri);\n      }\n      throw e;\n    }\n  }\n  return extractText(uri);\n}\n```\n\nIf you prefer error-as-data over try/catch, use `extractTextWithInfo` — it\nnever throws on password issues and returns `passwordRequired: true` plus an\n`errorCode` instead.\n\n## Usage with Document Picker\n\n```typescript\nimport { extractText, isAvailable } from 'expo-pdf-text-extract';\nimport * as DocumentPicker from 'expo-document-picker';\n\nasync function handlePdfUpload() {\n  // Check if extraction is available\n  if (!isAvailable()) {\n    Alert.alert(\n      'Not Available',\n      'PDF extraction requires a development build. Please rebuild the app.'\n    );\n    return;\n  }\n\n  // Pick PDF file\n  const result = await DocumentPicker.getDocumentAsync({\n    type: 'application/pdf',\n    copyToCacheDirectory: true,\n  });\n\n  if (result.canceled) {\n    return;\n  }\n\n  try {\n    // Extract text\n    const text = await extractText(result.assets[0].uri);\n\n    // Use the extracted text\n    console.log('Extracted text:', text.substring(0, 500));\n\n    // Parse the text, search for patterns, etc.\n    const hasKeyword = text.includes('invoice');\n\n  } catch (error) {\n    Alert.alert('Error', `Failed to extract text: ${error.message}`);\n  }\n}\n```\n\n## Error Handling\n\n```typescript\nimport { extractText, isAvailable } from 'expo-pdf-text-extract';\n\nasync function safeExtract(filePath: string): Promise<string | null> {\n  // Check availability first\n  if (!isAvailable()) {\n    console.warn('PDF extraction not available');\n    return null;\n  }\n\n  try {\n    return await extractText(filePath);\n  } catch (error) {\n    if (error.message.includes('not found')) {\n      console.error('File not found:', filePath);\n    } else if (error.message.includes('PDF_LOAD_ERROR')) {\n      console.error('Invalid or corrupted PDF');\n    } else {\n      console.error('Extraction failed:', error.message);\n    }\n    return null;\n  }\n}\n```\n\n## Platform Differences\n\n### iOS (PDFKit)\n- Uses Apple's native PDFKit framework\n- Built into iOS, no additional dependencies\n- Excellent support for standard PDF formats\n- Minimum iOS version: 15.1\n\n### Android (PDFBox)\n- Uses Apache PDFBox (Android port)\n- Text is sorted by position on page for better readability\n- Handles compressed PDF streams (FlateDecode, etc.)\n- Minimum API level: 21\n\n## Troubleshooting\n\n### \"PDF extraction is not available\"\n\nThis error occurs when running in Expo Go. Solution:\n\n```bash\n# Create a development build\nnpx expo run:ios\n# or\nnpx expo run:android\n```\n\n### Empty text returned\n\nIf `extractText()` returns empty string:\n1. **Scanned PDF** - The PDF contains images, not text. Use OCR instead.\n2. **Corrupted PDF** - Try opening the PDF in another app to verify it's valid.\n\n> Password-protected PDFs no longer return empty text — they throw an `Error`\n> with `.code === 'PASSWORD_REQUIRED'`. See [Password-Protected PDFs](#password-protected-pdfs).\n\n### Slow extraction on large PDFs\n\nFor PDFs with many pages, consider:\n1. Extract page by page using `extractTextFromPage()`\n2. Show progress indicator to users\n3. Process in background using a worker\n\n## Performance\n\n| PDF Size | Pages | Extraction Time (approx) |\n|----------|-------|--------------------------|\n| Small    | 1-5   | < 100ms |\n| Medium   | 10-50 | 100-500ms |\n| Large    | 100+  | 500ms-2s |\n\n*Times measured on iPhone 13 and Pixel 6*\n\n## Contributing\n\nContributions are welcome! Please read our contributing guidelines before submitting PRs.\n\n1. Fork the repository\n2. Create your feature branch (`git checkout -b feature/amazing-feature`)\n3. Commit your changes (`git commit -m 'Add amazing feature'`)\n4. Push to the branch (`git push origin feature/amazing-feature`)\n5. Open a Pull Request\n\n## License\n\nMIT License - see [LICENSE](LICENSE) for details.\n\n## Credits\n\n- iOS implementation uses Apple's [PDFKit](https://developer.apple.com/documentation/pdfkit)\n- Android implementation uses [PDFBox-Android](https://github.com/TomRoush/PdfBox-Android) by Tom Roush\n\n## Related Packages\n\n- [expo-document-picker](https://docs.expo.dev/versions/latest/sdk/document-picker/) - Pick documents from device\n- [react-native-pdf](https://github.com/wonday/react-native-pdf) - Display PDFs (viewing, not extraction)\n- [pdf-lib](https://pdf-lib.js.org/) - Create and modify PDFs in JavaScript\n","readmeFilename":"README.md"}