{"_id":"@cliftonc/pptx-to-json","_rev":"5-94b2ee5f1e042c75340875c1281ba6f6","name":"@cliftonc/pptx-to-json","dist-tags":{"latest":"0.2.1"},"versions":{"0.1.0":{"name":"@cliftonc/pptx-to-json","version":"0.1.0","_id":"@cliftonc/pptx-to-json@0.1.0","maintainers":[{"name":"cliftonc","email":"clifton.cunningham@gmail.com"}],"dist":{"shasum":"dc1ff1c76abdacf727907511e4d8131fbe772473","tarball":"https://registry.npmjs.org/@cliftonc/pptx-to-json/-/pptx-to-json-0.1.0.tgz","fileCount":70,"integrity":"sha512-pgzTyDb29w18lIf81iBBk91liTlLfpgvnOswk1OHc9tZYhEz5c57KSSMEALFJB0nu5xiUCVXYt54RtfSyIPXtg==","signatures":[{"sig":"MEUCIQDbaqyVKp6182NdqLgGE7facivLC+lU+TSKi8QkQwVvawIgQu1TXmdfu/9JwSKMyReuCXfGm67i6RpoGCCtL7RAUjM=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":434930},"main":"./dist/index.js","type":"module","_from":"file:cliftonc-pptx-to-json-0.1.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.js"}},"scripts":{"test":"vitest","build":"tsc","log-bin":"tsx src/scripts/log-paste-bin.ts","test:run":"vitest run","test:watch":"vitest --watch","type-check":"tsc --noEmit","build:watch":"tsc --watch","add-test-case":"node src/scripts/add-test-case.js","test:coverage":"vitest --coverage","validate-samples":"node src/scripts/validate-samples.js","download-fixtures":"node src/scripts/download-fixtures.js","regenerate-test-cases":"tsx src/scripts/regenerate-test-cases.ts"},"_npmUser":{"name":"cliftonc","email":"clifton.cunningham@gmail.com"},"_resolved":"/private/var/folders/nc/1sx8wk497kbdpgdxllrndr4w0000gn/T/5a8a6e9f36f1b0209213fe70aa20b007/cliftonc-pptx-to-json-0.1.0.tgz","_integrity":"sha512-pgzTyDb29w18lIf81iBBk91liTlLfpgvnOswk1OHc9tZYhEz5c57KSSMEALFJB0nu5xiUCVXYt54RtfSyIPXtg==","_npmVersion":"10.9.0","directories":{},"_nodeVersion":"22.12.0","dependencies":{"jszip":"^3.10.1","fast-xml-parser":"^5.2.5"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.20.5","vitest":"^1.6.0","ts-node":"^10.9.2","typescript":"^5.9.2","@types/node":"^22.0.0","fracturedjsonjs":"^2.2.1","@vitest/coverage-v8":"^1.6.0"},"optionalDependencies":{"node-fetch":"^3.3.2"},"_npmOperationalInternal":{"tmp":"tmp/pptx-to-json_0.1.0_1757481044058_0.9246822702982322","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@cliftonc/pptx-to-json","version":"0.1.1","_id":"@cliftonc/pptx-to-json@0.1.1","maintainers":[{"name":"cliftonc","email":"clifton.cunningham@gmail.com"}],"dist":{"shasum":"626adb702cb439352912226985a5d0b90d83411f","tarball":"https://registry.npmjs.org/@cliftonc/pptx-to-json/-/pptx-to-json-0.1.1.tgz","fileCount":71,"integrity":"sha512-S9s1R1Gn0IeZauZqQGVtZSmy28kKECEFwG1vF5OLYc2/UYfwrg17ezHR/FhozL3uw6UujoXuIpy62kKooCLzng==","signatures":[{"sig":"MEUCIQDQ3tc5hBiP7yAzoA+/W9ZcJ0y/Eu81MW2YCQkKUTx9jQIgdoeOPlYwgVFiAhpeb1PQX9adiM8qIQAKY7HO7ZfgkJ0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":458283},"main":"./dist/index.js","type":"module","_from":"file:cliftonc-pptx-to-json-0.1.1.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.js"}},"scripts":{"test":"vitest","build":"tsc","log-bin":"tsx src/scripts/log-paste-bin.ts","test:run":"vitest run","test:watch":"vitest --watch","type-check":"tsc --noEmit","build:watch":"tsc --watch","add-test-case":"node src/scripts/add-test-case.js","test:coverage":"vitest --coverage","validate-samples":"node src/scripts/validate-samples.js","download-fixtures":"node src/scripts/download-fixtures.js","regenerate-test-cases":"tsx src/scripts/regenerate-test-cases.ts"},"_npmUser":{"name":"cliftonc","email":"clifton.cunningham@gmail.com"},"_resolved":"/private/var/folders/nc/1sx8wk497kbdpgdxllrndr4w0000gn/T/35336146cf6254a648eb492b95560472/cliftonc-pptx-to-json-0.1.1.tgz","_integrity":"sha512-S9s1R1Gn0IeZauZqQGVtZSmy28kKECEFwG1vF5OLYc2/UYfwrg17ezHR/FhozL3uw6UujoXuIpy62kKooCLzng==","_npmVersion":"10.9.0","description":"A TypeScript library for parsing PowerPoint files (PPTX) and PowerPoint clipboard data into structured JSON. Extracts components like text, shapes, images, and tables with their positioning, styling, and content information.","directories":{},"_nodeVersion":"22.12.0","dependencies":{"jszip":"^3.10.1","fast-xml-parser":"^5.2.5"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.20.5","vitest":"^1.6.0","ts-node":"^10.9.2","typescript":"^5.9.2","@types/node":"^22.0.0","fracturedjsonjs":"^2.2.1","@vitest/coverage-v8":"^1.6.0"},"optionalDependencies":{"node-fetch":"^3.3.2"},"_npmOperationalInternal":{"tmp":"tmp/pptx-to-json_0.1.1_1757481593217_0.33974090370768506","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@cliftonc/pptx-to-json","version":"0.2.0","keywords":["pptx","powerpoint","ooxml","parser","json","slides"],"author":{"name":"Clifton Cunningham"},"license":"MIT","_id":"@cliftonc/pptx-to-json@0.2.0","maintainers":[{"name":"cliftonc","email":"clifton.cunningham@gmail.com"}],"homepage":"https://github.com/cliftonc/pptx-to-json#readme","bugs":{"url":"https://github.com/cliftonc/pptx-to-json/issues"},"dist":{"shasum":"1a0ceffd6e22f1cc9d92878e1a29fad96d367697","tarball":"https://registry.npmjs.org/@cliftonc/pptx-to-json/-/pptx-to-json-0.2.0.tgz","fileCount":46,"integrity":"sha512-aDqBQWLs+MFKpH7pAN9aS65EjpKax6W9X3WEctAvPShJXT6XfXf7UO1sX8BxhiynSck6RgYS32EFFd1C4g9akg==","signatures":[{"sig":"MEUCIHssmdAeABom2zVafDjYKKnwnqRQGwnSifHR7UQwtxqRAiEA0yVlXKMuN6c5vbkEUhkS5ukC4QwRKvVa68X07nlW+to=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":343653},"main":"./dist/index.js","type":"module","_from":"file:cliftonc-pptx-to-json-0.2.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.js"}},"scripts":{"test":"vitest","build":"tsc","log-bin":"tsx src/scripts/log-paste-bin.ts","test:run":"vitest run","test:watch":"vitest --watch","type-check":"tsc --noEmit","build:watch":"tsc --watch","add-test-case":"node src/scripts/add-test-case.js","test:coverage":"vitest --coverage","validate-samples":"node src/scripts/validate-samples.js","download-fixtures":"node src/scripts/download-fixtures.js","regenerate-test-cases":"tsx src/scripts/regenerate-test-cases.ts"},"_npmUser":{"name":"cliftonc","email":"clifton.cunningham@gmail.com"},"_resolved":"/private/var/folders/_z/hcj36sbj1kxcdpdnmjzkh3zc0000gn/T/a9127a3ba944d06a3c28a17b474fda68/cliftonc-pptx-to-json-0.2.0.tgz","_integrity":"sha512-aDqBQWLs+MFKpH7pAN9aS65EjpKax6W9X3WEctAvPShJXT6XfXf7UO1sX8BxhiynSck6RgYS32EFFd1C4g9akg==","repository":{"url":"git+https://github.com/cliftonc/pptx-to-json.git","type":"git","directory":"packages/pptx-to-json"},"_npmVersion":"11.13.0","description":"Parse PowerPoint files (PPTX) and PowerPoint clipboard data into structured JSON — slides in presentation order, titles, speaker notes, geometry, connectors, tables and images.","directories":{},"_nodeVersion":"24.16.0","dependencies":{"jszip":"^3.10.1","fast-xml-parser":"^5.2.5"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.20.5","vitest":"^1.6.0","ts-node":"^10.9.2","typescript":"^5.9.2","@types/node":"^22.0.0","fracturedjsonjs":"^2.2.1","@vitest/coverage-v8":"^1.6.0"},"optionalDependencies":{"node-fetch":"^3.3.2"},"_npmOperationalInternal":{"tmp":"tmp/pptx-to-json_0.2.0_1785851775283_0.5407098584832275","host":"s3://npm-registry-packages-npm-production"},"deprecated":"Incomplete build: dist/ shipped partial and the entry point fails to import. Use 0.2.1 (same source)."},"0.2.1":{"name":"@cliftonc/pptx-to-json","version":"0.2.1","keywords":["pptx","powerpoint","ooxml","parser","json","slides"],"author":{"name":"Clifton Cunningham"},"license":"MIT","_id":"@cliftonc/pptx-to-json@0.2.1","maintainers":[{"name":"cliftonc","email":"clifton.cunningham@gmail.com"}],"homepage":"https://github.com/cliftonc/pptx-to-json#readme","bugs":{"url":"https://github.com/cliftonc/pptx-to-json/issues"},"dist":{"shasum":"bed70cda883c746db8db6ab16285764e6423577b","tarball":"https://registry.npmjs.org/@cliftonc/pptx-to-json/-/pptx-to-json-0.2.1.tgz","fileCount":75,"integrity":"sha512-NSpCBb6/IvRpl6jQYBFY0Pu4olPbheNGbyyuYBxfeN8+IVx6CbQFlWOg6PhOyyqykRqJwi2sbUz+rF3L0/xK4A==","signatures":[{"sig":"MEUCIQCR8WEmP1l0bUzcAE+y1KzoibJotL0Ivf8w6+NDLUhpAQIgXslgALLRCQsxIbgyix54tyK+Fa0WJL+nbQZiXYAFPl0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":497722},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=22.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.js"}},"gitHead":"69e9796dd4118676d5137c1700449440767347fb","scripts":{"test":"vitest","build":"rm -rf dist tsconfig.tsbuildinfo && tsc","log-bin":"tsx src/scripts/log-paste-bin.ts","test:run":"vitest run","test:watch":"vitest --watch","type-check":"tsc --noEmit","build:watch":"tsc --watch","verify:dist":"node test/verify-dist.mjs","add-test-case":"node src/scripts/add-test-case.js","test:coverage":"vitest --coverage","prepublishOnly":"npm run build && npm run verify:dist","validate-samples":"node src/scripts/validate-samples.js","download-fixtures":"node src/scripts/download-fixtures.js","regenerate-test-cases":"tsx src/scripts/regenerate-test-cases.ts"},"_npmUser":{"name":"cliftonc","email":"clifton.cunningham@gmail.com"},"repository":{"url":"git+https://github.com/cliftonc/pptx-to-json.git","type":"git","directory":"packages/pptx-to-json"},"_npmVersion":"11.13.0","description":"Parse PowerPoint files (PPTX) and PowerPoint clipboard data into structured JSON — slides in presentation order, titles, speaker notes, geometry, connectors, tables and images.","directories":{},"_nodeVersion":"24.16.0","dependencies":{"jszip":"^3.10.1","fast-xml-parser":"^5.2.5"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.20.5","vitest":"^1.6.0","ts-node":"^10.9.2","typescript":"^5.9.2","@types/node":"^22.0.0","fracturedjsonjs":"^2.2.1","@vitest/coverage-v8":"^1.6.0"},"optionalDependencies":{"node-fetch":"^3.3.2"},"_npmOperationalInternal":{"tmp":"tmp/pptx-to-json_0.2.1_1785852004859_0.21792658374022267","host":"s3://npm-registry-packages-npm-production"}}},"time":{"created":"2025-09-10T05:10:43.937Z","modified":"2026-08-04T14:00:39.400Z","0.1.0":"2025-09-10T05:10:44.238Z","0.1.1":"2025-09-10T05:19:53.447Z","0.2.0":"2026-08-04T13:56:15.442Z","0.2.1":"2026-08-04T14:00:04.985Z"},"bugs":{"url":"https://github.com/cliftonc/pptx-to-json/issues"},"author":{"name":"Clifton Cunningham"},"license":"MIT","homepage":"https://github.com/cliftonc/pptx-to-json#readme","keywords":["pptx","powerpoint","ooxml","parser","json","slides"],"repository":{"url":"git+https://github.com/cliftonc/pptx-to-json.git","type":"git","directory":"packages/pptx-to-json"},"description":"Parse PowerPoint files (PPTX) and PowerPoint clipboard data into structured JSON — slides in presentation order, titles, speaker notes, geometry, connectors, tables and images.","maintainers":[{"name":"cliftonc","email":"clifton.cunningham@gmail.com"}],"readme":"# @cliftonc/pptx-to-json\n\nA TypeScript library for parsing PowerPoint files (PPTX) and PowerPoint clipboard data into structured JSON. Extracts components like text, shapes, images, and tables with their positioning, styling, and content information.\n\n## 🌐 Live Demo\n\nTry it out at **[ppt-paste.clifton-cunningham.workers.dev](https://ppt-paste.clifton-cunningham.workers.dev/)**\n\n## Features\n\n- **PPTX File Parsing**: Parse complete PowerPoint files\n- **Clipboard Data Parsing**: Handle PowerPoint clipboard paste operations\n- **Component Extraction**: Extract text, shapes, images, tables, and more\n- **Position & Styling**: Get accurate positioning, dimensions, and styling information\n- **Slide structure**: presentation order, titles and speaker notes, resolved through the\n  relationship graph rather than guessed from filenames (see [Slide structure](#slide-structure))\n- **TypeScript Support**: Full TypeScript definitions included\n- **Modular Architecture**: Extensible parser system for different component types\n\n## Installation\n\n```bash\nnpm install @cliftonc/pptx-to-json\n```\n\n## Usage\n\n### Parsing PPTX Files\n\n```typescript\nimport { PPTXParser, PowerPointParser } from '@cliftonc/pptx-to-json';\nimport fs from 'node:fs';\n\nconst buffer = fs.readFileSync('presentation.pptx');\n\n// 1. unzip + parse every part to JSON\nconst json = await new PPTXParser().buffer2json(buffer);\n\n// 2. turn that into slides and components\nconst result = await new PowerPointParser().parseJson(json);\n\nfor (const slide of result.slides) {\n  console.log(slide.slideNumber, slide.title);   // presentation order, title placeholder\n  console.log(slide.notes);                      // speaker notes, if any\n  console.log(slide.components.length);\n}\n```\n\n### Parsing Clipboard Data\n\n```typescript\nimport { PowerPointClipboardProcessor } from '@cliftonc/pptx-to-json';\n\nconst processor = new PowerPointClipboardProcessor();\nconst result = await processor.parseClipboardBuffer(clipboardBuffer);\n\nconsole.log(result.slides);\n```\n\n## Slide structure\n\nThree answers are resolved through the package relationship graph, because the obvious\nshortcut is wrong in ways that are hard to notice afterwards:\n\n| | Read from | Why not the obvious way |\n|---|---|---|\n| `slide.slideNumber` | `ppt/presentation.xml` → `p:sldIdLst` → `r:id` | Slide **filenames are creation order**. PowerPoint keeps a slide's part name when a deck is reordered, so `slide3.xml` can be the seventh slide shown. `result.slideOrderSource` reports whether this list could be read; `slide.metadata.fileSlideNumber` keeps the filename number. |\n| `slide.notes` | `ppt/slides/_rels/slideN.xml.rels` → `notesSlide` | A notes part exists only for slides that HAVE notes, so `notesSlideM` stops lining up with `slideN` as soon as one slide lacks them. Notes attributed to the wrong slide are worse than no notes at all. Only the body placeholder is read: nearly every real notes part contains just the auto-generated slide-number field, which would otherwise be returned as \"notes\". |\n| `slide.title` | `p:ph type=\"title\"\\|\"ctrTitle\"`, else the layout's placeholder for that `idx` | A slide's placeholder often omits `@type` and is named only by the layout it inherits from. `slide.metadata.titleSource` says which of the two answered. |\n\n> ⚠️ **Do not set `removeNSPrefix: true` on the `XMLParser`.** It merges `id` and `r:id`\n> into one key, and which one wins depends on attribute order in the source — slide-order\n> resolution would appear to work on most decks and silently mis-resolve on others. The\n> parser is configured with `ignoreAttributes: false` and `attributeNamePrefix: \"$\"`, and\n> does **not** strip prefixes. Keep it that way.\n\n# PowerPoint Parse Output Format Specification\n\nThis document provides a comprehensive specification for the structured JSON produced by the PowerPoint parsing pipeline (`pptx` files and clipboard paste data) in this repository. It consolidates details from:\n- Normalized intermediate types (`src/types/normalized.ts`)\n- Public component types (`src/types/index.ts`)\n- Parser implementations (`TextParser`, `ShapeParser`, `ImageParser`, `TableParser`, `DiagramParser`, `VideoParser`, `ConnectorParser`)\n- Aggregation logic in `PowerPointParser`\n- Existing test fixtures in `test/test-harness/expected/*.json`\n\nThe goal is to give consumers a stable contract for what a \"parse result\" contains and how to interpret each field.\n\n---\n## 1. High-Level Result Shapes\n\nTwo conceptual layers exist:\n1. Normalized Intermediate (`NormalizedResult`) – internal, z-order–preserving structure used during parsing.\n2. Parsed Result (`ParsedResult`) – external structure of slides, masters, layouts, and components returned by `PowerPointParser.parseJson` (and downstream helpers).\n\nDepending on which public API you call (e.g. legacy wrappers like `parsePptx` / `parseClipboard` if reintroduced), you will typically receive a `ParsedResult`-like object or a simpler wrapper exposing the `slides` array.\n\n### 1.1 ParsedResult (Primary External Shape)\n```\ninterface ParsedResult {\n  slides: ParsedSlide[];            // In presentation order\n  masters: Record<string, ParsedMaster>; // Keyed by master XML path (ppt/slideMasters/slideMasterX.xml)\n  layouts: Record<string, ParsedLayout>; // Keyed by layout XML path (ppt/slideLayouts/slideLayoutY.xml)\n  totalComponents: number;          // Total count across all slides (after expansions)\n  format: 'pptx' | 'clipboard';     // Source format detected by normalizer\n  slideOrderSource?: 'presentation' | 'filename'; // Whether p:sldIdLst decided the order\n  slideDimensions?: { width: number; height: number }; // In pixels (EMU converted)\n}\n```\n\n### 1.2 ParsedSlide\n```\ninterface ParsedSlide {\n  slideIndex: number;        // Zero-based position in presentation order\n  slideNumber: number;       // One-based position in presentation order\n  title?: string;            // Title placeholder text, when the slide has one\n  notes?: string;            // Speaker notes, when the slide has any\n  layoutId?: string;         // Derived from layout file name (slideLayoutX)\n  background?: PowerPointComponent; // Background image/shape (if non-white & accepted)\n  components: PowerPointComponent[]; // All foreground components, including background IF pushed into flow earlier\n  metadata: SlideMetadata;   // Provenance + counts (see below)\n}\n\ninterface SlideMetadata {\n  name: string;                    // The slide title, or 'Slide N' when it has none\n  componentCount: number;          // components.length\n  format: 'pptx' | 'clipboard';\n  slideFile: string | null;        // e.g. 'ppt/slides/slide1.xml'\n  layoutFile: string | null;       // e.g. 'ppt/slideLayouts/slideLayout1.xml'\n  masterFile: string | null;       // e.g. 'ppt/slideMasters/slideMaster1.xml'\n  layoutElementCount: number;      // Count of inherited layout elements (pre-filter)\n  masterElementCount: number;      // Count of inherited master elements (pre-filter)\n  notesFile: string | null;        // e.g. 'ppt/notesSlides/notesSlide2.xml'\n  fileSlideNumber: number | null;  // The number in the filename — creation order\n  titleSource: 'placeholder' | 'layout' | null; // How the title was identified\n}\n```\n\n### 1.3 ParsedMaster\n```\ninterface ParsedMaster {\n  id: string;                     // e.g. 'slideMaster1'\n  name: string;                   // 'Master 1'\n  background?: PowerPointComponent; // Non-white fill or image background\n  components: PowerPointComponent[]; // Master-level decorative or thematic components\n  sourceFile: string;             // Full master path\n  placeholders?: PlaceholderMap;  // Placeholder geometry by idx or type:*\n  textStyles?: {                  // Extracted txStyles for font inheritance\n    titleStyle?: XMLNode;\n    bodyStyle?: XMLNode;\n  };\n}\n```\n\n### 1.4 ParsedLayout\n```\ninterface ParsedLayout {\n  id: string;                      // e.g. 'slideLayout3'\n  name: string;                    // 'Layout 3'\n  masterId?: string;               // Associated master id\n  background?: PowerPointComponent;\n  components: PowerPointComponent[];\n  sourceFile: string;              // layout file path\n  placeholders?: PlaceholderMap;   // Layout-level placeholders override masters\n}\n```\n\n---\n## 2. Component Model\n\nAll returned components are discriminated by `type`. Base fields:\n```\ninterface PowerPointComponentBase {\n  id: string;              // Unique within result; generated: <Kind or Name> <OriginalName or Index>\n  type: ComponentType;     // 'text' | 'shape' | 'image' | 'table' | 'video' | 'connection' | 'diagram' | 'any'\n  content?: string;        // Human-readable summary or plain extracted text\n  x: number;               // Pixel position (EMU converted)\n  y: number;\n  width: number;           // Pixels\n  height: number;          // Pixels\n  rotation?: number;       // Degrees; may be 0; negative supported\n  style?: ComponentStyle;  // See Style section\n  metadata?: Record<string, any>; // Parser + provenance details\n  slideIndex: number;      // Owning slide zero-based index (masters/layouts use 0 when synthesized)\n  zIndex: number;          // Higher means on top; backgrounds use large negative or normalized values\n}\n```\n\n### 2.1 TextComponent\n```\ninterface TextComponent extends PowerPointComponentBase {\n  type: 'text';\n  richText?: RichTextDoc;              // TipTap-like structure (paragraphs & bulletLists)\n  backgroundShape?: {                  // Only present for text-with-visible background geometry\n    type: 'rectangle' | 'ellipse' | 'roundRect' | 'custom';\n    fill?: FillInfo;                   // { type, color, opacity }\n    border?: BorderInfo;               // { type, color, width, style, cap?, compound? }\n    geometry?: GeometryInfo;           // Normalized geometry description\n  };\n}\n```\nKey text-specific metadata fields:\n- `metadata.isTextBox`: boolean (derived from `cNvSpPr.$txBox`)\n- `metadata.paragraphCount`: number\n- `metadata.hasMultipleRuns`: boolean (true if any paragraph has multiple runs)\n- `metadata.namespace`: `'a' | 'p'` indicates clipboard vs pptx source segment\n\nSpacing heuristics: spaces may be synthesized between adjacent alphanumeric runs when PowerPoint omits them across run boundaries.\n\n### 2.2 ShapeComponent\n```\ninterface ShapeComponent extends PowerPointComponentBase {\n  type: 'shape';\n  shapeType?: string;        // Human-friendly descriptor (e.g. 'rectangle', 'up arrow')\n  geometry?: GeometryInfo;   // { type, preset, isCustom, paths? }\n}\n```\nStyle fields usually populated:\n- `style.fillColor`, `style.fillOpacity`\n- `style.borderColor`, `style.borderWidth`, `style.borderStyle`\n- Effects (if any) merged into style: possible keys include `effects`, `boxShadow`, etc.\n\nMetadata booleans:\n- `hasEffects`, `hasFill`, `hasBorder`\n\n### 2.3 ImageComponent\n```\ninterface ImageComponent extends PowerPointComponentBase {\n  type: 'image';\n  src?: string;      // Data URL or externally hosted URL (if R2 / remote)\n  alt?: string;      // Description from cNvPr.$descr\n}\n```\nMetadata additions:\n- `relationshipId`: underlying rId\n- `imageUrl`: dup of `src` for clarity\n- `imageType`: normalized extension (png, jpeg, gif, etc.)\n- `imageSize`: byte length (if known)\n- `hasEffects`: boolean\n\nNotes:\n- WMF/EMF images are skipped (treated as decorative backgrounds).\n- Background images use negative z-index only during intermediate; accepted backgrounds appear in slide `background` or as first component in flow depending on processing stage.\n\n### 2.4 TableComponent\n```\ninterface TableComponent extends PowerPointComponentBase {\n  type: 'table';\n  rows?: { cells: { content: string; style?: ComponentStyle; colSpan?: number; rowSpan?: number }[] }[];\n  columns?: number; // Count of columns\n}\n```\nMetadata:\n- `tableData`: 2D string matrix (raw textual extraction)\n- `rows`, `cols`: numeric dimensions\n- `hasHeader`: currently `true` if first row is treated as header in `richText` structure\n- `richText`: TipTap table representation mirroring client expectation\n- `format`: namespace `'a' | 'p'`\n\n### 2.5 DiagramComponent\n```\ninterface DiagramComponent extends PowerPointComponentBase {\n  type: 'diagram';\n  diagramType?: string;        // e.g. 'smartart' or 'unknown'\n  title?: string;              // Derived fallback name\n  smartArtData?: {             // Only for SmartArt when fully extracted\n    dataPoints: SmartArtDataPoint[];    // Logical model nodes\n    connections: SmartArtConnection[];  // Logical edges\n    shapes: SmartArtShape[];            // Visual shape placements\n    layout: SmartArtLayout;             // Layout metadata\n  };\n  extractedComponents?: PowerPointComponent[]; // Flattened shapes/text derived from SmartArt\n}\n```\nIf `options.returnExtractedComponents` is set during parse, the original `DiagramComponent` wrapper may be replaced by its `extractedComponents` array (each item a standard `PowerPointComponent` type – often `shape` or `text`).\n\n### 2.6 VideoComponent\n```\ninterface VideoComponent extends PowerPointComponentBase {\n  type: 'video';\n  url?: string;                 // Resolved target from slide relationships\n  thumbnailSrc?: string;        // Data URL for poster image (if available)\n  title?: string;               // From cNvPr.$name\n  embedType?: 'youtube' | 'vimeo' | 'generic';\n}\n```\nResolution logic distinguishes between PPTX mode (relationship XML) and clipboard mode (simpler relationship map). If no match is found, `url` may be `undefined`.\n\n### 2.7 ConnectionComponent (Connectors / Lines)\n```\ninterface ConnectionComponent extends PowerPointComponentBase {\n  type: 'connection';\n  startShapeId?: string;         // Target shape ID (parsed from connection metadata)\n  endShapeId?: string;\n  connectorType?: string;        // Derived line archetype\n  startPoint?: { x: number; y: number }; // Computed anchor point in pixels\n  endPoint?: { x: number; y: number };\n  lineStyle?: {                  // Extracted from <ln>\n    width?: number;              // Pixels\n    color?: string;              // Hex color\n    dashStyle?: string;          // e.g. 'dash'\n    startArrow?: string;         // Arrowhead descriptor\n    endArrow?: string;\n  };\n}\n```\nPost-processing (`fixConnectionPoints`) refines `startPoint` / `endPoint` after all shapes are known, using indices stored in `metadata.startConnectionIndex` / `metadata.endConnectionIndex` and bounding boxes of associated shapes.\n\n### 2.8 UnknownComponent\n```\ninterface UnknownComponent extends PowerPointComponentBase {\n  type: 'any';\n}\n```\nRare fallback when a parser cannot classify an element but still returns structural coordinates.\n\n---\n## 3. Styles & Visual Data\n\n```\ninterface ComponentStyle {\n  fontFamily?: string;\n  fontSize?: number;         // Points for text; numeric px-like scale after extraction\n  fontWeight?: string;       // 'bold', 'normal', etc.\n  fontStyle?: string;        // 'italic', 'normal'\n  textAlign?: string;        // 'left' | 'center' | 'right' | 'justify'\n  color?: string;            // Foreground text color (hex)\n  borderWidth?: number;      // Pixels\n  borderStyle?: string;      // 'solid' | 'none' | dash patterns\n  borderColor?: string;\n  fillColor?: string;        // Background fill (shape/table)\n  fillOpacity?: number;      // 0–1 normalized\n  rotation?: number;         // Degrees\n  // Additional dynamic keys include effects, shadows, filters\n  [key: string]: any;\n}\n```\n\nFill, border, geometry helpers (from shape parsing):\n```\ninterface FillInfo { type: 'solid' | 'gradient' | 'pattern' | 'none'; color: string; opacity: number; }\ninterface BorderInfo { type: 'solid' | 'none'; color: string; width: number; style: string; cap?: string; compound?: string; }\ninterface GeometryInfo { type: string; preset: string | null; isCustom: boolean; paths?: any[]; }\n```\n\nEffects (`EffectsInfo` – merged into style) may include:\n- `effects: string[]`\n- `boxShadow?: string`\n- Additional effect-specific keys\n\nImage effects augmented fields:\n- `opacity`, `filter`, `borderRadius`, `shadow`, `effectsList`\n\n---\n## 4. Coordinate & Unit Handling\n\n- PowerPoint internally uses EMUs (English Metric Units).\n- All exported `x`, `y`, `width`, `height` are converted to **pixels** using centralized constants (`emuToPixels` / `validatePixelRange`).\n- Rotation is converted from 1/60000 degree units to plain degrees.\n- Slide dimensions (`slideDimensions`) are also pixel-converted.\n- Guard rails: any dimension > 50,000 triggers a warning (likely unconverted EMU scenario).\n\n---\n## 5. Z-Ordering & Background Logic\n\n- Backgrounds extracted from masters/layouts/slides are given strongly negative `zIndex` during normalization:\n  - Master backgrounds: start at -2000 and decrement.\n  - Layout backgrounds: start at -1000 and decrement.\n  - Slide-specific backgrounds: promoted with `zIndex` of -500 when parsed and may be stored in both `slide.background` and `components` (depending on stage) if they pass the non-white criteria.\n- White / near-white solid fills are suppressed (ignored) for master/layout backgrounds to avoid noise.\n- Image backgrounds are always preserved.\n\n---\n## 6. Placeholder & Master Style Inheritance\n\nText and shape components with zero transform size/position will look up placeholder geometry if:\n1. They reference a placeholder (`nvSpPr > nvPr > ph`), and\n2. A matching placeholder (`idx` or `type`) exists in merged placeholder maps (master first, then layout override).\n\nMaster text styles (`titleStyle`, `bodyStyle`) affect font size inheritance when a text element lacks explicit `rPr` size declarations.\n\nPriority order for font size resolution:\n1. First explicit non-empty run with size\n2. Placeholder-derived master `titleStyle` (if placeholder type is 'title')\n3. Master `bodyStyle`\n4. Default fallback (Arial 18, or other parser default)\n\n---\n## 7. Rich Text Representation\n\nAll rich text surfaces (text boxes and table cells) are expressed in a simplified TipTap-like JSON:\n```\n{ type: 'doc', content: Array<Paragraph | BulletList | TableNode> }\n```\nSupported nodes:\n- `paragraph` – sequential inline `text` nodes\n- `bulletList` → `listItem` → nested `paragraph`\n- `table` (for tables only) → `tableRow` → `tableCell` / `tableHeader` → `paragraph`\n\nText marks (`marks`) include:\n- `{ type: 'bold' }`\n- `{ type: 'italic' }`\n- `{ type: 'textStyle', attrs: { fontSize: '22pt', color: '#FFFFFF', ... } }`\n\nSpacing heuristics maintain readability while avoiding double insertion of whitespace.\n\n---\n## 8. Diagram (SmartArt) Expansion\n\nWhen SmartArt is detected:\n- `diagramType` set to `smartart`\n- Raw relationship IDs are scanned to collect companion diagram/data/drawing files.\n- `smartArtData` includes:\n  - `dataPoints`: logical node set with hierarchy & content\n  - `connections`: edges linking data points\n  - `shapes`: positioned shapes representing each data node\n  - `layout`: summarizing layoutType, category, colorScheme, quickStyle\n- If `returnExtractedComponents` flag is provided, the parser returns the flattened `extractedComponents` instead of the diagram wrapper, allowing uniform downstream handling.\n\n---\n## 9. Connections (Lines/Arrows)\n\nConnectors rely on:\n- `startConnection` / `endConnection` from normalized XML (IDs + connection site indices)\n- A second pass to compute pixel anchor points (center, mid-left, mid-right, etc.) using bounding boxes of referenced shapes\n- Style extraction from `<ln>` (width, color, dash style, arrowheads)\n\nMetadata fields assist post-processing:\n- `startConnectionIndex`, `endConnectionIndex`\n- Boolean flags `hasStartConnection`, `hasEndConnection`\n\n---\n## 10. Media & Relationship Resolution\n\nImages & videos use relationship graphs:\n- PPTX mode: relationships are file-based (`ppt/slides/_rels/slideN.xml.rels`), and the\n  slide's own rels part is passed down explicitly. Deriving it from the slide's position\n  would read another slide's relationships once a deck has been reordered.\n- Clipboard mode: relationships stored in a simplified map keyed by slide index.\n- Background images restructure `blipFill` to surface `embed` id for uniform handling.\n- Video thumbnails resolve through nested `blipFill` -> `embed` in `thumbnailRelationshipId` if present.\n\nSkipped media:\n- `wmf` / `emf` images (often vector placeholders or noise).\n\n---\n## 11. Background Acceptance Rules\n\nA candidate background is added only if:\n- It is an image, OR\n- It has a non-white/near-white fill (solid, gradient, pattern).\n\nWhite-ish backgrounds are suppressed to reduce redundancy, especially on layout/master inheritance, letting slide content show default canvas background.\n\n---\n## 12. IDs & Names\n\n`id` strategy uses `BaseParser.generateComponentId(kind, componentIndex, originalName)` ensuring uniqueness. Connectors append semi-colon–delimited segments for later shape ID extraction. Connection component `id` format example:\n```\n<GeneratedConnectorName>;<UnderlyingShapeId>;<namespace><slideIndex>\n```\nShape IDs referenced by connectors are extracted from the second `;` segment (`shapeId`).\n\n---\n## 13. Error Handling & Null Returns\n\nEach specialized parser returns `null` to skip components when:\n- Missing structural nodes (`spPr`, `blipFill`, `graphicData`, etc.)\n- Zero width/height (for shapes, diagrams)\n- Empty text content (text parser)\n- Filtered image types (wmf/emf)\n\nThe aggregator silently ignores `null` (optionally logging in debug mode).\n\n---\n## 14. Examples\n\n### 14.1 Text Component (Clipboard)\n```\n{\n  \"id\": \"TextBox 4\",\n  \"type\": \"text\",\n  \"content\": \"TEXT\",\n  \"x\": 109,\n  \"y\": 41,\n  \"width\": 288,\n  \"height\": 139,\n  \"style\": { \"fontSize\": 80, \"color\": \"#4EA72E\", ... },\n  \"richText\": { \"type\": \"doc\", \"content\": [ { \"type\": \"paragraph\", ... } ] },\n  \"backgroundShape\": { \"type\": \"rectangle\", \"border\": { ... }, \"geometry\": { ... } },\n  \"metadata\": { \"namespace\": \"a\", \"paragraphCount\": 1 }\n}\n```\n\n### 14.2 Shape Component\n```\n{\n  \"id\": \"Arrow: Up 12\",\n  \"type\": \"shape\",\n  \"content\": \"up arrow shape\",\n  \"x\": 341,\n  \"y\": 471,\n  \"width\": 124,\n  \"height\": 176,\n  \"style\": { \"fillColor\": \"#4472C4\", \"borderStyle\": \"none\", ... },\n  \"shapeType\": \"up arrow\",\n  \"geometry\": { \"type\": \"up arrow\", \"preset\": \"upArrow\", \"isCustom\": false },\n  \"metadata\": { \"hasFill\": true, \"hasBorder\": false }\n}\n```\n\n### 14.3 Image Component (Background)\n```\n{\n  \"id\": \"Google Shape;308;p58\",\n  \"type\": \"image\",\n  \"content\": \"Google Shape;308;p58\",\n  \"x\": 25,\n  \"y\": 53,\n  \"width\": 569,\n  \"height\": 473,\n  \"src\": \"data:image/png;base64,...\",\n  \"metadata\": { \"relationshipId\": \"rId1\", \"imageType\": \"png\", \"imageSize\": 220591 }\n}\n```\n\n### 14.4 Connection Component\n```\n{\n  \"id\": \"Connector 5;42;p0\",\n  \"type\": \"connection\",\n  \"content\": \"Connector 5 Connection\",\n  \"x\": 100, \"y\": 120, \"width\": 200, \"height\": 0,\n  \"startShapeId\": \"1001\",\n  \"endShapeId\": \"1002\",\n  \"startPoint\": { \"x\": 300, \"y\": 140 },\n  \"endPoint\": { \"x\": 100, \"y\": 140 },\n  \"lineStyle\": { \"width\": 2, \"color\": \"#000000\", \"dashStyle\": \"solid\" },\n  \"metadata\": { \"startConnectionIndex\": 3, \"endConnectionIndex\": 1 }\n}\n```\n\n(Example generalized; actual IDs may differ.)\n\n---\n## 15. Versioning & Stability Notes\n\n- The discriminated union of `PowerPointComponent` is the primary stable contract.\n- New component types (e.g. future media or interactive elements) will extend the union with a distinct `type` value and additive fields.\n- Optional metadata fields are additive; consumers should feature-detect.\n- Normalized internal structures are subject to refinement but will continue to feed the same external component shapes.\n\n---\n## 16. Migration & Compatibility Guidelines\n\nWhen integrating:\n1. **Rely on `type`** for branching logic.\n2. **Use presence checks** (`if ('richText' in component)`) before accessing optional fields.\n3. **Do not assume** backgrounds appear only in `slide.background` – they may also appear as the first `components[]` entry for legacy consumers.\n4. **Expect arrays** like `richText.content` and `rows` to be empty rather than `null` when no data exists.\n5. **Ignore unknown keys** in `style` – future effect enrichments will appear there.\n\n---\n## 17. Known Limitations / TODO\n\n- SmartArt extraction: Coverage may vary; complex layouts might produce partial `smartArtData`.\n- Table styling: Cell-level font/color details are currently minimal (structure prioritized over rich formatting).\n- Video thumbnails: Not always available depending on relationship resolution.\n- Connectors: Complex routing (elbows, curves) simplified to bounding box endpoints; no path geometry yet.\n- Diagram expansion may return large component sets; consider filtering `extractedComponents` downstream.\n\n\n## Development\n\nThis library is part of a larger PowerPoint parsing ecosystem. The complete system includes:\n\n- **[@cliftonc/pptx-to-json](https://www.npmjs.com/package/@cliftonc/pptx-to-json)**: This parsing library\n- **Web Application**: Live demo and visual interface\n- **Cloudflare Worker**: API endpoints for parsing operations\n\n### Local Development\n\n```bash\n# Build the library\nnpm run build\n\n# Run tests\nnpm test\n\n# Type checking\nnpm run type-check\n\n# Test with sample files\nnpm run log-bin sample.pptx\n```\n\n## Technical Details\n\n### Architecture\n\nThe library uses a modular parser architecture:\n\n- **BaseParser**: Common utilities for EMU conversion, colors, transforms\n- **PowerPointParser**: Main coordinator handling format detection\n- **TextParser**: Text components with font/style parsing\n- **ShapeParser**: Geometric shapes with fill, border, effects\n- **ImageParser**: Images with data URL extraction\n- **TableParser**: Tables with cell structure and formatting\n\n### Format Support\n\n- **PPTX Files**: Complete Office Open XML PowerPoint files\n- **Clipboard Data**: PowerPoint clipboard format with proper namespace handling\n- **Component Detection**: Reliable classification with fallback logic\n\n## License\n\nMIT\n\n## Contributing\n\nIssues and pull requests are welcome! Please visit the [GitHub repository](https://github.com/cliftonc/pptx-to-json) for more information.\n\n## Related Projects\n\n- 🌐 **[Live Demo](https://ppt-paste.clifton-cunningham.workers.dev/)**: Try the parser in your browser\n- 📚 **Documentation**: Full implementation details in the main repository\n\n---\n\nMade with ❤️ by [Clifton Cunningham](https://github.com/cliftonc)\n","readmeFilename":"README.md"}