{"_id":"@bunkojp/text-segmentation","_rev":"2-a78e327efe9654a5166c5a4127bb4535","name":"@bunkojp/text-segmentation","dist-tags":{"latest":"0.2.0"},"versions":{"0.1.0":{"name":"@bunkojp/text-segmentation","version":"0.1.0","keywords":["text-segmentation","chunking","ncd","tfidf","nlp","japanese"],"license":"CC0-1.0","_id":"@bunkojp/text-segmentation@0.1.0","maintainers":[{"name":"trkbt10","email":"triple.quartet+npm@gmail.com"}],"homepage":"https://github.com/bunko-jp/text-segmentation#readme","bugs":{"url":"https://github.com/bunko-jp/text-segmentation/issues"},"dist":{"shasum":"1c15eb3eb6db9b76c422256f91532656e6185e23","tarball":"https://registry.npmjs.org/@bunkojp/text-segmentation/-/text-segmentation-0.1.0.tgz","fileCount":72,"integrity":"sha512-6y9X73SoY97uKGnpu8kV88w6iNq2NsAK0Wd0iqi49aRfeEKd0IePw4PaGRUUtctK3ym5vscVI5HKH7VFXc7/1g==","signatures":[{"sig":"MEYCIQCLU9P27/EVYMnnzR6pHjpJaz0G+htqdEeXCbj1ybYWjAIhAP98Y1BbEOCahWiWQNrj0CVUFl7Ato7QCd3jdUUI54/X","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":375612},"main":"dist/index.cjs","type":"module","types":"dist/index.d.ts","module":"dist/index.js","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"gitHead":"9386df21dc35bba8d7ac574d8fd15447691e0bb9","scripts":{"lint":"eslint .","test":"vitest --run","build":"vite build","clean":"rimraf dist","format":"prettier --write .","lint:fix":"eslint . --fix","test:cov":"vitest run --coverage","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"trkbt10","email":"triple.quartet+npm@gmail.com"},"repository":{"url":"git+https://github.com/bunko-jp/text-segmentation.git","type":"git"},"_npmVersion":"11.11.0","description":"Split text into semantic or structural chunks using purely algorithmic strategies. Supports mixed Japanese/English text.","directories":{},"sideEffects":false,"_nodeVersion":"22.18.0","dependencies":{"fflate":"^0.8.2"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"bun@1.3.11","devDependencies":{"ajv":"^8.18.0","vite":"^8.0.0","eslint":"^10.0.3","rimraf":"^6.1.3","vitest":"^4.1.0","prettier":"^3.8.1","@eslint/js":"^10.0.1","typescript":"^5.9.3","@types/node":"^25.5.0","@types/react":"^19.2.14","vite-plugin-dts":"^4","typescript-eslint":"^8.57.1","@vitest/coverage-v8":"^4.1.0","eslint-plugin-jsdoc":"^62.8.0","eslint-plugin-import":"^2.32.0","eslint-config-prettier":"^10.1.8","@typescript-eslint/parser":"^8.57.1","@typescript-eslint/eslint-plugin":"^8.57.1","@eslint-community/eslint-plugin-eslint-comments":"^4.7.1"},"_npmOperationalInternal":{"tmp":"tmp/text-segmentation_0.1.0_1773903358615_0.7637648157551593","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@bunkojp/text-segmentation","version":"0.2.0","description":"Split text into semantic or structural chunks using purely algorithmic strategies. Supports mixed Japanese/English text.","license":"CC0-1.0","type":"module","repository":{"type":"git","url":"git+https://github.com/bunko-jp/text-segmentation.git"},"keywords":["text-segmentation","chunking","ncd","tfidf","nlp","japanese"],"main":"dist/index.cjs","module":"dist/index.js","types":"dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js","require":"./dist/index.cjs"}},"publishConfig":{"access":"public"},"packageManager":"bun@1.3.11","sideEffects":false,"scripts":{"build":"vite build","clean":"rimraf dist","lint":"eslint .","lint:fix":"eslint . --fix","format":"prettier --write .","typecheck":"tsc -p tsconfig.json --noEmit","test:cov":"vitest run --coverage","test":"vitest --run"},"devDependencies":{"@eslint-community/eslint-plugin-eslint-comments":"^4.7.1","@eslint/js":"^10.0.1","@types/node":"^25.5.0","@types/react":"^19.2.14","@typescript-eslint/eslint-plugin":"^8.57.1","@typescript-eslint/parser":"^8.57.1","@vitest/coverage-v8":"^4.1.0","ajv":"^8.18.0","eslint":"^10.0.3","eslint-config-prettier":"^10.1.8","eslint-plugin-import":"^2.32.0","eslint-plugin-jsdoc":"^62.8.0","prettier":"^3.8.1","rimraf":"^6.1.3","typescript":"^5.9.3","typescript-eslint":"^8.57.1","vite":"^8.0.0","vite-plugin-dts":"^4","vitest":"^4.1.0"},"dependencies":{"fflate":"^0.8.2"},"gitHead":"664b58e371c05edda75ecbf4945474b3d40ff534","_id":"@bunkojp/text-segmentation@0.2.0","bugs":{"url":"https://github.com/bunko-jp/text-segmentation/issues"},"homepage":"https://github.com/bunko-jp/text-segmentation#readme","_nodeVersion":"22.18.0","_npmVersion":"11.11.0","dist":{"integrity":"sha512-LHJC9jWaesKOFFJDqH+AmSqCUJvH4A7FEeO6xX8ehun7haeHsypBuWVYmHz+r+h1lD5rHgiwdhwidhHJp0dx8g==","shasum":"688bb792c2ac3542331c9d45435aa7e4d9ad12d6","tarball":"https://registry.npmjs.org/@bunkojp/text-segmentation/-/text-segmentation-0.2.0.tgz","fileCount":62,"unpackedSize":418496,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDwdCnVdzpoIHjfIqE74tF5B4TWeSNf8WQJ1ijkGxmSTwIgJCBRxg6H7RXRF1uJ+w0cmj0zm31ySBEzq9yKI6Jn55o="}]},"_npmUser":{"name":"trkbt10","email":"triple.quartet+npm@gmail.com"},"directories":{},"maintainers":[{"name":"trkbt10","email":"triple.quartet+npm@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/text-segmentation_0.2.0_1774067278685_0.9324684331164641"},"_hasShrinkwrap":false}},"time":{"created":"2026-03-19T06:55:58.472Z","modified":"2026-03-21T04:27:59.007Z","0.1.0":"2026-03-19T06:55:58.771Z","0.2.0":"2026-03-21T04:27:58.881Z"},"bugs":{"url":"https://github.com/bunko-jp/text-segmentation/issues"},"license":"CC0-1.0","homepage":"https://github.com/bunko-jp/text-segmentation#readme","keywords":["text-segmentation","chunking","ncd","tfidf","nlp","japanese"],"repository":{"type":"git","url":"git+https://github.com/bunko-jp/text-segmentation.git"},"description":"Split text into semantic or structural chunks using purely algorithmic strategies. Supports mixed Japanese/English text.","maintainers":[{"name":"trkbt10","email":"triple.quartet+npm@gmail.com"}],"readme":"# text-segmentation\n\nSplit text into semantic or structural chunks using purely algorithmic strategies. No LLM or external API dependencies. Supports mixed Japanese/English text.\n\n## Install\n\n```bash\nnpm install @bunkojp/text-segmentation\n```\n\n## Quick Start\n\n```ts\nimport { segmentByNcdTfidf } from \"@bunkojp/text-segmentation\";\n\nconst text = `Today the weather is nice. I went for a walk. I saw flowers in the park.\n\nYesterday I went to the supermarket. I bought vegetables and meat. I cooked dinner.`;\n\nconst segments = segmentByNcdTfidf(text, {\n  targetChunkSize: 80,\n  minChunkSize: 20,\n  maxChunkSize: 300,\n  windowSize: 2,\n});\n\nfor (const seg of segments) {\n  console.log(`[${seg.start}:${seg.end}]`, text.slice(seg.start, seg.end));\n}\n```\n\n## Strategies\n\nFour segmentation strategies are provided, listed from fastest/simplest to most semantically accurate.\n\n### Punctuation\n\nAccumulates sentences up to a target size and splits at sentence boundaries. The fastest strategy.\n\n```ts\nimport { segmentByPunctuation } from \"@bunkojp/text-segmentation\";\n\nconst segments = segmentByPunctuation(text, {\n  targetChunkSize: 500,  // Target chunk size in characters\n  minChunkSize: 100,     // Minimum chunk size\n  maxChunkSize: 2000,    // Hard limit on chunk size\n});\n```\n\n### Compression (NCD)\n\nComputes Normalized Compression Distance between adjacent sentence windows to detect semantic boundaries.\n\n```ts\nimport { segmentByCompression } from \"@bunkojp/text-segmentation\";\n\nconst segments = segmentByCompression(text, {\n  targetChunkSize: 500,\n  minChunkSize: 100,\n  maxChunkSize: 2000,\n  ncdThreshold: 0.4,    // Boundary detection threshold (higher = fewer splits)\n  windowSize: 3,        // Number of sentences per window\n  adaptive: false,      // Set true for percentile-based automatic thresholding\n  ncdPercentile: 0.2,   // Percentile used in adaptive mode\n});\n```\n\n### TF-IDF\n\nUses TF-IDF cosine distance between adjacent sentence windows. Strong at detecting lexical topic shifts.\n\n```ts\nimport { segmentByTfidf } from \"@bunkojp/text-segmentation\";\n\nconst segments = segmentByTfidf(text, {\n  targetChunkSize: 500,\n  minChunkSize: 100,\n  maxChunkSize: 2000,\n  tfidfThreshold: 0.45,\n  windowSize: 3,\n  adaptive: false,\n  tfidfPercentile: 0.2,\n});\n```\n\n### NCD + TF-IDF\n\nWeighted combination of compression distance and TF-IDF cosine distance. The most robust strategy.\n\n```ts\nimport { segmentByNcdTfidf } from \"@bunkojp/text-segmentation\";\n\nconst segments = segmentByNcdTfidf(text, {\n  targetChunkSize: 500,\n  minChunkSize: 100,\n  maxChunkSize: 2000,\n  ncdTfidfThreshold: 0.42,\n  windowSize: 3,\n  ncdWeight: 0.5,       // Weight for NCD component\n  tfidfWeight: 0.5,     // Weight for TF-IDF component\n  adaptive: false,\n  ncdTfidfPercentile: 0.2,\n});\n```\n\n## Streaming\n\nAll strategies provide an `AsyncGenerator`-based streaming API.\n\n```ts\nimport { streamSegmentByCompression } from \"@bunkojp/text-segmentation\";\n\nfor await (const event of streamSegmentByCompression(text)) {\n  if (event.type === \"segment\") {\n    console.log(`Segment #${event.index}:`, event.point);\n  }\n  if (event.type === \"done\") {\n    console.log(\"Total segments:\", event.points.length);\n  }\n}\n```\n\n## Types\n\n```ts\ntype SegmentPoint = {\n  start: number;   // Start position (inclusive)\n  end: number;     // End position (exclusive)\n  type: \"heading\" | \"section\" | \"paragraph\";\n};\n```\n\nUse `text.slice(segment.start, segment.end)` to extract segment text. All segments are contiguous with no gaps and cover the entire input text.\n\n## Utilities\n\nThe sentence splitter is also available as a standalone utility.\n\n```ts\nimport { splitIntoSentences } from \"@bunkojp/text-segmentation\";\n\nconst sentences = splitIntoSentences(\"First sentence. Second sentence.\");\n// [{ index: 1, text: \"First sentence.\", start: 0, end: 16 }, ...]\n```\n\n## Example Results\n\nPre-generated segmentation results for Akutagawa Ryunosuke's \"Rashomon\" (5,839 characters) are included in [`spec/fixtures/results/`](spec/fixtures/results/). Each JSON file contains the strategy name, configuration, and full segment list with positions and full text content.\n\nPunctuation uses a fixed `targetChunkSize=500`. Semantic strategies use **adaptive mode** with a wide size range (`min=100, max=3000`) so that semantic boundaries dominate over size constraints.\n\n| Strategy | Segments | Avg Length | Config |\n|----------|----------|------------|--------|\n| [Punctuation](spec/fixtures/results/rashomon-punctuation.json) | 11 | 531 chars | target=500, min=100, max=2000 |\n| [Compression](spec/fixtures/results/rashomon-compression.json) | 24 | 243 chars | adaptive, percentile=0.25, window=3 |\n| [TF-IDF](spec/fixtures/results/rashomon-tfidf.json) | 28 | 209 chars | adaptive, percentile=0.25, window=3 |\n| [NCD+TF-IDF](spec/fixtures/results/rashomon-ncd-tfidf.json) | 27 | 216 chars | adaptive, percentile=0.25, window=3 |\n\nTo regenerate these results:\n\n```bash\nbun spec/fixtures/generate-results.ts\n```\n\n## How It Works\n\nThe semantic strategies (Compression, TF-IDF, NCD+TF-IDF) share a common window-based algorithm:\n\n1. Split text into sentences\n2. Create sliding windows of N sentences\n3. Compute divergence between adjacent windows\n4. Select local maxima as boundary candidates\n5. Apply size constraints (min/max/target)\n\n**Adaptive mode** replaces fixed thresholds with percentile-based automatic threshold selection and recursively subdivides oversized chunks.\n\n## License\n\n[CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/)\n","readmeFilename":"README.md"}