{"_id":"@bitofsky/merge-streams","_rev":"4-6b3912fb51ac3688d671739a958d2584","name":"@bitofsky/merge-streams","dist-tags":{"latest":"1.1.0"},"versions":{"1.0.0":{"name":"@bitofsky/merge-streams","version":"1.0.0","keywords":["stream","merge","csv","json","arrow","apache-arrow","concat","streaming","fetch"],"author":{"name":"BumSeok Hwang","email":"bitofsky@naver.com"},"license":"MIT","_id":"@bitofsky/merge-streams@1.0.0","maintainers":[{"name":"bitofsky","email":"bitofsky@naver.com"}],"homepage":"https://github.com/bitofsky/merge-streams#readme","bugs":{"url":"https://github.com/bitofsky/merge-streams/issues"},"dist":{"shasum":"f19a91835a449c1d3b34826a95ec720bf6c91c12","tarball":"https://registry.npmjs.org/@bitofsky/merge-streams/-/merge-streams-1.0.0.tgz","fileCount":27,"integrity":"sha512-XKNX2JKzMuX0JWSnHzTmRW7jRGF4CGJfFjiF33zrBEnNtavGf2QwncRR8W+MfRiVAmDJ0d5RYyeDnD4WtwQddw==","signatures":[{"sig":"MEQCIHzic3Yk4KaQs3wGclkbSLhHU4KRhre42xRtmLneyTc7AiBVkAb+tjtP4KEXZ/LvcQsNiZxMSp+nhsnim1XX7d/IOg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":43333},"main":"dist/index.js","type":"module","types":"dist/index.d.ts","engines":{"node":">=18.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsc","test:watch":"vitest","prepublishOnly":"npm run build"},"_npmUser":{"name":"bitofsky","email":"bitofsky@naver.com"},"repository":{"url":"git+ssh://git@github.com/bitofsky/merge-streams.git","type":"git"},"_npmVersion":"11.3.0","description":"Merge multiple chunked streams (CSV, JSON_ARRAY, ARROW_STREAM) into one unified stream - perfect for Databricks External Links","directories":{},"_nodeVersion":"24.0.2","dependencies":{"apache-arrow":"^18.0.0"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^1.6.0","typescript":"^5.0.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/merge-streams_1.0.0_1766587294837_0.5788603496021518","host":"s3://npm-registry-packages-npm-production"}},"1.0.3":{"name":"@bitofsky/merge-streams","version":"1.0.3","keywords":["stream","merge","csv","json","arrow","apache-arrow","concat","streaming","fetch"],"author":{"name":"BumSeok Hwang","email":"bitofsky@naver.com"},"license":"MIT","_id":"@bitofsky/merge-streams@1.0.3","maintainers":[{"name":"bitofsky","email":"bitofsky@naver.com"}],"homepage":"https://github.com/bitofsky/merge-streams#readme","bugs":{"url":"https://github.com/bitofsky/merge-streams/issues"},"dist":{"shasum":"c05a1c9b797dc8d16397de2574c18c44daa8bcd9","tarball":"https://registry.npmjs.org/@bitofsky/merge-streams/-/merge-streams-1.0.3.tgz","fileCount":27,"integrity":"sha512-DleQCeJ12iek5vSqE44rUxxWhTbl5NOrZufTlXRLn+pLcqLNJ0F8zAOuS4gD5V/tYEwDN0UW74QG4tG3EGj4pA==","signatures":[{"sig":"MEYCIQCm5ANw+K2Mhy2Lg4e4H/C9O0OyucNx3xNmieeWxxFxhQIhAOFyiGDHJz/7ozOM4YotR3gWmOZFZGkV8xO51lwdh4O/","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bitofsky%2fmerge-streams@1.0.3","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":43685},"main":"dist/index.js","type":"module","types":"dist/index.d.ts","engines":{"node":">=18.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"4bbeeb79a0e225f58219cd24550ee95049c03c7b","scripts":{"test":"vitest run","build":"tsc","test:watch":"vitest","prepublishOnly":"npm run build"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:55a58a54-fc84-407f-8354-36ce0cefd340"}},"repository":{"url":"git+ssh://git@github.com/bitofsky/merge-streams.git","type":"git"},"_npmVersion":"11.6.2","description":"Merge multiple chunked streams (CSV, JSON_ARRAY, ARROW_STREAM) into one unified stream - perfect for Databricks External Links","directories":{},"_nodeVersion":"24.12.0","dependencies":{"apache-arrow":"^18.0.0"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^1.6.0","typescript":"^5.0.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/merge-streams_1.0.3_1766589761704_0.018406690054118657","host":"s3://npm-registry-packages-npm-production"}},"1.0.4":{"name":"@bitofsky/merge-streams","version":"1.0.4","keywords":["stream","merge","csv","json","arrow","apache-arrow","concat","streaming","fetch"],"author":{"name":"BumSeok Hwang","email":"bitofsky@naver.com"},"license":"MIT","_id":"@bitofsky/merge-streams@1.0.4","maintainers":[{"name":"bitofsky","email":"bitofsky@naver.com"}],"homepage":"https://github.com/bitofsky/merge-streams#readme","bugs":{"url":"https://github.com/bitofsky/merge-streams/issues"},"dist":{"shasum":"e80d0503104434b285959d92d77fe295cab2d91b","tarball":"https://registry.npmjs.org/@bitofsky/merge-streams/-/merge-streams-1.0.4.tgz","fileCount":27,"integrity":"sha512-2IgRX1gY8ODrY3QJrHlMKJ9LUUoYzOkZoFjDQLfy/20fmDxMn3m7Ymd3G6ubUKtjfHSHKa5Is/R+Pvv+8fEb1g==","signatures":[{"sig":"MEQCIE9BMoJCoVtfYiZmN6JRXtjSl1ztB2ffLSAKkewyn+PYAiATcpPeqMu/w1RGlHZKEacfcGhBMuwQ/btbiinhNKEHeQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bitofsky%2fmerge-streams@1.0.4","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":43685},"main":"dist/index.js","type":"module","types":"dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"4065d2a18005af4c5a2dfe96a1104d0c5c8f3a0a","scripts":{"test":"vitest run","build":"tsc","test:watch":"vitest","prepublishOnly":"npm run build"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:55a58a54-fc84-407f-8354-36ce0cefd340"}},"repository":{"url":"git+ssh://git@github.com/bitofsky/merge-streams.git","type":"git"},"_npmVersion":"11.6.2","description":"Merge multiple chunked streams (CSV, JSON_ARRAY, ARROW_STREAM) into one unified stream - perfect for Databricks External Links","directories":{},"_nodeVersion":"24.12.0","dependencies":{"apache-arrow":"^18.0.0"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^1.6.0","typescript":"^5.0.0","@types/node":"^20.0.0"},"_npmOperationalInternal":{"tmp":"tmp/merge-streams_1.0.4_1766590202547_0.614183131349469","host":"s3://npm-registry-packages-npm-production"}},"1.1.0":{"name":"@bitofsky/merge-streams","version":"1.1.0","description":"Merge multiple chunked streams (CSV, JSON_ARRAY, ARROW_STREAM) into one unified stream - perfect for Databricks External Links","main":"dist/index.cjs","module":"dist/index.js","types":"dist/index.d.ts","type":"module","exports":{".":{"import":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"require":{"types":"./dist/index.d.cts","default":"./dist/index.cjs"}}},"scripts":{"build":"tsup","build:tsc":"tsc --noEmit","test":"vitest run","test:watch":"vitest","prepublishOnly":"npm run build:tsc && npm run build"},"keywords":["stream","merge","csv","json","arrow","apache-arrow","concat","streaming","fetch"],"author":{"name":"BumSeok Hwang","email":"bitofsky@naver.com"},"license":"MIT","repository":{"type":"git","url":"git+ssh://git@github.com/bitofsky/merge-streams.git"},"homepage":"https://github.com/bitofsky/merge-streams#readme","bugs":{"url":"https://github.com/bitofsky/merge-streams/issues"},"engines":{"node":">=20.0.0"},"dependencies":{"apache-arrow":"^18.0.0"},"devDependencies":{"@types/node":"^20.0.0","tsup":"^8.5.1","typescript":"^5.0.0","vitest":"^1.6.0"},"gitHead":"a28efebcf816024297ac9a0c1c69b63438b3569c","_id":"@bitofsky/merge-streams@1.1.0","_nodeVersion":"24.12.0","_npmVersion":"11.6.2","dist":{"integrity":"sha512-Lqz+4hWyCJwJb6T2Y8YWcAYnsmsfMO0RVPop/Fg2pb9ZrGW48xjuyLy5nPtUs9xobSwJ/VKpgfhHElChKsPnLQ==","shasum":"f64b2f2ae523f0dd81f661c9bbe666820abc03f2","tarball":"https://registry.npmjs.org/@bitofsky/merge-streams/-/merge-streams-1.1.0.tgz","fileCount":9,"unpackedSize":86828,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@bitofsky%2fmerge-streams@1.1.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQDR8oOD9/TRagDnJIyXHqtU/3LrxD9EaD5ZPFzNFZvd2AIhAMUhrut07SpxvdYAcqCMXz/7H1xwYqR7Xn7I4UOfr/Id"}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:55a58a54-fc84-407f-8354-36ce0cefd340"}},"directories":{},"maintainers":[{"name":"bitofsky","email":"bitofsky@naver.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/merge-streams_1.1.0_1766636967716_0.23908275471510443"},"_hasShrinkwrap":false}},"time":{"created":"2025-12-24T14:41:34.711Z","modified":"2025-12-25T04:29:28.210Z","1.0.0":"2025-12-24T14:41:34.972Z","1.0.3":"2025-12-24T15:22:41.860Z","1.0.4":"2025-12-24T15:30:02.700Z","1.1.0":"2025-12-25T04:29:27.854Z"},"bugs":{"url":"https://github.com/bitofsky/merge-streams/issues"},"author":{"name":"BumSeok Hwang","email":"bitofsky@naver.com"},"license":"MIT","homepage":"https://github.com/bitofsky/merge-streams#readme","keywords":["stream","merge","csv","json","arrow","apache-arrow","concat","streaming","fetch"],"repository":{"type":"git","url":"git+ssh://git@github.com/bitofsky/merge-streams.git"},"description":"Merge multiple chunked streams (CSV, JSON_ARRAY, ARROW_STREAM) into one unified stream - perfect for Databricks External Links","maintainers":[{"name":"bitofsky","email":"bitofsky@naver.com"}],"readme":"# @bitofsky/merge-streams\n\n**When Databricks gives you 90+ presigned URLs, merge them into one.**\n\n> *Because nobody wants to explain to their MCP client why it needs to juggle dozens of chunk URLs.*\n\n---\n\n## Why I Made This\n\nI was building an MCP Server that queries Databricks SQL for large datasets. I chose External Links format because INLINE would blow up memory.\n\nBut then Databricks handed me back something like this:\n\n```\nchunk_0.arrow (presigned URL)\nchunk_1.arrow (presigned URL)\nchunk_2.arrow (presigned URL)\n...\nchunk_89.arrow (presigned URL)\n```\n\nMy client would have to:\n1. Fetch each chunk sequentially\n2. Parse and merge them correctly (CSV headers? JSON array brackets? Arrow EOS markers?)\n3. Handle errors across 90 HTTP requests\n4. Pray nothing times out\n\nThat was unacceptable. So I built this.\n\n---\n\n## The Solution\n\n`merge-streams` takes those chunked External Links and merges them into a single, unified stream.\n\n```\n90+ presigned URLs → merge-streams → 1 clean stream → S3 → 1 presigned URL\n```\n\nNow my MCP client gets one URL. Done.\n\n### What Makes It Fast\n\n- **Pre-connected**: Next chunk's connection opens while current chunk streams. No idle time.\n- **Zero accumulation**: Pure stream piping. Memory stays flat regardless of data size.\n- **Format-aware**: Not byte concatenation — actual format understanding.\n\n---\n\n## Features\n\n- **CSV**: Automatically deduplicates headers across chunks\n- **JSON_ARRAY**: Properly concatenates JSON arrays (handles brackets and commas)\n- **ARROW_STREAM**: Merges Arrow IPC streams batch-by-batch (doesn't just byte-concat)\n- **Memory-efficient**: Streaming-based, never loads entire files into memory\n- **AbortSignal support**: Cancel mid-stream when needed\n- **Progress tracking**: Monitor merge progress with byte-level granularity\n\n---\n\n## Installation\n\n```bash\nnpm install @bitofsky/merge-streams\n```\n\nRequires Node.js 20+ (uses native `fetch()` and `Readable.fromWeb()`)\n\n---\n\n## Quick Start: The Databricks Use Case\n\nSee [test/databricks.spec.ts](test/databricks.spec.ts) for a complete working example.\n\n```bash\n# Run the integration test\nDATABRICKS_TOKEN=dapi... \\\nDATABRICKS_HOST=xxx.cloud.databricks.com \\\nDATABRICKS_HTTP_PATH=/sql/1.0/warehouses/xxx \\\nnpm test -- test/databricks.spec.ts\n```\n\n---\n\n## API\n\n### URL-based (for Databricks External Links)\n\n```ts\nimport { mergeStreamsFromUrls } from '@bitofsky/merge-streams'\n\nawait mergeStreamsFromUrls('CSV', { urls, output })\nawait mergeStreamsFromUrls('JSON_ARRAY', { urls, output })\nawait mergeStreamsFromUrls('ARROW_STREAM', { urls, output })\n```\n\n### With AbortSignal\n\n```ts\nconst controller = new AbortController()\n\nawait mergeStreamsFromUrls('CSV', {\n  urls,\n  output,\n  signal: controller.signal,\n})\n\n// Cancel anytime\ncontroller.abort()\n```\n\n### With Progress Tracking\n\n```ts\nawait mergeStreamsFromUrls('CSV', {\n  urls,\n  output,\n  onProgress: ({ inputIndex, totalInputs, inputedBytes, mergedBytes }) => {\n    console.log(`Processing ${inputIndex + 1}/${totalInputs}: ${inputedBytes} bytes read, ${mergedBytes} bytes merged`)\n  },\n})\n```\n\n### Stream-based (for custom input sources)\n\n```ts\nimport { mergeStreams, mergeCsv, mergeJson, mergeArrow } from '@bitofsky/merge-streams'\n\n// Using unified API\nawait mergeStreams('CSV', { inputs, output })\n\n// Or use format-specific functions directly\nawait mergeCsv({ inputs, output, signal })\nawait mergeJson({ inputs, output, signal })\nawait mergeArrow({ inputs, output, signal })\n```\n\nInputs can be:\n- `Readable` streams directly\n- Sync factories: `() => Readable`\n- Async factories: `() => Promise<Readable>` (recommended for lazy fetching)\n\n---\n\n## Format Details\n\n| Format | Behavior |\n|--------|----------|\n| `CSV` | Writes header once, skips duplicate headers from subsequent chunks |\n| `JSON_ARRAY` | Wraps in `[]`, strips brackets from chunks, inserts commas |\n| `ARROW_STREAM` | Re-encodes RecordBatches into single IPC stream (not byte-concat) |\n\n---\n\n## Types\n\n```ts\nimport type { Readable, Writable } from 'node:stream'\n\ntype MergeFormat = 'ARROW_STREAM' | 'CSV' | 'JSON_ARRAY'\ntype InputSource = Readable | (() => Readable) | (() => Promise<Readable>)\n\ninterface MergeOptions {\n  inputs: InputSource[]\n  output: Writable\n  signal?: AbortSignal\n  onProgress?: (progress: MergeOptionsProgress) => void\n  progressIntervalMs?: number  // Throttle interval (default: 1000, 0 = no throttle)\n}\n\ninterface MergeOptionsProgress {\n  inputIndex: number    // Index of the input being processed\n  totalInputs: number   // Total number of inputs\n  inputedBytes: number  // Total bytes read from all inputs\n  mergedBytes: number   // Total bytes written to output\n}\n\nfunction mergeStreams(\n  format: MergeFormat,\n  options: MergeOptions\n): Promise<void>\n\nfunction mergeStreamsFromUrls(\n  format: MergeFormat,\n  options: { urls: string[]; output: Writable; signal?: AbortSignal; onProgress?: (progress: MergeOptionsProgress) => void; progressIntervalMs?: number }\n): Promise<void>\n```\n\n---\n\n## Why Not Just Byte-Concatenate?\n\n- **CSV**: You'd get duplicate headers scattered throughout\n- **JSON_ARRAY**: `[1,2][3,4]` is not valid JSON\n- **Arrow**: Most Arrow readers stop at the first EOS marker\n\nEach format needs format-aware merging. That's what this library does.\n\n---\n\n## Scope\n\nThis library was born from a specific pain point: making Databricks External Links usable in MCP Server development. It does that one thing well.\n\nIf you have other use cases in mind, PRs are welcome.\n\n---\n\n## License\n\nMIT\n","readmeFilename":"README.md"}