{"_id":"@dotdo/hyparquet-writer","_rev":"2-33d4e2109dac5445ee3eddd4739e3811","name":"@dotdo/hyparquet-writer","dist-tags":{"latest":"0.12.1"},"versions":{"0.12.0":{"name":"@dotdo/hyparquet-writer","version":"0.12.0","keywords":["ai","data","dotdo","hyparquet","ml","parquet","snappy","thrift","variant"],"author":{"name":"dotdo"},"license":"MIT","_id":"@dotdo/hyparquet-writer@0.12.0","maintainers":[{"name":"crisner1978","email":"chris@driv.ly"},{"name":"nathanclevenger","email":"nateclev@gmail.com"}],"homepage":"https://github.com/dot-do/hyparquet-writer","bugs":{"url":"https://github.com/dot-do/hyparquet-writer/issues"},"dist":{"shasum":"31fd6eaebc3f95b6d306ee3852d237d312a3a18f","tarball":"https://registry.npmjs.org/@dotdo/hyparquet-writer/-/hyparquet-writer-0.12.0.tgz","fileCount":73,"integrity":"sha512-u12Td/ESkUFAOb6KVmXgeMH4rjCZkFK8jk1pEhGswCYXHIS+44OmhipEeo/JxBfrnO8BzBbtuVIYzFMsoCE/SQ==","signatures":[{"sig":"MEQCID69yjEnyDuKmrz3YyfhFDEgOwWG2JBskS7RVZyIwyBSAiB7aivIFKi54ID0h0JL0jH2IPNg1L3pR578GHjp2i+EfA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":173578},"main":"src/index.js","type":"module","types":"types/index.d.ts","exports":{".":{"browser":{"types":"./types/index.d.ts","default":"./src/index.js"},"default":{"types":"./types/node.d.ts","default":"./src/node.js"}},"./src/*.js":{"types":"./types/*.d.ts","default":"./src/*.js"}},"gitHead":"5435efe8460f29f9fe843c71a211a6c7ec45c5e9","scripts":{"lint":"eslint","test":"vitest run","prepare":"npm run build:types","coverage":"vitest run --coverage --coverage.include=src","lint:fix":"eslint --fix","build:types":"tsc -p ./tsconfig.build.json"},"_npmUser":{"name":"nathanclevenger","email":"nateclev@gmail.com"},"repository":{"url":"git+https://github.com/dot-do/hyparquet-writer.git","type":"git"},"_npmVersion":"10.9.4","description":"Parquet file writer for JavaScript (dotdo fork with Variant support)","directories":{},"sideEffects":false,"_nodeVersion":"22.21.1","dependencies":{"@dotdo/hyparquet":"^1.17.1"},"_hasShrinkwrap":false,"devDependencies":{"eslint":"9.39.2","vitest":"4.0.18","hysnappy":"1.1.0","typescript":"5.9.3","@types/node":"25.0.10","@vitest/coverage-v8":"4.0.18","eslint-plugin-jsdoc":"62.4.1","@babel/eslint-parser":"7.28.6"},"_npmOperationalInternal":{"tmp":"tmp/hyparquet-writer_0.12.0_1770036917702_0.5652701883158544","host":"s3://npm-registry-packages-npm-production"}},"0.12.1":{"name":"@dotdo/hyparquet-writer","version":"0.12.1","description":"Parquet file writer for JavaScript (dotdo fork with Variant support)","author":{"name":"dotdo"},"homepage":"https://github.com/dot-do/hyparquet-writer","keywords":["ai","data","dotdo","hyparquet","ml","parquet","snappy","thrift","variant"],"license":"MIT","repository":{"type":"git","url":"git+https://github.com/dot-do/hyparquet-writer.git"},"bugs":{"url":"https://github.com/dot-do/hyparquet-writer/issues"},"main":"src/index.js","type":"module","types":"types/index.d.ts","exports":{".":{"browser":{"types":"./types/index.d.ts","default":"./src/index.js"},"default":{"types":"./types/node.d.ts","default":"./src/node.js"}},"./src/*.js":{"types":"./types/*.d.ts","default":"./src/*.js"}},"sideEffects":false,"scripts":{"build:types":"tsc -p ./tsconfig.build.json","coverage":"vitest run --coverage --coverage.include=src","lint":"eslint","lint:fix":"eslint --fix","prepare":"npm run build:types","test":"vitest run"},"dependencies":{"@dotdo/hyparquet":"^1.17.1"},"devDependencies":{"@babel/eslint-parser":"7.28.6","@types/node":"25.0.10","@vitest/coverage-v8":"4.0.18","eslint":"9.39.2","eslint-plugin-jsdoc":"62.4.1","hysnappy":"1.1.0","typescript":"5.9.3","vitest":"4.0.18"},"_id":"@dotdo/hyparquet-writer@0.12.1","gitHead":"5435efe8460f29f9fe843c71a211a6c7ec45c5e9","_nodeVersion":"22.21.1","_npmVersion":"10.9.4","dist":{"integrity":"sha512-BWiMFsX9tCRAozrdyB5hMOjgkjRujyRmoSB0gStg1W3XLOai3kNbojx/SiudpQVvCoxKBAkFw5Gvg5zUo8/KAg==","shasum":"cb6f7c2df2705748c4c0bb1c6ed046519003b051","tarball":"https://registry.npmjs.org/@dotdo/hyparquet-writer/-/hyparquet-writer-0.12.1.tgz","fileCount":73,"unpackedSize":173584,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIFeZ5OQ98+QdsKX34P/a3ncs5oRlpZFFE3uzHsnTRuYfAiEA09lbsnouzan3LM9mBZJn7Um6h64NZBw3Rd3VhbPxrpc="}]},"_npmUser":{"name":"nathanclevenger","email":"nateclev@gmail.com"},"directories":{},"maintainers":[{"name":"crisner1978","email":"chris@driv.ly"},{"name":"nathanclevenger","email":"nateclev@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/hyparquet-writer_0.12.1_1770056339959_0.7371717764395602"},"_hasShrinkwrap":false}},"time":{"created":"2026-02-02T12:55:17.592Z","modified":"2026-02-02T18:19:00.274Z","0.12.0":"2026-02-02T12:55:17.868Z","0.12.1":"2026-02-02T18:19:00.118Z"},"bugs":{"url":"https://github.com/dot-do/hyparquet-writer/issues"},"author":{"name":"dotdo"},"license":"MIT","homepage":"https://github.com/dot-do/hyparquet-writer","keywords":["ai","data","dotdo","hyparquet","ml","parquet","snappy","thrift","variant"],"repository":{"type":"git","url":"git+https://github.com/dot-do/hyparquet-writer.git"},"description":"Parquet file writer for JavaScript (dotdo fork with Variant support)","maintainers":[{"name":"crisner1978","email":"chris@driv.ly"},{"name":"nathanclevenger","email":"nateclev@gmail.com"}],"readme":"# @dotdo/hyparquet-writer\n\n![hyparquet writer parakeet](hyparquet-writer.jpg)\n\n[![npm](https://img.shields.io/npm/v/@dotdo/hyparquet-writer)](https://www.npmjs.com/package/@dotdo/hyparquet-writer)\n[![minzipped](https://img.shields.io/bundlephobia/minzip/@dotdo/hyparquet-writer)](https://www.npmjs.com/package/@dotdo/hyparquet-writer)\n[![workflow status](https://github.com/dot-do/hyparquet-writer/actions/workflows/ci.yml/badge.svg)](https://github.com/dot-do/hyparquet-writer/actions)\n[![mit license](https://img.shields.io/badge/License-MIT-orange.svg)](https://opensource.org/licenses/MIT)\n![coverage](https://img.shields.io/badge/Coverage-95-darkred)\n[![dependencies](https://img.shields.io/badge/Dependencies-1-blueviolet)](https://www.npmjs.com/package/@dotdo/hyparquet-writer?activeTab=dependencies)\n\nHyparquet Writer is a JavaScript library for writing [Apache Parquet](https://parquet.apache.org) files. It is designed to be lightweight, fast and store data very efficiently. This is the dotdo fork with Variant support. It is a companion to the [@dotdo/hyparquet](https://github.com/dot-do/hyparquet) library, which is a JavaScript library for reading parquet files.\n\n## Quick Start\n\nTo write a parquet file to an `ArrayBuffer` use `parquetWriteBuffer` with argument `columnData`. Each column in `columnData` should contain:\n\n- `name`: the column name\n- `data`: an array of same-type values\n- `type`: the parquet schema type (optional)\n\n```javascript\nimport { parquetWriteBuffer } from '@dotdo/hyparquet-writer'\n\nconst arrayBuffer = parquetWriteBuffer({\n  columnData: [\n    { name: 'name', data: ['Alice', 'Bob', 'Charlie'], type: 'STRING' },\n    { name: 'age', data: [25, 30, 35], type: 'INT32' },\n  ],\n})\n```\n\nNote: if `type` is not provided, the type will be guessed from the data. The supported `BasicType` are a superset of the parquet primitive types:\n\n| Basic Type | Equivalent Schema Element |\n|------|----------------|\n| `BOOLEAN` | `{ type: 'BOOLEAN' }` |\n| `INT32` | `{ type: 'INT32' }` |\n| `INT64` | `{ type: 'INT64' }` |\n| `FLOAT` | `{ type: 'FLOAT' }` |\n| `DOUBLE` | `{ type: 'DOUBLE' }` |\n| `BYTE_ARRAY` | `{ type: 'BYTE_ARRAY' }` |\n| `STRING` | `{ type: 'BYTE_ARRAY', converted_type: 'UTF8' }` |\n| `JSON` | `{ type: 'BYTE_ARRAY', converted_type: 'JSON' }` |\n| `TIMESTAMP` | `{ type: 'INT64', converted_type: 'TIMESTAMP_MILLIS' }` |\n| `UUID` | `{ type: 'FIXED_LEN_BYTE_ARRAY', type_length: 16, logical_type: { type: 'UUID' } }` |\n| `FLOAT16` | `{ type: 'FIXED_LEN_BYTE_ARRAY', type_length: 2, logical_type: { type: 'FLOAT16' } }` |\n| `GEOMETRY` | `{ type: 'BYTE_ARRAY', logical_type: { type: 'GEOMETRY' } }` |\n| `GEOGRAPHY` | `{ type: 'BYTE_ARRAY', logical_type: { type: 'GEOGRAPHY' } }` |\n\nMore types are supported but require defining the `schema` explicitly. See the [advanced usage](#advanced-usage) section for more details.\n\n### Write to Local Parquet File (nodejs)\n\nTo write a local parquet file in node.js use `parquetWriteFile` with arguments `filename` and `columnData`:\n\n```javascript\nconst { parquetWriteFile } = await import('@dotdo/hyparquet-writer')\n\nparquetWriteFile({\n  filename: 'example.parquet',\n  columnData: [\n    { name: 'name', data: ['Alice', 'Bob', 'Charlie'], type: 'STRING' },\n    { name: 'age', data: [25, 30, 35], type: 'INT32' },\n  ],\n})\n```\n\nNote: hyparquet-writer is published as an ES module, so dynamic `import()` may be required on the command line.\n\n## Advanced Usage\n\nBy default, hyparquet-writer generates parquet files that are optimized for large text datasets and fast previews. Parquet file parameters can be configured via options:\n\n```typescript\ninterface ParquetWriteOptions {\n  writer: Writer // generic writer\n  columnData: ColumnSource[]\n  schema?: SchemaElement[] // explicit parquet schema\n  codec?: CompressionCodec // compression codec (default 'SNAPPY')\n  compressors?: Compressors // custom compressors (default includes snappy)\n  statistics?: boolean // enable column statistics (default true)\n  pageSize?: number // target page size in bytes (default 1 mb)\n  rowGroupSize?: number | number[] // target row group size in rows (default [1000, 100000])\n  kvMetadata?: { key: string; value?: string }[] // extra key-value metadata\n}\n```\n\nNote: `rowGroupSize` can be either constant or an array of row group sizes, with the last size repeating. The default `[1000, 100000]` means the first row group will have 1000 rows, and all subsequent row groups will have 100,000 rows. This is optimized for fast previews of large datasets.\n\nPer-column options:\n\n```typescript\ninterface ColumnSource {\n  name: string\n  data: DecodedArray\n  type?: BasicType\n  nullable?: boolean // allow nulls (default true)\n  encoding?: Encoding // parquet encoding (PLAIN, RLE, DELTA_BINARY_PACKED, BYTE_STREAM_SPLIT, etc)\n  columnIndex?: boolean // enable page-level column index (default false)\n  offsetIndex?: boolean // enable page-level offset index (default true)\n}\n```\n\nExample:\n\n```javascript\nimport { ByteWriter, parquetWrite } from '@dotdo/hyparquet-writer'\nimport { snappyCompress } from 'hysnappy'\n\nconst writer = new ByteWriter()\nparquetWrite({\n  writer,\n  columnData: [\n    { name: 'name', data: ['Alice', 'Bob', 'Charlie'] },\n    { name: 'age', data: [25, 30, 35] },\n    { name: 'dob', data: [new Date(1000000), new Date(2000000), new Date(3000000)] },\n  ],\n  // explicit schema:\n  schema: [\n    { name: 'root', num_children: 3 },\n    { name: 'name', type: 'BYTE_ARRAY', converted_type: 'UTF8' },\n    { name: 'age', type: 'FIXED_LEN_BYTE_ARRAY', type_length: 4, converted_type: 'DECIMAL', scale: 2, precision: 4 },\n    { name: 'dob', type: 'INT32', converted_type: 'DATE' },\n  ],\n  compressors: { SNAPPY: snappyCompresss }, // high performance wasm compressor\n  statistics: false, // disable statistics\n  rowGroupSize: 1000000, // large row groups\n  kvMetadata: [\n    { key: 'key1', value: 'value1' },\n    { key: 'key2', value: 'value2' },\n  ],\n})\nconst arrayBuffer = writer.getBuffer()\n```\n\n## Column Types\n\nHyparquet-writer supports several ways to define the parquet schema. The simplest way is to provide basic types in the `columnData` elements.\n\nIf you don't provide types, the types will be auto-detected from the data. However, it is still recommended that you provide type information when possible. (zero rows would throw an exception, floats might be typed as int, etc)\n\n### Explicit Schema\n\nYou can provide your own parquet schema of type `SchemaElement` (see [parquet-format](https://github.com/apache/parquet-format/blob/master/src/main/thrift/parquet.thrift)):\n\n```typescript\nimport { ByteWriter, parquetWrite } from '@dotdo/hyparquet-writer'\n\nconst writer = new ByteWriter()\nparquetWrite({\n  writer,\n  columnData: [\n    { name: 'name', data: ['Alice', 'Bob', 'Charlie'] },\n    { name: 'age', data: [25, 30, 35] },\n  ],\n  // explicit schema:\n  schema: [\n    { name: 'root', num_children: 2 },\n    { name: 'name', type: 'BYTE_ARRAY', converted_type: 'UTF8', repetition_type: 'REQUIRED' },\n    { name: 'age', type: 'INT32', repetition_type: 'REQUIRED' },\n  ],\n})\n```\n\n### Schema Overrides\n\nYou can use mostly automatic schema detection, but override the schema for specific columns. This is useful if most of the column types can be automatically determined, but you want to use a specific schema element for one particular element.\n\n```javascript\nconst { ByteWriter, parquetWrite, schemaFromColumnData } = await import(\"@dotdo/hyparquet-writer\")\n\n// one unsigned and one signed int column\nconst columnData = [\n  { name: 'unsigned_int', data: [1000000, 2000000] },\n  { name: 'signed_int', data: [1000000, 2000000] },\n]\nconst writer = new ByteWriter()\nparquetWrite({\n  writer,\n  columnData,\n  // override schema for unsigned_int column\n  schema: schemaFromColumnData({\n    columnData,\n    schemaOverrides: {\n      unsigned_int: {\n        name: 'unsigned_int',\n        type: 'INT32',\n        converted_type: 'UINT_32',\n        repetition_type: 'REQUIRED',\n      },\n    },\n  }),\n})\n```\n\n## References\n\n - https://github.com/dot-do/hyparquet\n - https://github.com/dot-do/hyparquet-writer\n - https://github.com/hyparam/hyparquet-compressors\n - https://github.com/apache/parquet-format\n - https://github.com/apache/parquet-testing\n","readmeFilename":"README.md"}