/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * @file Writes Parquet files and the manifest that describes their contents. * * This module exposes two writers. {@linkcode writeParquetFile} writes one file from rows already in memory. * {@linkcode writeParquetSplits} streams per-split row iterables into as many files as the row cap requires. It * returns the manifest. Both writers create their parent directory, as `@mailwoman/core/fs` writers do * encode rather than repeat at each call site. * * Layout under ``: * * ``` * corpus-v/ * manifest.json * train/ * part-0000.parquet * part-0001.parquet * val/ * part-0000.parquet * test/ * part-0000.parquet * ``` * * Each file caps at `rowsPerFile` (default 1,000,000). Within a file DuckDB writes a row group every * {@linkcode ROW_GROUP_SIZE} rows. The manifest captures every file's path, row count, byte size and SHA-256, the * digest computed by re-reading the file once after close — cheap against the cost of writing it. * * A part is closed by its row count and never by a source boundary, so one part may hold the tail of one source and * the head of the next. A reader must therefore take a file's sources from all of its rows. */ import { type PathBuilderLike } from "path-ts"; import { PARQUET_COMPRESSION, type ParquetRow } from "#parquet/schema"; import type { LabeledRow } from "#types"; import type { SplitName } from "#utils/split"; /** * Write one parquet file from rows held in memory. */ export declare function writeParquetFile(rows: readonly ParquetRow[], path: PathBuilderLike): Promise; /** * Per-file metadata captured in `manifest.json`, one entry per `.parquet` file of a split. * * The `slices` key it sits under is the wire interface the Python loader reads * (`manifest_files` in `corpus_files.py`, with its pre-rename fallback); every corpus * on disk includes it, so the key name is not the writer's to change. */ export interface ParquetFileDescriptor { split: SplitName; path: string; format: "parquet"; compression: typeof PARQUET_COMPRESSION; rows: number; bytes: number; sha256: string; first_source_id: string; last_source_id: string; /** * The file's corpus source slug, when the writer knows it. * * `audit.ts` prefers this over inferring the source from `first_source_id`'s prefix; * {@linkcode writeParquetSplits} itself writes multi-source files and leaves it unset. */ source?: string; } export interface ParquetManifest { corpus_version: string; schema: readonly string[]; rows_per_slice: number; row_group_size: number; slices: ParquetFileDescriptor[]; counts: Record; total_rows: number; } export interface WriteParquetSplitsOptions { /** * Root output directory. * * The corpus version directory is created beneath it. */ outputDir: PathBuilderLike; /** * Corpus version stamped onto rows and into the output directory name. */ corpusVersion: string; /** * Max rows per `.parquet` file. * * Default 1,000,000. */ rowsPerFile?: number; /** * Continue from the files `/MANIFEST.json` already records. * * The manifest is rewritten after each finished file. * A descriptor is appended only after its `.parquet` file closes, so each recorded file is complete. * * A resumed run rereads each descriptor's size. * It discards that many rows from the front of each split's iterable and opens the next file index. * * The caller therefore has to supply the same iterables in the same order. * For `buildCorpus` that means the same labeled files and the same shuffle seed. */ resume?: boolean; } /** * Pre-partitioned labeled-row streams, one per split. * * Callers (`buildCorpus`) decide each row's split inline at align time via `splitForRow` * and route rows to the matching stream. * Splits with no rows can be omitted or passed as an empty iterable. * {@linkcode writeParquetSplits} skips them. */ export type PerSplitRows = Partial>>; /** * Streams labeled rows into `.parquet` files, with one set per split. * It writes the manifest that describes them. * * Splits are processed sequentially so only one file is open at a time. * Rows are staged to newline-delimited JSON with backpressure, then DuckDB * writes the Parquet file from disk. */ export declare function writeParquetSplits(perSplit: PerSplitRows, opts: WriteParquetSplitsOptions): Promise; //# sourceMappingURL=writers.d.ts.map