/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * @file Merges one source's overlay parquet files into shuffled output files. * * The training sampler can read a whole per-source draw from one row group, * so a source split into per-country files would reach only one country per draw. * A shuffle across the inputs mixes each source's countries within each row group. */ import type { PathBuilderLike } from "path-ts"; /** * Options for {@link mergeSourceFiles}. */ export interface MergeSourceOptions { /** * The parquet files to merge, read in the order given. */ inputs: readonly PathBuilderLike[]; /** * The output path. * Each written file receives a five-digit index in its stem. */ output: PathBuilderLike; /** * The number of rows held in memory while shuffling. * It defaults to the corpus writer's window. */ windowSize?: number; /** * The shuffle seed. * * The default matches the corpus writer's seed. * This makes rebuilds reproducible. */ seed?: number; /** * The maximum rows per output file. * * `writeParquetFile` builds one Arrow table per file. * Arrow's list builder overflows on very large tables. */ rowsPerFile?: number; } /** * The files that {@link mergeSourceFiles} wrote and the row counts they hold. */ export interface MergeSourceResult { inputs: readonly string[]; /** * The files written, in order. */ outputs: readonly string[]; rows: number; /** * Row counts per `country` value. */ byCountry: Record; /** * Row counts per `source` value. * A successful merge has one entry. */ bySource: Record; } /** * Merges the inputs through a windowed shuffle into one or more parquet files and reports their contents. * * @throws When the inputs contain more than one `source`, because a manifest * entry records one source per file. * The check runs after the stream ends, so files flushed before it remain on disk. */ export declare function mergeSourceFiles(options: MergeSourceOptions): Promise; //# sourceMappingURL=merge-source.d.ts.map