/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * Convert a jsonl of LabeledRow objects to a Parquet file matching the schema the corpus writes. * * DuckDB reproduces the PyArrow file's logical schema and column order. A PyArrow reader therefore sees an identical table. * The trainer reads Parquet by column name, so physical layout does not affect it. * * Schema: `PARQUET_COLUMNS` from `#parquet/schema`, the same list the native writer uses. * * Every row must contain the span triple. A row without it came from a producer that has not migrated. * The writer would silently drop the character-offset labels if it wrote that row. Fail loudly and report the row number. */ import type { PathBuilderLike } from "path-ts"; /** * Options for {@linkcode jsonlToParquet}. */ export interface JSONLToParquetOptions { /** * The labeled-row jsonl to convert. */ input: PathBuilderLike; /** * The parquet file to write. */ output: PathBuilderLike; /** * Parquet row-group size. * * Default 50000. */ rowGroupSize?: number; } /** * Summary returned by {@linkcode jsonlToParquet}. */ export interface JSONLToParquetSummary { read: number; written: number; outPath: string; } /** * Convert a labeled-row jsonl to a v0.5.0-schema Parquet file. */ export declare function jsonlToParquet(options: JSONLToParquetOptions, report?: (line: string) => void): Promise; //# sourceMappingURL=jsonl-to-parquet.d.ts.map