/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * @file The corpus parquet schema defines its columns and logical types. It also defines the projection from a `LabeledRow`. * * Every file in this family agrees on this one definition. A column added here and nowhere else fails to compile * against {@linkcode ParquetRow}, which keeps the writer, the reader and the manifest's `schema` key from drifting * apart. * * Compression is `snappy` throughout. PyArrow uses it by default. It is also the standard ML-corpus codec. A reader * outside this repository therefore opens the files without configuration. */ import type { LabeledRow } from "#types"; /** * Row groups are written at this cadence within a parquet file. */ export declare const ROW_GROUP_SIZE = 50000; /** * Rows per parquet file when a writer's caller names no limit. * * A file is closed here rather than grown, because `writeParquetFile` materializes * one Arrow table and Arrow's list builder raises before a whole source fits: * 4,228,212 rows raised where 1,603,143 wrote. * The manifest records the value in force as `rows_per_slice`. */ export declare const ROWS_PER_FILE = 1000000; /** * Snappy is the codec selected for corpus parquet files. */ export declare const PARQUET_COMPRESSION = "SNAPPY"; export interface ParquetFieldDefinition { type: "UTF8" | "INT32"; compression: typeof PARQUET_COMPRESSION; repeated?: boolean; optional?: boolean; } export type ParquetSchemaDefinition = Record, ParquetFieldDefinition>; /** * A single Parquet row shape. * * The index signature allows callers to retain source fields before projection. * * Optional fields are represented as null in the Arrow table and read back as null. */ export interface ParquetRow { raw: string; tokens: readonly string[]; labels: readonly string[]; span_starts: readonly number[]; span_ends: readonly number[]; span_tags: readonly string[]; country: string; locale?: string | null; source: string; source_id: string; corpus_version: string; license: string; register?: string | null; surface: string; recipe?: string | null; base_source_id?: string | null; [key: string]: unknown; } /** * Column names emitted into every parquet file. * * Matches `ParquetRow`. */ export declare const PARQUET_COLUMNS: readonly ["raw", "tokens", "labels", "span_starts", "span_ends", "span_tags", "country", "locale", "source", "source_id", "corpus_version", "license", "register", "surface", "recipe", "base_source_id"]; /** * The DuckDB type each column is read and written as. * * Paired with {@linkcode PARQUET_COLUMNS} so a `read_json` column map and a `copy` * select list are built from one list rather than two that can disagree. */ export declare const PARQUET_COLUMN_TYPES: Record<(typeof PARQUET_COLUMNS)[number], string>; /** * Parquet schema for `LabeledRow`. * * Optional fields set `optional: true`. * Repeated UTF8 columns capture the tokens and labels arrays. */ export declare const LABELED_ROW_SCHEMA: ParquetSchemaDefinition; /** * Project a labeled row to the Parquet schema. * * The span triple is required because `alignRow` emits it on every labeled row. * A row without it came from a producer that has not migrated. * * The writer would drop the labels from the file if it wrote that row. * A thrown error identifies the row instead. */ export declare function rowToParquet(row: LabeledRow): ParquetRow; //# sourceMappingURL=schema.d.ts.map