/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * Pre-compute corpus-wide token + bigram label distributions for the corpus linter. * * Reads one or more parquet files, builds per-(token, label) and per-(bigram, label-bigram) * histograms. The tool serializes them as JSON. `lint/recipe-output/index.ts` consumes the output file * as the baseline against which a new recipe output is compared. * * Stats are cheap to compute (~5–30s per 100K rows) but expensive enough that we cache them between * linter invocations. Re-run this whenever the corpus changes substantially (a new mainline * recipe output added, a source-pool re-weighted, etc.). * * Output schema (`slice_paths` is the stats file's own key. a stats file already on disk includes it, * so the linter reads it under that spelling): * * ```ts * interface CorpusStats { * row_count: number * slice_paths: string[] * tokens: { [token: string]: { [label: string]: number } } * bigrams: { [token_bigram: string]: { [label_bigram: string]: number } } * // token_bigram = "tok1tok2" (US sep), label_bigram = "lab1lab2" * // For memory: only keep bigrams with count >= MIN_BIGRAM_COUNT (2). * } * ``` * * Usage: mailwoman corpus stats\ * --parquet \ * --out * * For a quick local-corpus baseline (limited but useful for linter testing): mailwoman corpus stats\ * --parquet $MAILWOMAN_DATA_ROOT/corpus/versioned/v0.4.0/corpus-v0.4.0/train/\ * --out /tmp/corpus-stats-local.json */ /** * Options for {@linkcode buildCorpusStats}. */ export interface CorpusStatsOptions { /** * A directory of parquet files, one parquet file, or a literal path. */ parquetPath: string; outputPath: string; /** * Read at most this many rows per parquet file. */ limitPerFile?: number; } export declare function buildCorpusStats(args: CorpusStatsOptions): Promise; //# sourceMappingURL=corpus-stats.d.ts.map