/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * #511 base-consistency lint, GENERALIZED + COUNTRY-SCOPED (v2) — any synthetic slice vs the base. * * Ported from scripts/lint-slice-vocab.py (pyarrow → @duckdb/node-api); behavior preserved * byte-for-byte (same flags, same stdout, same verdicts). The base-root default routes through * `dataRootPath` so the lab `/mnt/playpen` literal stays in its one home * (core/utils/data-root.ts) and `$MAILWOMAN_DATA_ROOT` is honored; with the env unset it equals * the Python default. * * The #511 lesson: a synthetic slice must not label a token a tag the BASE dominantly labels * something else, or training gets conflicting gradients on the same token and the minority (the * slice) loses. This reads a slice's own (token -> tag) and checks each token against the base. * * WHY v2 IS COUNTRY-SCOPED + FULL-COUNT (the night-2026-06-18 lesson, learned the hard way over * three tries): a token's correct tag is COUNTRY-specific — "Paris" is locality in FR data and * street in US "Paris Ave"; "Marion" is a US town AND many US "Marion" streets. So: * * 1. A cross-COUNTRY aggregate mis-judges any country-specific token (v1 uniform AND a proportional * retry both false-flagged FR cities as "street" from US street-contexts). * 2. A SMALL sample is street-BIASED regardless, because the street sources (tiger 39 + nad 378 parts) * dwarf the locality sources (a small US-scoped spot-check read Indianapolis 54% street vs * its true 219700:29 LOCALITY). The fix: tally each slice token's base tag SCOPED to the * country the slice uses it in (the base has a `country` column), over a LARGE/FULL scan * (`fraction`, default 1.0). Pure-numeric tokens excluded (house_number/postcode are * context-determined). An affix-split flag (slice street_suffix/_prefix vs base "street") is * EXPECTED — the loader's affix-relabel handles it; weigh those separately. * * Usage: mailwoman dev lint slice-vocab --slice * [--base-version v0.5.0] [--base-root ] [--fraction 1.0] [--threshold 0.7] [--min-count * 50] */ /** * Options for {@linkcode lintSliceVocab}. */ export interface LintSliceVocabOptions { /** * The slice parquet to lint. */ slice: string; /** * Base corpus version. Default `v0.5.0`. */ baseVersion?: string; /** * Base corpus root. Default `$MAILWOMAN_DATA_ROOT/corpus/versioned`. */ baseRoot?: string; /** * Base-majority confidence floor for a contradiction. Default 0.7. */ threshold?: number; /** * Minimum base support to judge a token. Default 50. */ minCount?: number; /** * Fraction of base parts to scan (proportional per-source slice below 1.0). Default 1.0. */ fraction?: number; } /** * One contradiction row: token, slice tag, base tag, base fraction, base total. */ export type SliceVocabRow = [token: string, sliceTag: string, baseTag: string, baseFrac: number, baseTotal: number]; /** * Findings summary returned by {@linkcode lintSliceVocab}. */ export interface LintSliceVocabSummary { /** * Real contradictions — the command exits 1 when nonzero. */ errors: number; /** * Affix-split rows (EXPECTED — the loader's affix-relabel handles them). */ warnings: number; findings: { contradictions: SliceVocabRow[]; affixSplits: SliceVocabRow[]; }; } /** * Lint a synthetic slice's (token → tag) vocabulary against the base corpus, country-scoped. */ export declare function lintSliceVocab(options: LintSliceVocabOptions): Promise; //# sourceMappingURL=vocab.d.ts.map