/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * @file What reached the trainer, as distinct from what the corpus holds. * * `freezeTrainingManifest` records the sources that contributed rows to a corpus. A run reads that * corpus through a config. Four stages between them remove rows: `country_weights` rejects a row * whose country it does not list. `source_weights` draws only the sources it lists. A source set to `0.0` * is drawn zero times. `augment_exclude_sources` withholds a source from augmentation. A model card * quoting the corpus manifest therefore attributes sources the checkpoint never saw. * * Measured over `v0.6.0-register-surface` with `v6.0.0-register-surface-60k.yaml` on 2026-09-28: the corpus's * frozen manifest names 11 sources. The audited epoch emitted 9 of those sources and 48 sources * in total. The config admits 136 countries. Forty countries drew rows, so 96 admitted countries drew zero of * the 1,000,000 rows the epoch emitted. * * **What this establishes and what it does not.** The counts come from one audited epoch at one seed. * A source drawn zero times here is one the sampler did not reach in that epoch. The measurement does not show * that no epoch can reach it. The audit reports its seed and draw count. This record stores both. A source the corpus holds * and the config omits is excluded by construction. The record reports that exclusion instead of a zero draw. */ import type { PathBuilder } from "path-ts"; import type { TrainingManifest } from "#source-register/training-manifest"; /** * Where an `audit_epoch_mixture --json` output for one config is looked for. * * Under the data root rather than committed because the file is a measurement of one * corpus at one seed and takes about ten minutes to produce. * The config determines the filename so two training arms' audits stay apart. * * This avoids the mismatch {@linkcode deriveEffectiveTrainingManifest} refuses. */ export declare function epochMixtureAuditPath(configPath: string): PathBuilder; /** * Why a source the corpus holds contributed no row to the audited epoch. */ export declare const ExclusionReason: { /** * The config's `source_weights` map does not name the source, so the sampler never considers it. */ readonly UnweightedSource: "unweighted-source"; /** * `source_weights` names the source at `0.0`. */ readonly ZeroWeight: "zero-weight"; /** * The source is weighted and the audited epoch drew zero of its rows. * * This is a statement about one seed and one epoch length rather than about the source. */ readonly DrewZeroRows: "drew-zero-rows"; }; export type ExclusionReason = (typeof ExclusionReason)[keyof typeof ExclusionReason]; /** * One source as the audited epoch saw it. */ export interface EffectiveSourceRecord { source: string; /** * Rows the corpus holds for this source, from the frozen manifest. */ corpusRows: number; license: string; /** * The weight `source_weights` gives it, or `null` where the map omits it. */ weight: number | null; /** * Rows the sampler drew, before augmentation. */ drawnRows: number; /** * Rows that survived augmentation and reached the trainer. */ emittedRows: number; /** * Whether `augment_exclude_sources` withholds this source from augmentation. */ augmentExcluded: boolean; /** * Why it contributed no row, or `null` where it contributed at least one. */ excludedBecause: ExclusionReason | null; } /** * Which stage of the draw chain a manifest's row counts come from. * * The chain runs eligible rows, requested mixture, expected draws, realized draws, checkpoint. * A manifest records the last stage it could read, and a reader treats an earlier * stage as an estimate of the later ones rather than as the fact. */ export declare const DrawEvidence: { /** * The epoch-mixture audit replays epoch 1 of the sampler with the trainer's seed. * * It reconstructs the draw. * The trainer's own count is {@link DrawEvidence.TrainerLogged}. */ readonly ReplayedEpochAudit: "replayed-epoch-audit"; /** * The trainer's `exposure-realized-from-.json`, which counts the rows of every batch it trained on. */ readonly TrainerLogged: "trainer-logged"; }; export type DrawEvidence = (typeof DrawEvidence)[keyof typeof DrawEvidence]; /** * The head a checkpoint trained: its label set and where each head tag sits in the shared * semantic tag registry (`corpus-python/src/mailwoman_train/semantic_tags.py`). */ export interface LabelSetContract { labelSetID: string; semanticTagRegistryVersion: number; /** * The global semantic id of each head tag, in head order. */ headMapping: number[]; } /** * An attribution a model card owes because a source under that license reached the trainer. */ export interface RequiredAttribution { license: string; /** * The training sources carrying the license. */ sources: string[]; /** * The statement to print, or `null` where no statement is recorded for the license * and the attribution has to be written from each source's register entry. */ statement: string | null; } /** * The attributions owed by the sources that reached the trainer, grouped by license. * * Only `trainingSources` are read, so a source the corpus holds and the run never drew owes no attribution. * A license is included when `summarizeLicense` records an attribution obligation for it, or * when it is unrecognized, since an unrecognized license's obligations are unknown rather than absent. */ export declare function requiredAttributions(sources: readonly EffectiveSourceRecord[]): RequiredAttribution[]; /** * What one config drew from one corpus in one audited epoch. */ export interface EffectiveTrainingManifest { manifestID: "corpus-effective-training-manifest"; schemaVersion: 2; /** * The stage of the draw chain the row counts below come from. */ drawEvidence: DrawEvidence; /** * The head the run trained, or `null` where the derivation was given none. */ labelSet: LabelSetContract | null; /** * The attributions the sources in {@linkcode trainingSources} owe. */ requiredAttributions: RequiredAttribution[]; /** * The corpus the rows came from and the digest of its frozen manifest. * A reader can identify which record this was derived from. */ corpusVersion: string; corpusManifestDigest: string; /** * The config file the audit ran under, as the audit recorded it. */ config: string; /** * The seed and draw count of the audited epoch. * Both decide the counts below. */ seed: number; drawsRequested: number; drawsRealized: number; /** * Number of countries `country_weights` admits and number that drew a row. */ admittedCountries: number; countriesDrawingRows: number; /** * Countries admitted at a weight that drew zero rows in this epoch. */ admittedCountriesDrawingZero: string[]; /** * Every source the corpus holds, in descending emitted order. */ sources: EffectiveSourceRecord[]; /** * Sources that reached the trainer. * A model card may attribute this set as training data. */ trainingSources: string[]; /** * Sources the corpus holds that reached no row, each with its reason. */ excludedSources: Record; /** * Sources the epoch emitted that the corpus's frozen manifest does not name, with their emitted rows. * * `freezeTrainingManifest` runs inside `buildCorpus` and records the base build's sources. * An overlay is merged into the corpus afterwards by `overlay-manifest`, * so its rows are in the corpus and outside that manifest. * * Measured over `v0.6.0-register-surface` on 2026-09-28: the frozen manifest names 11 sources * and the audited epoch emitted 48, of which 39 have no entry in it. * Those 39 account for 890,666 of the 1,000,000 rows emitted, 89.1%, so the frozen * manifest's license set covers 10.9% of what trained the model. * * The set keeps this record from reading as a complete one. * A release assertion over it has to treat a non-empty value as an unread input * rather than as an absence of rows. */ emittedButUnrecorded: Record; /** * The three stages a source passes through, for each source in {@linkcode emittedButUnrecorded}. * * The entries above give only the emitted rows. * These give the rows the corpus holds, the rows the epoch drew and the rows * it emitted after augmentation. * * A reader can distinguish a source drawn heavily from one drawn barely. * * `corpusRows` comes from `draw_level.per_source`. * That field counts the corpus files. * * A source outside the frozen manifest has a corpus row count nowhere else. * `-1` records that the audit reported none rather than that the corpus holds no such row. */ unrecordedSourceStages: Record; totalEmittedRows: number; /** * Sha256 over this manifest with `contentDigest` emptied. */ contentDigest: string; } /** * The fields this reads from an `audit_epoch_mixture --json` output. */ export interface EpochMixtureAudit { draw_level?: { totals?: Record; by_country?: Record; admitted_countries_drawn?: Record; admitted_countries_drawing_nothing?: string[]; /** * Per-source detail the audit computes from the corpus files rather than from the corpus manifest. * * `rows` is how many rows of that source the corpus holds. * It is the only place a source outside the frozen manifest has a corpus row count, so it is what * lets a reader read a source's rows in the corpus beside the rows one epoch drew and emitted. */ per_source?: Record; }; emitted_level?: { totals?: Record; }; meta?: { seed?: number; draws_requested?: number; draws_realized?: number; config?: string; }; } /** * The config fields that decide which of a corpus's sources a run reaches. */ export interface EffectiveConfigView { sourceWeights: Readonly>; augmentExcludeSources: readonly string[]; } /** * Read `data.source_weights` and `data.augment_exclude_sources` from a training config. * * A line scanner instead of a YAML parse. * `corpus audit` already reads these blocks this way. * * The `wire-identifiers` check uses the same approach. * The configs use a small hand-written subset: two-space indentation, one entry per line and `#` comments. * They contain no flow syntax. * * YAML 1.1 coerces the bare key `NO` to boolean false. * Norway is a `country_weights` key, so a parser would drop it. * * @throws When the text contains no `source_weights` block. * That map decides which sources the sampler considers. * A config without one is a truncated file rather than a config drawing from every source. */ export declare function readConfigView(text: string): EffectiveConfigView; /** * The digest a manifest should include, computed over the manifest with its digest field emptied. */ export declare function effectiveManifestDigest(manifest: EffectiveTrainingManifest): string; /** * Derive what reached the trainer from the corpus's frozen manifest, one audited epoch and the config. * * @throws When the audit contains no `emitted_level.totals`. * That field answers the question. * An audit without it is either a different report or a truncated one. * A derivation from only `draw_level` would attribute rows augmentation removed. * @throws When the audit ran under a different config file than the one supplied, because the two are * two training arms and one record over both would report a source one of them never weighted. */ export declare function deriveEffectiveTrainingManifest(input: { corpusManifest: TrainingManifest; audit: EpochMixtureAudit; config: EffectiveConfigView; configPath: string; labelSet?: LabelSetContract | null; }): EffectiveTrainingManifest; /** * Sources a record declares as training data that the audited epoch never reached, * plus the reverse mismatch. * * A release reads both directions. * A declared source the checkpoint never saw overstates what the model learned from. * * An emitted source the record omits is an attribution a consumer never receives. */ export declare function provenanceDisagreement(manifest: EffectiveTrainingManifest, declared: readonly string[]): { declaredButNotTrained: string[]; trainedButNotDeclared: string[]; }; /** * Why a release may not assert that a model's declared provenance equals what trained it. * * An empty array is the only value a caller may read as agreement. * It is reachable only when the effective manifest covers every emitted source. * * A `null` manifest means no release read one at all. * The function reports it rather than passing it. * * `declared` is a list of corpus source ids. * A caller holding a model card's attribution entries passes `null`. * * Those entries are prose naming publishers. * Every field of the card contains prose rather than a source id, so comparing the two * lists as strings reports every entry as a source the epoch never drew. * * The refusal list then records the coverage refusal and one line recording the comparison as unmeasured. */ export declare function provenanceRefusals(input: { manifest: EffectiveTrainingManifest | null; declared: readonly string[] | null; packageName: string; }): string[]; //# sourceMappingURL=effective-manifest.d.ts.map