import type { EvaluationScore, Experiment } from '../types/evaluate.js'; /** How one metric moved between two experiments. */ export interface MetricComparison { /** The score key. */ key: string; /** Mean in the baseline. */ baseline: number; /** Mean in the candidate. */ candidate: number; /** Mean of the per-example differences, candidate minus baseline. */ delta: number; /** 95% interval for the delta, from a paired bootstrap. Excludes zero when the change is real. */ ci95: [number, number]; /** `better`, `worse`, or `unchanged` — unchanged when the interval spans zero. */ verdict: 'better' | 'worse' | 'unchanged'; } /** How one example's score moved. */ export interface ExampleComparison { /** The example. */ exampleId: string; /** The score key. */ key: string; /** Its score in the baseline. */ baseline: number; /** Its score in the candidate. */ candidate: number; /** The change, signed so a positive number is always an improvement. */ delta: number; } /** What changed between a baseline experiment and a candidate. */ export interface ExperimentComparison { /** The baseline experiment. */ baseline: { id: string; name: string; }; /** The candidate experiment. */ candidate: { id: string; name: string; }; /** Present when the two experiments did not run over the same dataset version. */ datasetMismatch?: { baseline: string; candidate: string; }; /** One entry per metric both experiments scored. */ metrics: MetricComparison[]; /** The examples that moved most, worst first. */ regressions: ExampleComparison[]; /** The examples that improved most, best first. */ improvements: ExampleComparison[]; /** True when any metric got worse beyond noise, or a new error appeared. */ regressed: boolean; /** Examples that failed in the candidate but not the baseline. */ newErrors: string[]; /** Examples that failed in the baseline but not the candidate. */ fixedErrors: string[]; } /** * Options for `compareExperiments()`. `latency`, `total-cost`, and `cost` are lower-is-better * unless replaced. */ export interface CompareOptions { /** Metrics where a lower number is better, such as latency or cost. */ lowerIsBetter?: readonly string[]; /** Bootstrap resamples used for the interval. Defaults to 2,000. */ resamples?: number; /** Deterministic sampling, so a comparison in CI is reproducible. Defaults to a fixed seed. */ seed?: number; /** Examples listed in `regressions` and `improvements`. Defaults to 10. */ topExamples?: number; } /** * Compares two experiments and says whether the change is real. * * A mean that moved from 0.81 to 0.83 says nothing on its own — with twenty examples that is noise, * and shipping on it is how a quality gate becomes a coin toss. The interval comes from a paired * bootstrap over the per-example deltas, which needs no assumption about how the scores are * distributed, and pairing is what removes the variation caused by the examples themselves. */ export declare function compareExperiments(baseline: Experiment, candidate: Experiment, options?: CompareOptions): ExperimentComparison; /** A comparison as text, for a pull-request comment or a CI log. */ export declare function formatComparison(comparison: ExperimentComparison): string; /** Summary scores as a map, for asserting on one in a test or a gate. */ export declare function scoreMap(scores: EvaluationScore[]): Record;