/** * PURE deterministic regression gate for the replay corpus. * * PURE — compareToBaseline must not import from ../providers; no network, no * Date.now(), no side effects, no filesystem access. All inputs are parameters. * Fresh verdicts are re-derived from frozen eval_details_json — NOT by * re-running the generator or evaluator LLM. * * Regression / improvement classification: * baseline 'pass' + fresh 'fail' → regression * baseline 'fail' + fresh 'pass' → improvement * same verdict → unchanged * * A regression is STRICTLY pass→fail. CaseIds present only in fresh (not in * baseline) are not classified at all — the baseline corpus defines the gate. * * Fresh-verdict derivation rule (deterministic, no LLM): * Parse evalDetailsJson (JSON array of result objects). The fresh verdict is * 'fail' iff any element's failures[] contains an entry with * passed === false AND severity === 'error'. Otherwise 'pass'. * Malformed / missing fields → treated as 'pass' (no error-severity failure). */ import type { BoberConfig } from "../../config/schema.js"; export type Verdict = "pass" | "fail"; export interface ReplayComparison { /** caseIds where baseline was 'pass' and fresh is 'fail'. */ regressions: string[]; /** caseIds where baseline was 'fail' and fresh is 'pass'. */ improvements: string[]; /** caseIds where verdict did not change. */ unchanged: string[]; } /** Extended comparison result returned by runReplayHarness. */ export interface ReplayHarnessResult extends ReplayComparison { /** Total number of cases in the corpus. */ total: number; /** Fresh verdicts keyed by caseId (for CLI table rendering). */ fresh: Map; /** Baseline verdicts keyed by caseId (for CLI table rendering). */ baseline: Map; } /** * PURE comparator. Classifies each caseId present in `baseline` as regression, * improvement, or unchanged based on the transition of verdict. * * PURE — no clock, no fs, no LLM. All inputs are parameters. * * @param baseline Map from the captured corpus. * @param fresh Map re-derived from frozen evalDetailsJson. * @returns Sorted arrays for deterministic, byte-identical output. */ export declare function compareToBaseline(baseline: Map, fresh: Map): ReplayComparison; /** * Open the Sprint-1 ReplayStore, re-derive fresh verdicts deterministically * from each case's frozen evalDetailsJson, and return a regression/improvement * breakdown via compareToBaseline. * * NEVER re-runs the generator or evaluator LLM — fresh verdicts come only from * the frozen captured eval_details_json (the deterministic fresh-verdict rule). * * @param projectRoot Absolute path to the project root. * @param config Already-loaded BoberConfig. config.selfImprove is OPTIONAL. */ export declare function runReplayHarness(projectRoot: string, config: BoberConfig): Promise; //# sourceMappingURL=replay-harness.d.ts.map