/** * Judging the frozen set with the skills exactly as they stand. * * This is the gate, and it runs the real pipeline rather than a simplified one: * the judge files, then the adversarial refuter gets to kill findings before * anything is counted, once per view-group batch exactly as `check` runs it * (src/check/batches.ts), because an amendment to either skill has to be graded * in the context shape and index space it will actually run in. * * One deliberate divergence from `check`: the judge is NOT shown the "ALREADY * FILED" aid that production builds from the live backlog (src/check/plan.ts). * That aid exists to keep the judge's freehand `attribute` stable, and the gate * ignores the attribute on purpose, deciding a claim at its panel's granularity * instead (src/skills/verdict.ts). Rebuilding the aid from the frozen claims would * hand the judge the answer key: every must-file claim would appear as an * already-open defect the aid instructs the judge to re-file by name, so an * amendment that blinded the judge could still pass the "lost" check by * parroting the list. Each frozen claim was first filed by a judge that had no * such aid; the replay holds every candidate to the same conditions. */ import { type AiFinding } from "../judge/engine.js"; import { type RegressionSet } from "./regression.js"; import { type Drift, type Violation } from "./verdict.js"; import { type ResolvedConfig } from "../types.js"; /** * The skills the frozen set can actually exercise. An amendment to anything * else is written down as a proposal rather than applied: auto-applying a * change nothing can grade is the exact thing the gate exists to prevent. */ export declare const GATED_SKILLS: Set; export interface ReplayScope { /** * The skill the candidate amendment touched. A panel amendment changed only * that panel's prompt, so only that panel replays: every other judge's * bytes, and therefore its verdicts, are identical to the run that settled * the claims. An amendment to the core or the refuter changes every * prompt, so every claim-owning panel replays. */ amendedSkill?: string; /** * Re-judge only the view groups these shots belong to. * * A confirmation round re-tests what failed rather than the whole set, and a * control round re-tests it with the candidate withdrawn. The unit is the view * GROUP and not the shot because the rubric compares within a group, so * dropping a shot's siblings would change the question being asked. */ onlyShots?: ReadonlySet; } export declare function replayRegression(resolved: ResolvedConfig, set: RegressionSet, model: string, scope?: ReplayScope): Promise<{ violations: Violation[]; drift: Drift[]; findings: AiFinding[]; costUsd: number; }>;