export type WindtunnelVerdictKind = 'PASS' | 'HONEST-NEGATIVE' | 'FAIL'; export type GroundTruthLabel = 'TP' | 'FP'; export interface CullLedgerEntry { ruleId: string; pr: number; filePath: string; matchedLine: string; reason: 'negative-control-fired'; } export interface WindtunnelDiagnostics { /** * Precision (TP/(TP+FP)) over labeled, surviving (non-culled, * non-negative-control) firings. DESCRIPTIVE ONLY — informative even on * no-claim verdicts ("culled 8/10, the 2 survivors were clean"). NEVER * consulted for the gate decision and never mistakable for the certifying * `precision`. null when no surviving firing is labeled. */ survivorPrecision: number | null; } export interface WindtunnelVerdict { verdict: WindtunnelVerdictKind; /** * Certifying precision claim. A real value ONLY on verdicts that make a * precision claim: PASS (1.0) and confirmed-FP FAIL (the breaching value, * which IS the evidence). `null` on every no-claim verdict (exposure-floor / * cull-rate / needs-adjudication HONEST-NEGATIVE, and vacuous-control FAIL). * `null` ⟺ no precision claim; `0` is reserved for a real all-FP measurement * and NEVER means "not computed" (#2189 ruling, strategy-claude 2026-06-17). */ precision: number | null; mintedRuleCount: number; culledCount: number; survivingRuleCount: number; /** 3-tuple: [activeRulesEvaluated, filesTouchedInWindow, positiveControlsExercised]. Never collapsed to a product. */ exposureTuple: [number, number, number]; cullLedger: CullLedgerEntry[]; /** True when all positive controls fired their target rule. */ nonVacuity: boolean; /** Label ids of firings with no ground-truth label (operator adjudication required). */ needsAdjudication: string[]; /** Separately-namespaced descriptive diagnostics — never part of the gate decision. */ diagnostics: WindtunnelDiagnostics; } /** * fold-D — one raw engine match that collapsed into a logical `RuleFiring`. * A rule can match more than once under the same `labelId` (multiple AST nodes * on one line, or distinct physical lines whose only difference is trailing * whitespace that `normalizeMatchedLine` drops). Each such raw match is retained * here so the cert-run report can show what backed a deduped firing. */ export interface FiringEvidence { /** 1-based physical line number of the raw match in the post-image. */ lineNumber: number; /** The raw (pre-normalization) matched line text. */ rawLine: string; } export interface RuleFiring { ruleId: string; pr: number; filePath: string; matchedLine: string; controlKind: 'corpus' | 'positive' | 'negative'; /** For positive controls: the rule that MUST fire to prove non-vacuousness. */ targetRuleId?: string; /** Content-based label id (A2). Compute via firingLabelId from windtunnel-lock. */ labelId: string; /** * fold-D — the raw engine matches that collapsed into this one logical firing. * `buildFirings` dedups same-`labelId` matches to a SINGLE firing (the collapse * is verdict-safe under ADR-110's 1.0 precision floor — it only shrinks the * precision denominator, never flips PASS/FAIL) and retains the raw matches * here for the report. Optional: scorer- and test-built firings need not carry * it; an un-collapsed firing carries a single-element array. */ evidence?: FiringEvidence[]; } export interface ScorerInput { firings: RuleFiring[]; /** Maps firingLabelId → TP/FP label. Unlabeled firings ⇒ needsAdjudication. */ groundTruth: Map; positiveControlTargets: Array<{ pr: number; targetRuleId: string; }>; mintedRuleIds: string[]; cullRateThreshold: number; exposureFloors: { activeRulesEvaluated: number; filesTouchedInWindow: number; positiveControlsExercised: number; }; actualExposure: { activeRulesEvaluated: number; filesTouchedInWindow: number; positiveControlsExercised: number; }; } /** * Score a wind-tunnel run. Pure function: no IO, no clock, no randomness. * Implements ADR-110 §4/§5 done-criterion exactly per spec invariants. * * Verdict ordering (highest precedence first) — #2189 ruling: * 1. Any firing labeled FP → FAIL (confirmed FP is a claim; precision = breaching value) * 2. Positive control does not fire its target → FAIL (vacuous pass; precision = null) * 3. Exposure floor below minimum → HONEST-NEGATIVE (masquerade guard; precision = null) * 4. Cull rate exceeds threshold → HONEST-NEGATIVE (cull-laundering guard; precision = null) * 5. Any unlabeled firing → HONEST-NEGATIVE (needs adjudication, not PASS; precision = null) * 6. All labeled TP → PASS (precision = 1.0) * * The FAIL tier (1–2) outranks the masquerade guards (3–4): a guard may only * DEMOTE a would-be PASS, never UPGRADE a FAIL. survivorPrecision (diagnostics) * carries the informative survivor ratio distinct from the certifying precision. */ export declare function scoreWindtunnel(input: ScorerInput): WindtunnelVerdict; //# sourceMappingURL=windtunnel-scorer.d.ts.map