import type { ExtractionHarness } from "../extract/harness.js"; import { type Output } from "../shared.js"; import { type ResolveEngineOptions } from "./engine.js"; /** * The judgment channel: read the deterministic catalog, ask a model to grade it, * ask a SECOND model to tear the first one's answer apart, and write only what * survives into `.vendo/judgments.json`. * * The shape exists because of one measured failure mode. A single model pass * that is allowed to grade capability will confidently justify a grade the code * does not support — in either direction. An over-tight grade silently breaks a * working product; a loose one hands out capability. So: * * - the JUDGE proposes, and every proposal costs a VERBATIM quote from the * handler. No quote, no proposal — rejected at parse, counted out loud; * - the SKEPTIC is a second, independent run (fresh conversation, same engine) * whose only job is to check each field against the real source, including * whether the quote exists at all. It rejects hardenings as readily as * loosenings; * - anything the skeptic never looked at gets ONE re-ask and is then REJECTED. * Unexamined must never mean applied, and the narrative says how many; * - what survives is routed by the deterministic direction rule in * `@vendoai/vendo/actions`: hardenings and prose apply themselves, loosenings wait * for a human. * * Every model-originated string and every evidence snippet is untrusted repo * content and is sanitized before it reaches a terminal. */ /** One judge call reads this many tools. Big enough that a normal catalog is one * or two calls, small enough that the model actually opens each handler instead * of skimming a wall of names. */ export declare const JUDGE_BATCH_LIMIT = 20; export interface JudgmentPassOptions { root: string; /** The `.vendo` directory (sync's `out`). */ out: string; /** full: judge the whole catalog. incremental: only what moved. */ mode: "full" | "incremental"; /** review: ask about loosenings now. queue: park them as `pending`. */ loosenings: "review" | "queue"; env: Record; output: Output; /** Adapter rule: an explicitly passed harness always wins over the ladder. */ harness?: ExtractionHarness; /** `--engine` family pin. An unavailable pin never falls back. */ engine?: string; confirm?: (question: string, defaultYes: boolean) => Promise; /** Ladder seams. */ harnesses?: ExtractionHarness[]; resolveCredential?: ResolveEngineOptions["resolveCredential"]; appName?: string; onProgress?: (line: string) => void; } export interface JudgmentPassCounts { /** Tools whose judgment entry this pass wrote or updated. */ judged: number; /** (tool, field) hardenings and prose edits applied. */ hardened: number; /** Loosenings left waiting as `pending`. */ queued: number; /** Loosenings a human accepted this run. */ approved: number; rejectedBySkeptic: number; unexaminedRejected: number; /** Proposals thrown out at parse for carrying no evidence. */ evidenceless: number; /** Advisory and prose strings truncated or dropped so they could not discard a * judgment. Never fatal — surfaced so a clamped lead is visible. */ advisoriesClamped: number; /** Risk grades dropped for contradicting their own stated reason. */ inconsistentRisk: number; /** Blind schema slots the judge filled and the skeptic upheld (both slots). */ schemasInferred: number; /** Schema proposals the skeptic vetoed, plus the ones `patchToolSchemas` * refused because the slot was already occupied or the tool was rebound. */ schemasRejected: number; } export type JudgmentPassResult = ({ status: "judged"; } & JudgmentPassCounts) | { status: "structural-only"; unjudged: number; } | { status: "up-to-date"; } | { status: "skipped"; }; export declare function runJudgmentPass(options: JudgmentPassOptions): Promise;