import { CAPACITY_PROBE_SYSTEM_PROMPT, CURATION_DIGEST_SYSTEM_PROMPT, SEARCH_PROBE_SYSTEM_PROMPT, TOOL_CALL_PROBE_SYSTEM_PROMPT } from "../provider-prompt-contracts.ts"; export { CAPACITY_PROBE_SYSTEM_PROMPT, CURATION_DIGEST_SYSTEM_PROMPT as DIGEST_PROBE_SYSTEM_PROMPT, SEARCH_PROBE_SYSTEM_PROMPT, TOOL_CALL_PROBE_SYSTEM_PROMPT, }; /** * Model fitness probe: measures whether a candidate model can actually drive the harness's * subagent contracts — the research lane, the delegated-worker lane, and the routing judge — by * running each real runner against the model and scoring parse/success rates plus judge * discrimination. Provider-free: the completion executor is injected, so this works against any * registered model (local Ollama models included) and against faux providers in tests. */ export interface FitnessCompletion { text: string; costUsd: number; stopReason: string; /** Output tokens generated (for tok/s). Optional: providers that don't report it are skipped. */ outputTokens?: number; /** Pure generation time in ms (e.g. Ollama eval_duration). Falls back to wall-clock if absent. */ evalMs?: number; } export type FitnessComplete = (args: { systemPrompt: string; userPrompt: string; signal?: AbortSignal; }) => Promise; export interface JudgeFitnessPrompt { prompt: string; /** True when the prompt is planning-shaped and must never route cheap. */ planning: boolean; } /** Default judge probe set: three planning-shaped prompts, three trivial lookups. */ export declare const DEFAULT_JUDGE_FITNESS_PROMPTS: readonly JudgeFitnessPrompt[]; export interface CapacityProbeOptions { registeredContextWindow: number; /** Smallest candidate window to try before declaring the capacity unknown. Default 1024. */ minContextWindow?: number; } export interface ModelFitnessOptions { complete: FitnessComplete; /** Trials per lane surface. Default 3. */ trials?: number; /** Wall-clock budget per call in ms. Default 120000. */ maxWallClockMs?: number; judgePrompts?: readonly JudgeFitnessPrompt[]; /** Optional local-model capacity lane: measures the actually served context window. */ capacityProbe?: CapacityProbeOptions; signal?: AbortSignal; /** Injected clock for latency measurement (test seam). Defaults to Date.now. */ now?: () => number; } export interface LaneFitnessScore { succeeded: number; total: number; outcomes: string[]; meanMs: number; /** Mean output tokens/second across the surface's calls; undefined when not reported. */ tokensPerSecond?: number; } export interface JudgeFitnessScore { parsed: number; planningElevated: number; planningTotal: number; trivialCheap: number; trivialTotal: number; total: number; outcomes: string[]; meanMs: number; /** Mean output tokens/second across the judge calls; undefined when not reported. */ tokensPerSecond?: number; } export interface CapacityFitnessScore { registeredContextWindow: number; servedContextWindow: number; outcomes: string[]; meanMs: number; } export interface ModelFitnessReport { trials: number; /** Aggregate output tokens/second across ALL probe calls (the headline speed number). */ tokensPerSecond?: number; research: LaneFitnessScore; worker: LaneFitnessScore; judge: JudgeFitnessScore; /** Heavy-lifter surface: can the model formulate a structured search plan? */ search: LaneFitnessScore; /** Heavy-lifter surface: can the model emit a well-formed tool call against a schema? */ toolCall: LaneFitnessScore; /** Curator surface: can the model digest a context chunk to strict JSON WITHOUT losing key facts? */ digest: LaneFitnessScore; /** Local-model capacity lane: measured context window the server actually serves. */ capacity?: CapacityFitnessScore; totalCostUsd: number; } export declare function runModelFitnessProbe(options: ModelFitnessOptions): Promise; /** * Pure verdict: true when the probe found ZERO successes on every LANE surface it actually graded * AND the judge (if it ran) also failed. A lane/judge with total 0 (i.e. never run) carries no * evidence and is excluded from the lane check — but at least one lane must actually have been * graded for an all-failed verdict at all: `gradedLanes.every(...)` is vacuously true over an * empty array, so a report where only the judge ran (every research/worker/search/toolCall/digest * lane is ungraded) is excluded explicitly rather than misread as "all lanes failed" on zero lane * evidence. An empty/degenerate report (nothing graded at all, lanes AND judge) is likewise never * mistaken for a failed one. This is the gate adoption flows must consult before assigning a role — * see `isProbeAllFailed` callers in interactive-mode.ts and agent-session.ts. */ export declare function isProbeAllFailed(report: ModelFitnessReport): boolean; /** Compact human-readable report for tool output / interactive display. Bounded, no raw dumps. */ export declare function formatModelFitnessReport(model: string, report: ModelFitnessReport): string; //# sourceMappingURL=model-fitness.d.ts.map