import type { EvaluationContext, Evaluator, SummaryEvaluator } from '../types/evaluate.js'; /** * Evaluators that need no model call. * * Cheap, deterministic checks first: most regressions are caught by "did it still contain the order * number" rather than by a judge. An LLM judge is available through `nexus-ai-pro/evals/judge`, and * plugs in here as an ordinary evaluator. */ /** Exact equality against the example's expected output. */ export declare function exactMatch(options?: { key?: string; normalize?: (value: unknown) => unknown; }): Evaluator; /** Whether the output contains every required phrase, case-insensitively by default. */ export declare function contains(phrases: readonly string[], options?: { key?: string; caseSensitive?: boolean; }): Evaluator; /** Fails an output that matches any forbidden pattern: a leaked key, a refusal, a placeholder. */ export declare function mustNotMatch(patterns: readonly RegExp[], options?: { key?: string; }): Evaluator; /** The example either produced an output or it did not. Worth measuring on its own. */ export declare function completed(options?: { key?: string; }): Evaluator; /** Latency as a pass or fail, so a quality gate can hold a budget as well as a score. */ export declare function underLatency(maxMs: number, options?: { key?: string; }): Evaluator; /** * Cost as a pass or fail, per example. * * Reads the cost `evaluate()` recorded for the output, so an example whose output carries no cost * scores 0 and passes: an unpriced call is not evidence of overspending. */ export declare function underCost(maxCost: number, options?: { key?: string; }): Evaluator; /** Cosine similarity against the expected answer, through any embedder. */ export declare function embeddingSimilarity(options: { embed: (texts: string[]) => Promise; key?: string; threshold?: number; }): Evaluator; /** Options for `trajectory()`. */ export interface TrajectoryOptions { /** Tool or node names expected, in order. */ expected?: readonly string[]; /** Reads the path actually taken. Defaults to the agent's tool calls. */ path?: (context: EvaluationContext) => string[]; /** `exact` requires the same sequence; `subset` only requires each expected step to appear. */ mode?: 'exact' | 'subset'; /** Score key. Defaults to `trajectory`. */ key?: string; } /** * Scores how an answer was reached, not just what it said. * * An agent that reaches the right answer by calling the refund tool three times is not working. The * trajectory is the part a final-answer check cannot see. */ export declare function trajectory(options?: TrajectoryOptions): Evaluator; /** * Compares two outputs for the same example and says which is better. * * Absolute scores drift; a side-by-side judgement is what people are actually good at, and what * distinguishes two candidate versions when both look acceptable alone. */ export declare function pairwise(options: { compare: (context: { example: EvaluationContext['example']; a: unknown; b: unknown; }) => Promise<-1 | 0 | 1> | (-1 | 0 | 1); baseline: (exampleId: string) => unknown; key?: string; }): Evaluator; /** Pass rate over every example, as a summary score for the experiment. */ export declare function passRate(key?: string): SummaryEvaluator; /** Total cost of the experiment, so a quality gain that tripled the bill is visible. */ export declare function totalCost(key?: string): SummaryEvaluator;