import type { Experiment, ExperimentStore } from '../types/evaluate.js'; import type { CompletionRequest } from '../types/messages.js'; import type { EvalMetrics, MetricInputs } from './metrics.js'; /** * The one method `EvalRunner` needs from a client. A `NexusAI` client fits, and so does a test * double. */ export interface EvalClient { /** Runs one completion. */ complete(request: CompletionRequest): Promise; } /** A judge's verdict on one response. */ export interface EvalJudgment { /** Score from 0 to 1. */ score: number; /** Whether the response passed. */ passed: boolean; /** Why the judge decided as it did. */ rationale?: string; /** Labels the judge applied. */ labels?: string[]; /** The judge's raw output. */ raw?: unknown; /** Provider the judge ran on. */ providerUsed?: string; /** Model the judge ran on. */ modelUsed?: string; } /** * Judges a response: `true`/`false`, or a full judgment with a score and rationale. The LLM judge * from `nexus-ai-pro/evals/judge` produces one. */ export type EvalJudge = (response: Response, testCase: EvalCase) => boolean | EvalJudgment | Promise; /** One completion case: a request and how its response is checked. */ export interface EvalCase { /** Names the case in reports. Duplicate names get a numbered suffix in the experiment. */ name: string; /** The request to send. */ request: CompletionRequest; /** Checks the response in code. */ assert?: (response: Response) => boolean | Promise; /** Judges the response, typically with a model. */ judge?: EvalJudge; /** The expected answer, available to the judge and to metrics. */ expected?: string; /** Computes metric inputs from the response, for `calculateEvalMetrics`. */ metrics?: (response: Response) => MetricInputs | Promise; /** Labels for filtering. */ tags?: string[]; } /** The outcome of one case. */ export interface EvalResult { /** The case's name. */ name: string; /** True when the assertion and the judge both passed. */ passed: boolean; /** Time from the request to the last check, in milliseconds. */ durationMs: number; /** The response, when the case did not error. */ response?: Response; /** Metrics computed from the response. */ metrics?: EvalMetrics; /** The judge's verdict. */ judgment?: EvalJudgment; /** * Why the case failed to run: the provider failed, a check threw, or it had neither `assert` nor * `judge`. */ error?: string; /** The case's tags. */ tags?: string[]; } /** The outcome of a run of cases. */ export interface EvalRunResult { /** True when every case passed. */ passed: boolean; /** Cases run. */ total: number; /** Cases that passed. */ passedCount: number; /** Cases that failed. */ failedCount: number; /** Total duration in milliseconds. */ durationMs: number; /** Every case's outcome, in order. */ results: EvalResult[]; /** * The same run as an experiment, so it can be stored, compared with `compareExperiments()`, and * gated in CI like any other evaluation. Absent only for a run with no cases. */ experiment?: Experiment; } /** Options for one run. */ export interface EvalRunOptions { /** Names the experiment. Defaults to `eval-runner` plus a timestamp. */ name?: string; /** Cases run at once. Defaults to 1, which keeps them in order. */ concurrency?: number; /** Stores the experiment, for a later comparison. */ store?: ExperimentStore; /** Application data recorded on the experiment. */ metadata?: Record; } /** * Runs completion cases and checks each response. * * Built on `evaluate()`: cases become a dataset, the client is the target, and the assertion, judge, * and metrics become one evaluator. The result keeps the shape this runner has always returned and * adds the experiment underneath it. */ export declare class EvalRunner { private client; constructor(client: EvalClient); /** * Runs the cases, in order by default, and returns the results with the experiment underneath. */ run(cases: EvalCase[], options?: EvalRunOptions): Promise>; }