import type { Dataset, DatasetExample, Evaluator, ExampleResult, Experiment, ExperimentStore, MetricSummary, SummaryEvaluator } from '../types/evaluate.js'; /** Turns one example into an output. A completion, an agent run, a graph, or plain code. */ export type EvaluationTarget = (inputs: I, context: { example: DatasetExample; run: number; signal?: AbortSignal; }) => Promise | unknown; /** Options for `evaluate()`. */ export interface EvaluateOptions { /** Names the experiment. Defaults to the dataset name plus a timestamp. */ name?: string; /** Examples evaluated at once. Defaults to 4; `1` runs them in order. */ concurrency?: number; /** * Times each example is run. Above 1 for a target that is not deterministic; a function gives * each example its own count, for a dataset where some cases are noisier than others. */ repetitions?: number | ((example: DatasetExample) => number); /** Evaluators over the whole experiment, such as a pass rate. */ summary?: SummaryEvaluator[]; /** Where the experiment is saved. */ store?: ExperimentStore; /** Application data stored with the experiment. */ metadata?: Record; /** Stops an example that hangs. */ timeoutMs?: number; /** * Cancels the evaluation. No further examples start, the examples in flight receive the signal, * and `evaluate()` rejects with the abort reason instead of storing a partial experiment that * could later be mistaken for a complete baseline. */ signal?: AbortSignal; /** * Reads the cost of one output. Defaults to `meta.cost.amount` on a completion, a numeric * `meta.cost` on an image result, or a numeric `cost` field on anything else. */ cost?: (output: unknown, example: DatasetExample) => number | undefined; /** Called after each example finishes, for progress reporting. */ onResult?: (result: ExampleResult) => void; /** Replaces the system clock, for tests. */ now?: () => Date; } /** * Runs a target over a dataset and scores it. * * One entry point for every kind of target, because the question is always the same: for each * example, produce an output, score it, and keep the result somewhere a later run can be compared * against. Everything that differs — how an output is produced, what counts as good — is a function * the caller supplies. */ export declare function evaluate(target: EvaluationTarget, dataset: Dataset, evaluators: Array>, options?: EvaluateOptions): Promise; /** Mean, spread, and an interval per metric, so one lucky run is not mistaken for an improvement. */ export declare function summarize(results: ExampleResult[]): MetricSummary[]; /** Count, mean, sample standard deviation, minimum, maximum, and 95% interval of the mean for a set of scores. */ export declare function stats(values: number[]): Omit;