import type { AssetDescriptor, AssetInput, ImageEditRequest, ImageGenerateRequest, ImageOperation, ImageResult, MediaSafetyFinding } from '../types/images.js'; import type { Experiment, ExperimentStore } from '../types/evaluate.js'; import type { RgbaImage } from './codec.js'; /** * Evaluation for generated media. * * A string golden file cannot tell whether an image matches its prompt, kept the parts of a photo an * edit was told to keep, rendered the text it was asked for, or was blocked when it should have been. * Generation is also stochastic, so one run proves little. This runner scores each of those, * repeats every case, reports distributions rather than single numbers, and routes the uncertain * middle to people. */ export interface MediaEvalExpectations { /** Text the image should contain, scored by the configured OCR function. */ text?: string; /** Minimum OCR accuracy, 0–1. Defaults to 0.8 when `text` is set. */ minTextAccuracy?: number; /** Minimum prompt alignment, 0–1, scored by the configured alignment function. */ minAlignment?: number; /** For an edit: an image the result should stay perceptually close to, outside the edited area. */ preserve?: { asset: AssetInput; minSimilarity: number; }; /** Whether this case should be blocked. Drives the false-positive and false-negative counts. */ shouldBlock?: boolean; /** Fails a run slower than this, in milliseconds. */ maxLatencyMs?: number; /** Fails a run that cost more than this. */ maxCost?: number; } /** One media evaluation case: a request, how often to run it, and what counts as a pass. */ export interface MediaEvalCase { /** Identifies the case in reports and comparisons. */ id: string; /** Whether the request generates or edits. */ operation: ImageOperation; /** The image request to run. */ request: ImageGenerateRequest | ImageEditRequest; /** Runs per case. Stochastic output needs several; defaults to the runner's `runs`. */ runs?: number; /** What the result must satisfy. Without it, a run passes when it produces an image. */ expect?: MediaEvalExpectations; /** Labels for filtering and grouping reports. */ tags?: string[]; } /** Produces one result for one run of a case — usually a thin call into `ImageManager`. */ export type MediaEvalTarget = (evalCase: MediaEvalCase, run: number) => Promise; /** * Scoring functions a case's expectations need. Alignment and OCR are injected, so any vision model * or OCR engine works. */ export interface MediaEvalScorers { /** Returns prompt alignment from 0 to 1, typically a vision model acting as judge. */ alignment?: (asset: AssetDescriptor, prompt: string) => Promise | number; /** Returns the text found in an image. */ ocr?: (asset: AssetDescriptor) => Promise | string; /** Decodes pixels for perceptual similarity. Defaults to the bundled PNG decoder. */ decode?: (asset: AssetInput) => Promise | RgbaImage; } /** * A run sent to a person, because its score fell in an uncertain band or a safety policy asked for * review. */ export interface ReviewItem { /** The case that produced it. */ caseId: string; /** Which run of the case. */ run: number; /** Why it needs a person. */ reason: string; /** The run's scores. */ scores: Record; /** The image to look at. */ asset?: AssetDescriptor; /** Safety findings on the run. */ findings?: readonly MediaSafetyFinding[]; } /** Where uncertain results go for a person to judge. */ export interface ReviewQueue { /** Adds an item for review. */ enqueue(item: ReviewItem): Promise | void; } /** A review queue that keeps items in memory, for tests and small setups. */ export declare class MemoryReviewQueue implements ReviewQueue { /** Items queued so far, in order. */ readonly items: ReviewItem[]; /** Adds an item for review. */ enqueue(item: ReviewItem): void; } /** * Configuration for `MediaEvalRunner`: scorers, where uncertain results go, and how often to run * each case. */ export interface MediaEvalOptions { /** Scoring functions for alignment, OCR, and decoding. */ scorers?: MediaEvalScorers; /** Where uncertain results go for a person to judge. */ reviewQueue?: ReviewQueue; /** * Scores inside this band are neither a clear pass nor a clear fail, so they go to review instead * of being decided automatically. Keyed by metric: `alignment`, `textAccuracy`, `similarity`. */ reviewBands?: Partial>; /** Default runs per case. Defaults to 1. */ runs?: number; } /** Distribution of one metric across runs. */ export interface MetricStats { /** Values measured. */ n: number; /** Their mean. */ mean: number; /** Their sample standard deviation. */ stddev: number; /** The smallest value. */ min: number; /** The largest value. */ max: number; /** Normal-approximation 95% interval for the mean. Wide intervals mean more runs are needed. */ ci95: [number, number]; } /** The outcome of one run of a case. */ export interface MediaEvalRunReport { /** Which run, starting at 0. */ run: number; /** True when every expectation held. */ passed: boolean; /** True when a safety policy blocked the run. */ blocked: boolean; /** Each expectation that failed, as a readable reason. */ failures: string[]; /** Every metric measured: `latencyMs`, `cost`, `alignment`, `textAccuracy`, `similarity`. */ scores: Record; /** Why the run failed to produce a result, when it did. */ error?: string; } /** A case's runs and how consistently it passed. */ export interface MediaEvalCaseReport { /** The case's id. */ id: string; /** The case's tags. */ tags?: string[]; /** Every run. */ runs: MediaEvalRunReport[]; /** Share of runs that passed. */ passRate: number; /** True only when every run passed. */ passed: boolean; /** Distribution of each metric across the runs. */ stats: Record; } /** A media evaluation: per-case results, distributions, safety accuracy, and operational health. */ export interface MediaEvalReport { /** Per-case results, in order. */ cases: MediaEvalCaseReport[]; /** Share of all runs that passed. */ passRate: number; /** Distribution of each metric across every run. */ metrics: Record; /** * How well blocking matched expectations: true and false positives and negatives, and their * rates. */ safety: { truePositives: number; falsePositives: number; trueNegatives: number; falseNegatives: number; /** Share of runs expected to pass that were blocked anyway. */ falsePositiveRate: number; /** Share of runs expected to be blocked that got through. */ falseNegativeRate: number; }; /** Runs, errors, error rate, and failovers across the evaluation. */ operational: { runs: number; errors: number; errorRate: number; /** Runs whose result names more than one provider on its route. */ failovers: number; }; /** Runs sent to review. */ reviewQueued: number; /** The same evaluation as an experiment, for storage and `compareExperiments()`. */ experiment?: Experiment; } /** Options for one evaluation run. */ export interface MediaEvalRunOptions { /** Names the experiment. Defaults to `media-eval` plus a timestamp. */ name?: string; /** Stores the experiment, for a later comparison. */ store?: ExperimentStore; /** Application data recorded on the experiment. */ metadata?: Record; /** Stops the evaluation. No further runs start, and nothing is stored. */ signal?: AbortSignal; } /** * Runs media cases repeatedly and reports distributions, safety accuracy, and operations. * * Built on `evaluate()`: each case is an example, `runs` is its repetition count, and the report is * assembled from the experiment, which is returned with it so a media evaluation can be stored and * compared like any other. */ export declare class MediaEvalRunner { private readonly target; private readonly options; constructor(target: MediaEvalTarget, options?: MediaEvalOptions); /** * Runs every case its configured number of times and reports the results. Rejects before running * anything when a case asks for fewer than one run, and stops when a case needs a scorer that is * not configured. */ evaluate(cases: readonly MediaEvalCase[], runOptions?: MediaEvalRunOptions): Promise; private runOnce; private inBand; private requireScorer; } /** * Perceptual similarity from 0 to 1, using a 64-bit difference hash. * * Robust to re-encoding, mild resizing, and compression, which is the point: an edit that kept the * rest of a photo should still score high after the provider re-encoded it, while a pixel diff would * report it as entirely changed. */ export declare function perceptualSimilarity(a: RgbaImage, b: RgbaImage): number; /** OCR accuracy from 0 to 1: one minus normalised edit distance, after case and whitespace folding. */ export declare function textAccuracy(expected: string, actual: string): number; /** Mean, spread, and a 95% interval, so a single lucky run is not mistaken for a capability. */ export declare function stats(values: readonly number[]): MetricStats;