import type { DatasetExample, EvaluationScore, Evaluator } from '../types/evaluate.js'; import type { Run, RunQuery, TraceStore } from '../types/tracing.js'; /** One item waiting for a person, with its claims and the answers it has. */ export interface ReviewItem { /** The item's id. */ id: string; /** What is being judged, with enough context for a person to judge it. */ subject: { inputs?: unknown; output?: unknown; runId?: string; exampleId?: string; }; /** The questions a reviewer answers. */ rubric: ReviewQuestion[]; /** * `pending` until claimed, `claimed` while someone works on it, `reviewed` once it has enough * answers. */ status: 'pending' | 'claimed' | 'reviewed'; /** Live claims, one per reviewer. Several at once when the item needs consensus. */ claims: Array<{ reviewer: string; until: string; }>; /** Answers submitted so far. */ answers: ReviewAnswer[]; /** ISO-8601 time the item was queued. */ createdAt: string; /** Application data carried with the item. */ metadata?: Record; } /** One question a reviewer answers. */ export interface ReviewQuestion { /** Key the answer is recorded under, which becomes a score key. */ key: string; /** The question, as the reviewer sees it. */ prompt: string; /** What kind of answer is expected. */ type: 'score' | 'boolean' | 'text' | 'choice'; /** The options, for a `choice` question. */ choices?: string[]; } /** One reviewer's answers to an item. */ export interface ReviewAnswer { /** Who answered. */ reviewer: string; /** The answers, as scores. */ scores: EvaluationScore[]; /** A note from the reviewer. */ comment?: string; /** ISO-8601 time the answers were submitted. */ submittedAt: string; } /** Configuration for an annotation queue. */ export interface AnnotationQueueOptions { /** Questions every item asks. */ rubric: ReviewQuestion[]; /** How long a claim lasts before the item returns to the queue. Defaults to 15 minutes. */ leaseMs?: number; /** Answers needed before an item counts as reviewed. Defaults to 1. */ consensus?: number; /** Replaces the system clock, for tests. */ now?: () => Date; } /** * Work waiting for a person. * * Some judgements only a human can make, and the failure mode of "send it to a human" is losing * track of what was sent, what came back, and what is still waiting. Claims expire, so a reviewer * who closes the tab does not strand an item; consensus lets two reviewers see the same item when * one opinion is not enough. */ export declare class AnnotationQueue { private readonly options; private readonly items; private readonly now; constructor(options: AnnotationQueueOptions); /** Queues something for review and returns the new item. */ enqueue(subject: ReviewItem['subject'], metadata?: Record): ReviewItem; /** * Claims the oldest item this reviewer can work on. * * An item needing two opinions may be claimed by two reviewers at once, but never twice by the * same one, and a claim that expires frees the slot again. */ claim(reviewer: string): ReviewItem | undefined; /** Records a reviewer's answers. The item is reviewed once it has `consensus` answers. */ submit(itemId: string, answer: Omit & { submittedAt?: string; }): ReviewItem; /** Items in the queue, optionally with one status. */ list(status?: ReviewItem['status']): ReviewItem[]; /** Averaged scores per key across reviewers, which is what consensus is for. */ consensusScores(itemId: string): EvaluationScore[]; /** Reviewed items as dataset examples, so human judgement becomes a regression test. */ toExamples(): Array & { id: string; }>; } /** Options for `evaluateOnline()`. */ export interface OnlineEvaluationOptions { /** Where the runs to score are. */ store: TraceStore; /** Evaluators applied to each sampled run. */ evaluators: Evaluator[]; /** Which runs to score. Defaults to finished model runs. */ query?: RunQuery; /** Share of matching runs scored, 0 to 1. Defaults to 1. */ sampleRate?: number; /** Sends an uncertain result to people instead of scoring it automatically. */ reviewQueue?: AnnotationQueue; /** Decides whether a run's scores are uncertain enough to send it to people. */ reviewWhen?: (scores: EvaluationScore[], run: Run) => boolean; /** Writes scores back as trace feedback. Defaults to true. */ recordFeedback?: boolean; /** Replaces the system clock, for feedback timestamps. */ now?: () => Date; } /** What an online evaluation pass did. */ export interface OnlineEvaluationReport { /** Runs matching the query. */ scanned: number; /** Runs scored after sampling. */ evaluated: number; /** Runs sent to the review queue. */ queuedForReview: number; /** Every score given. */ scores: EvaluationScore[]; } /** * Scores production runs after the fact. * * A dataset tells you whether a change works on the cases you thought of; online evaluation tells you * how it is doing on the ones you did not. Scores are written back as feedback on the run, so an * alert rule can watch them and a bad sample can become a dataset example. */ export declare function evaluateOnline(options: OnlineEvaluationOptions): Promise;