import type { RagChunk } from '../hallucination/rag.js'; import { type EmbeddingProvider } from '../hallucination/retrieval.js'; /** How close an answer is to the expected one. */ export interface QualityMetrics { /** * 1 when the answer equals the expected text after case, punctuation, and whitespace are * normalized, otherwise 0. */ exactMatch?: number; /** Token-overlap F1 against the expected text, from 0 to 1. */ f1?: number; /** Cosine similarity of the answer and the expected text under the configured embedder. */ semanticSimilarity?: number; /** 1 when any of the first k candidates passed, otherwise 0. */ passAtK?: number; /** * Perplexity from the answer's token log-probabilities. Lower means the model was more confident. */ perplexity?: number; } /** How fast and expensive an answer was. */ export interface OperationalMetrics { /** Output tokens per second of total latency. */ tokensPerSecond?: number; /** Time to the first streamed token, in milliseconds. */ timeToFirstTokenMs?: number; /** Total latency, in milliseconds. */ latencyMs?: number; /** Cost as estimated by the caller. */ estimatedCost?: number; /** Input tokens. */ inputTokens?: number; /** Output tokens. */ outputTokens?: number; } /** How well a retrieval-augmented answer used its sources. */ export interface RagMetrics { /** * Share of the answer's factual statements supported by the retrieved context, by lexical * entailment. */ faithfulness?: number; /** Average precision of the retrieved chunks: relevant chunks ranked early score higher. */ contextualPrecision?: number; /** Share of the relevant chunks that were retrieved at all. */ contextualRecall?: number; /** Similarity of the answer to the query, as a proxy for whether it addressed the question. */ answerRelevancy?: number; } /** * Heuristic safety signals. Word lists and patterns, not trained classifiers: useful as a tripwire, * not as a verdict. */ export interface SafetyMetrics { /** Share of the answer's facts not supported by the context: one minus `faithfulness`. */ hallucinationRate?: number; /** Share of words that appear on a short list of abusive terms. */ toxicityScore?: number; /** 1 when the answer generalizes about a group by one of a few fixed patterns, otherwise 0. */ biasScore?: number; /** Average of two checks: no forbidden term appears, and the share of required terms that do. */ policyAdherence?: number; /** 1 when a safe prompt was refused, otherwise 0. */ refusalRate?: number; } /** * Every metric computed for one answer, grouped by kind. A group's fields are absent when their * inputs were not supplied. */ export interface EvalMetrics { /** Closeness to the expected answer. */ quality?: QualityMetrics; /** Speed and cost. */ operational?: OperationalMetrics; /** Use of retrieved context. */ rag?: RagMetrics; /** Heuristic safety signals. */ safety?: SafetyMetrics; } /** * What `calculateEvalMetrics` needs. Each metric is computed only when the inputs it depends on are * present. */ export interface MetricInputs { /** The answer being scored. */ actual: string; /** The expected answer, for quality metrics. */ expected?: string; /** The question asked, for answer relevancy. */ query?: string; /** Retrieved passages, for faithfulness and hallucination rate. */ contexts?: string[]; /** Retrieved chunks in rank order, for contextual precision and recall. */ retrievedChunks?: RagChunk[]; /** Ids of the chunks that should have been retrieved. */ relevantChunkIds?: string[]; /** Candidate answers, for pass@k. */ candidates?: string[]; /** Whether each candidate passed, for pass@k. */ passedCandidates?: boolean[]; /** Token log-probabilities of the answer, for perplexity. */ tokenLogProbs?: number[]; /** Total latency, in milliseconds. */ latencyMs?: number; /** Time to the first streamed token, in milliseconds. */ timeToFirstTokenMs?: number; /** Input tokens. */ inputTokens?: number; /** Output tokens. */ outputTokens?: number; /** Cost, as estimated by the caller. */ estimatedCost?: number; /** Terms the answer must avoid or include, and whether the prompt was safe to answer. */ policy?: { forbiddenTerms?: string[]; requiredTerms?: string[]; safePrompt?: boolean; }; /** * Embeds text for similarity metrics. Defaults to a local hash embedding, which suits tests but * not real semantic comparison. */ embed?: EmbeddingProvider; } /** Computes every metric the inputs allow. */ export declare function calculateEvalMetrics(input: MetricInputs): Promise; /** 1 when two texts are equal after normalizing case, punctuation, and whitespace, otherwise 0. */ export declare function exactMatch(actual: string, expected: string): number; /** Token-overlap F1 between two texts, from 0 to 1. */ export declare function f1Score(actual: string, expected: string): number; /** Cosine similarity of two texts under an embedder. Defaults to a local hash embedding. */ export declare function semanticSimilarity(a: string, b: string, embed?: EmbeddingProvider): Promise; /** 1 when any of the first `k` candidates passed, otherwise 0. */ export declare function passAtK(passedCandidates: boolean[], k?: number): number; /** Perplexity from token log-probabilities. */ export declare function perplexity(tokenLogProbs: number[]): number; /** Output tokens per second, or `undefined` when either input is missing or zero. */ export declare function tokensPerSecond(outputTokens?: number, latencyMs?: number): number | undefined; /** Share of the answer's factual statements that the contexts support lexically, from 0 to 1. */ export declare function faithfulness(answer: string, contexts: string[]): number; /** Average precision of a ranked retrieval against the relevant chunk ids. */ export declare function contextualPrecision(retrievedChunks: RagChunk[], relevantChunkIds: string[]): number; /** Share of the relevant chunk ids that the retrieval returned. */ export declare function contextualRecall(retrievedChunks: RagChunk[], relevantChunkIds: string[]): number; /** * Share of words in the text that appear on a short list of abusive terms. A tripwire, not a * classifier. */ export declare function toxicityScore(text: string): number; /** * 1 when the text generalizes about a group by one of a few fixed patterns, otherwise 0. A * tripwire, not a classifier. */ export declare function biasScore(text: string): number; /** * Average of two checks: no forbidden term appears (1 or 0), and the share of required terms * present. */ export declare function policyAdherence(text: string, policy: { forbiddenTerms?: string[]; requiredTerms?: string[]; }): number; /** 1 when a safe prompt was answered with a refusal phrase, otherwise 0. */ export declare function refusalRate(text: string, safePrompt: boolean): number;