import type { PingRecord } from "./metrics.js"; import { type QuotaObservation } from "../quota-observation.js"; export declare const DEFAULT_PROBE_TTL_MS: number; export declare const BROKEN_PROBE_BACKOFF_BASE_MS = 60000; /** * Bumped to 3 when probe quota changed from an ambiguous scalar percentage to typed observations. * A version mismatch marks a model due for probing (see `getModelsDueForProbe`), so old entries * refill naturally instead of migrating a scalar whose axis and period were already lost. */ export declare const CURRENT_PROBE_VERSION = 3; /** * How many recent probes are kept per model. * * p95 and jitter need a DISTRIBUTION, so a rolling window is unavoidable; `totals` below carries * the long-run figures the window drops. 25 keeps a full roster's cache in the low hundreds of KB. */ export declare const MAX_SAMPLES = 25; /** * Long-run counters that OUTLIVE the rolling window. * * The window answers "how does this model behave lately"; these answer "how has it behaved since * we first saw it". Keeping both is what stops a single bad afternoon from erasing a model's * record — and what makes uptime mean something after a restart. */ export interface ProbeTotals { probes: number; ok: number; sumMs: number; firstProbedAt: number; } export interface ProbeEntry { modelId: string; status: "ok" | "broken"; lastProbedAt: number; probeVersion: number; ms: number; code: string; quotaObservations: QuotaObservation[]; /** Rolling window, oldest first. Absent on v1 entries. */ samples?: PingRecord[]; /** Cumulative counters. Absent on v1 entries. */ totals?: ProbeTotals; } export interface ProbeCacheData { version: number; providers: Record; }>; } export declare function getProbeCachePath(): string; export declare function loadProbeCache(opts?: { path?: string; reload?: boolean; }): ProbeCacheData; export declare function flushProbeCache(opts?: { path?: string; cache?: ProbeCacheData; }): void; export declare function getModelsDueForProbe(providerKey: string, modelIds: string[], opts?: { ttlMs?: number; now?: number; probeVersion?: number; path?: string; lastSuccessfulCallAt?: (modelId: string) => number | null; }): string[]; /** * Append a REAL SERVED-REQUEST latency sample to a deployment's rolling window. * * Owner decision 2026-08-30: `routing.latency` reads the PROBE dataset — it persists across * restarts and it is what `llm-relay candidates` displays — and that dataset is EXPANDED to carry * request latency too, so a per-token rate can be derived from actual traffic. * * ⚠ **It deliberately touches NOTHING that schedules probing.** `status`, `lastProbedAt`, * `probeVersion`, the scalar `ms`/`code` and `quotaObservations` are all left exactly as they were, * because `getModelsDueForProbe` reads `lastProbedAt`: refreshing it here would mean a deployment * carrying real traffic silently STOPPED being probed, so its independent health signal would go * stale precisely for the models that matter most. * * ⚠ **`totals` are untouched too.** They count PROBES (`probes`/`ok`/`sumMs`) and feed uptime; * folding request samples in would change what uptime has always meant. * * ⚠ **An unknown deployment is SKIPPED, not created.** Minting an entry here would mean inventing * probe-scheduling fields for a model nobody has probed. The ping loop creates the entry on its * first real probe and request samples land from then on — a short warm-up, in exchange for never * writing a fabricated probe record. */ export declare function recordRequestSample(providerKey: string, modelId: string, sample: { ms: number; tokens?: number; }, opts?: { now?: number; path?: string; }): void; export declare function recordProbeResult(providerKey: string, modelId: string, result: { code: string; ms: number; quotaObservations: QuotaObservation[]; }, opts?: { now?: number; probeVersion?: number; path?: string; }): ProbeEntry; /** * Every persisted sample for a model, oldest first — the history a restarted process needs to * avoid starting from zero. * * `PingLoop` kept its history in an in-memory `Map` and read only that, while `recordProbeResult` * wrote to disk and nothing ever read it back. So every restart reset every model to `Pending` * with `p95: -1`, and a proxy that restarts (a laptop that sleeps, an upgrade, a crash) never * accumulated anything at all. This is the read side that was missing. */ export declare function loadPersistedSamples(providerKey: string, modelId: string, opts?: { path?: string; }): PingRecord[]; /** Long-run counters for a model, or null when it has never been probed. */ export declare function loadTotals(providerKey: string, modelId: string, opts?: { path?: string; }): ProbeTotals | null; /** * How many REAL SERVED-REQUEST samples a deployment has in the probe dataset. * * The count the probation band (`routing.probation`) reads: a free deployment with fewer than * `minSamples` of these leads its pool so the relay gathers data on it. PROBE samples do NOT * count — a probe asks for one token and proves availability, not served traffic; only samples * `recordRequestSample` wrote carry `source: "request"` (absence means `"probe"`, the * `PingRecord` contract). Unknown deployment, missing entry or corrupt samples read as 0 — * "unmeasured", which is exactly what the band is for. */ export declare function countRequestSamples(providerKey: string, modelId: string, opts?: { path?: string; }): number; /** * Typed quota observations persisted by the default synthetic probe, or none when this entry is * not current. In particular, a v2 scalar `quotaPercent` is intentionally never reconstructed. */ export declare function loadPersistedQuotaObservations(providerKey: string, modelId: string, opts?: { path?: string; }): QuotaObservation[]; /** Every (provider, model) the cache holds samples for — what a restarting PingLoop rehydrates. */ export declare function persistedModels(opts?: { path?: string; }): Array<{ provider: string; model: string; }>;