export interface PingRecord { ms: number; code: string; timestamp: number; /** * Output tokens this sample generated, when the provider reported them. * * ⚠ ABSENT means UNKNOWN, never zero — the standing provenance rule. Present only on * `source: "request"` samples: a probe asks for one token, so its rate would be pure fixed * overhead and is deliberately not comparable with a real generation. */ tokens?: number; /** * Where the sample came from. ⚠ ABSENT means `"probe"` — that is every sample written before * 2026-08-30 and every sample the ping loop writes, so absence must never be read as "unknown * provenance". Only the request path writes `"request"`. */ source?: "probe" | "request"; } export type Verdict = "Perfect" | "Normal" | "Slow" | "Spiky" | "Overloaded" | "Unstable" | "Not Active" | "Pending"; /** * Codes whose round-trip time is a real LATENCY sample. A 401 came back from the provider, * so it timed the network path and belongs in avg/p95/jitter/spike. * * ⚠ Latency only — this is NOT an availability set. Availability is decided by `getUptime()` * below, which counts `"200"` and nothing else, so a target answering only 401s reports 0% * uptime and `getStabilityScore` MULTIPLIES its composite to 0. That claim used to read "capped * accordingly" while the arithmetic merely subtracted a 20% term, which capped an all-401 target * at ~80 — above every all-success target with mediocre latency. The cap is real now because * availability scales the score rather than contributing a share of it. Two other sites used to * treat 401 as equivalent to 200 for availability (`probe-cache.ts`, `cadence.ts`) and a provider * with a revoked key read as healthy; both now follow the 200-only rule. */ export declare const MEASURABLE_CODES: ReadonlySet; /** Calculate average latency from measurable pings (HTTP 200/401). Returns Infinity if none. */ export declare function getAvg(pings: PingRecord[]): number; /** Calculate 95th percentile latency from measurable pings. Returns Infinity if none. */ export declare function getP95(pings: PingRecord[]): number; /** * 90th percentile latency from measurable pings. Returns Infinity if none. * * Added for the hedge trigger, which asks "is this attempt slower than this deployment normally * is?" — a question p90 answers better than p95, because p95 over a short window is the worst * sample and would almost never be exceeded. */ export declare function getP90(pings: PingRecord[]): number; /** * 95th-percentile latency PER OUTPUT TOKEN, in ms — the only figure that compares a real * generation with anything else, because absolute latency scales with how much was generated. * * ⚠ REQUEST samples only, and only those carrying a reported token count. A probe sends * `max_tokens: 1`, so its ms/token is almost entirely fixed overhead (connect, queue, prompt * processing) and would read as catastrophically slow beside a 500-token answer that amortises * the same overhead. Mixing the two would not be a noisy measurement; it would be a wrong one. * * ⚠ Returns `Infinity` when nothing qualifies — "unmeasured", never "infinitely slow", exactly as * `getP95` does. Callers must test `Number.isFinite` rather than compare against a ceiling. * * Measured on this machine 2026-08-30 over 68 real requests, and this is what calibrates the * default ceiling: p50 40.4, p75 70.5, p90 292.0, p95 967.1 ms/token. Per deployment the * separation is clean — a healthy `nemotron-3-ultra` ran a median 36.3 while `gemini-3.6-flash` * ran 687.8. */ export declare function getP95MsPerToken(pings: PingRecord[]): number; /** * 90th-percentile ms per output token — the hedge trigger's signal. * * Same filter and same convention as the p95 above, so the two can never describe different * populations. p90 rather than p95 because the question is "is THIS attempt slower than this * deployment normally is?", and a p95 taken over a short window is the worst sample ever seen, * which almost nothing exceeds. */ export declare function getP90MsPerToken(pings: PingRecord[]): number; /** * How many samples actually back `getP95MsPerToken` and `getP90MsPerToken`. * * ⚠ It used to REPEAT their filter, under a comment saying the copies "must stay identical" — a * requirement nothing enforced. All three now read `msPerTokenRates`, which is what enforces it. * The reason is unchanged: a count taken over a WIDER set than the statistic it describes is how a * sample floor comes to admit a figure resting on one measurement. */ export declare function countMsPerTokenSamples(pings: PingRecord[]): number; /** Calculate latency standard deviation (jitter) in ms from measurable pings. */ export declare function getJitter(pings: PingRecord[]): number; /** Calculate spike rate: fraction (0–1) of measurable pings with latency > 3000ms. */ export declare function getSpikeRate(pings: PingRecord[]): number; /** Calculate uptime percentage (0–100) of HTTP 200 pings over total pings. */ export declare function getUptime(pings: PingRecord[]): number; /** * Composite Stability Score (0–100). Returns -1 only when the deployment has NEVER been probed. * * Shape: **latency quality, SCALED by availability** — not latency quality plus a fifth of * availability. The distinction is the whole point, because `MEASURABLE_CODES` is a latency set: * 403/404/429/5xx leave p95, jitter and spike entirely, so under the old additive form * (`0.3*p95 + 0.3*jitter + 0.2*spike + 0.2*uptime`) a failing deployment kept a clean latency * profile and paid only 20%. Measured on live probe data: 1 success in 12 scored **81** while * 3 of 3 scored **27**, and 27 zero-success deployments scored above 50 — on a machine whose * pools are all `{include: "free"}`, so this score IS the pool order. The comment above * `MEASURABLE_CODES` claimed such a target was "capped accordingly"; only a multiplier delivers * that cap. * * The latency weights are renormalized to sum to 1 across the three latency terms, so a * deployment at 100% uptime scores exactly its latency quality and nothing is silently rescaled. * * ⚠ 401 deliberately STAYS in the latency terms: the response came back from the provider, so it * really did time the network path. It contributes nothing to uptime, so the multiplier is what * stops a revoked key from reading as a fast healthy target — no need to discard real timing data. * * ⚠ `-1` means "no samples", not "no MEASURABLE samples". Twelve consecutive 402s is evidence that * this deployment fails, not absence of evidence: returning -1 there made every consumer read * "unmeasured" and the ordering substitute a neutral 50, which put a fully-exhausted deployment * above one that answered every probe. */ export declare function getStabilityScore(pings: PingRecord[]): number; /** * How many of the MOST RECENT probes failed, consecutively. * * The unit of "is this thing broken" — one failure in a row is weather, five in a row is a * pattern. Anything reading the single last sample cannot tell those apart. */ export declare function trailingFailures(pings: PingRecord[]): number; /** * Consecutive recent failures before a target is called down rather than merely erratic. * * A VPN the provider blocks, a DNS hiccup, a rate-limit window, a laptop resuming from sleep — * all produce isolated failures against a model that is completely fine. Requiring a RUN of them * is what stops one such blip from disqualifying a model that has answered hundreds of times. */ export declare const DOWN_AFTER_CONSECUTIVE_FAILURES = 3; /** * Is this target persistently down, as opposed to having just had a bad moment? * * Three ways to be down, and only the last one is forgiving: * 1. the latest probe was a credential refusal — deterministic, see above; * 2. it has never once answered 200 — there is no good record to protect; * 3. a RUN of recent failures AND a poor overall record. * * (3) is what stops a blip disqualifying a model: one with 95% uptime that just hit three * rate-limited probes is rate-limited, not dead. The caller can still reach it, and the circuit * breaker will step over it independently if it really is failing. */ export declare function isPersistentlyDown(pings: PingRecord[]): boolean; /** * Determine human-readable health verdict for a model based on average latency and tail latency. * * ⚠ Down-ness is derived from the ACCUMULATED history, never from the caller's reading of the * most recent ping. `cadence.ts` used to pass `isDown: lastPing.code !== "200"`, so a single * transient 503 — a VPN block, a resumed laptop, one rate-limited probe — overrode fifty good * samples and reported a healthy model as `Not Active`/`Unstable`. `opts.isDown` is still * honoured when a caller genuinely knows better, but nothing has to supply it. */ export declare function getVerdict(pings: PingRecord[], opts?: { httpCode?: string | null; isDown?: boolean; }): Verdict;