import type { Config } from "../config.js"; import type { ModelCatalog } from "../catalog.js"; import { type PingResult } from "./ping.js"; import { type PingRecord } from "./metrics.js"; import { type CredentialId } from "../credential-id.js"; import { type QuotaObservation } from "../quota-observation.js"; export type PingMode = "speed" | "normal" | "slow" | "forced"; export declare const PING_MODE_INTERVALS: Record; export declare const SPEED_MODE_DURATION_MS = 60000; export declare const IDLE_SLOW_AFTER_MS = 300000; /** * How often one credential's provider-stated spend headroom is re-asked (`spend-headroom.ts`). * Slow on purpose: the figure moves with the account's own spend, and the `allowance-exhausted` * fact it feeds carries a 1h TTL — a 15-minute cadence keeps the fact alive while a condition * holds and notices bought credits within one interval, at four requests per credential per hour. */ export declare const SPEND_POLL_INTERVAL_MS: number; /** * How often ONE cell cooling on a guessed 429 rung is re-probed (`reprobeRateLimited`). A minute * bounds the spend to one request per cooling cell per minute — against a cell the walk is not * serving anyway — while cutting a 2 m / 10 m / 1 h / 24 h guess down to at most a minute past * the moment the deployment actually recovers. A stated `Retry-After` is never re-probed early * (`REPROBE_TARGETS_COOLDOWN` in `circuit-breaker.ts`), so this cadence only ever shortens the * relay's OWN guesses. */ export declare const RATE_LIMIT_REPROBE_INTERVAL_MS = 60000; /** Re-probes spent per tick, soonest-lifting cells first — the same flooding bound as the catalog probe. */ export declare const MAX_RATE_LIMIT_REPROBES_PER_TICK = 3; /** * The ONE thing the ping loop may do to the breaker (2026-09-15): end a 429-sourced cooldown when * a probe answers 200, and ask which cells are worth probing for that. A narrow structural port, * not a `CircuitBreaker` import — the loop must not be able to reach outcome recording, and * `CircuitBreaker` satisfies it directly so `server.ts` passes the breaker itself. */ export interface RateLimitRecoveryPort { rateLimitCoolingCells(now: number): ReadonlyArray<{ readonly provider: string; readonly model: string | null; readonly credentialId: string; readonly cooldownUntil: number; }>; endRateLimitCooldown(cell: { credentialId: string; model: string | null; }, at: number): boolean; } export interface ModelHealthSummary { providerKey: string; modelId: string; avgMs: number; p95Ms: number; jitterMs: number; stabilityScore: number; uptimePct: number; verdict: string; lastPingCode: string | null; lastPingMs: number | null; } /** * Concrete deployments that routing can currently select, ordered for useful early coverage. * Pool leaders come first, then pinned routes/subagent choices, then the remaining pool members. */ export declare function collectRoutableModels(cfg: Config): Map; export declare class PingLoop { private cfg; private catalog; private opts; private mode; private modeSource; private intervalMs; private speedUntil; private lastActivityAt; private resumeSpeedOnActivity; private timerObj; private running; private pingHistory; /** Quota is credential and model scoped; provider-wide percentages were always ambiguous. */ private latestQuota; /** Request-local fleet cursor persisted between ticks so one-due-model ticks still rotate. */ private credentialCursors; /** When each credential's spend headroom was last asked for — the SPEND_POLL_INTERVAL_MS gate. */ private spendPolledAt; /** When each cooling cell was last re-probed — the RATE_LIMIT_REPROBE_INTERVAL_MS gate, keyed `credentialId/model`. */ private rateLimitReprobedAt; constructor(cfg: Config, catalog: ModelCatalog, opts?: { fetchFn?: typeof fetch; autoStart?: boolean; probeCachePath?: string; /** * Called once per SELF-SCHEDULED loop iteration, contained — the lane cadence's entry * point (`lane-cadence.ts`). A hook, not an await: lane work runs detached, so a * minutes-long lane command can never delay an HTTP probe tick. ⚠ Deliberately NOT fired * from `tickOnce` itself: the admitted `GET /ping` route calls `tickOnce` directly, and * the request path must not be able to initiate lane work (the closeout audit of * 2026-08-30 caught exactly that leak) — only the timer loop advances the lane cadence. */ onTick?: ((now: number) => void) | undefined; /** * The breaker, through `RateLimitRecoveryPort` (2026-09-15). Absent ⇒ a probe success * still retracts cooling FACTS as before and touches no breaker cooldown, and no cell is * re-probed for recovery — the pre-2026-09-15 behaviour exactly. */ rateLimitRecovery?: RateLimitRecoveryPort | undefined; }); getMode(): PingMode; getIntervalMs(): number; setPingMode(nextMode: PingMode, source?: string): void; noteUserActivity(): void; refreshAutoPingMode(): void; recordPing(providerKey: string, modelId: string, res: PingResult, timestamp?: number, credentialId?: CredentialId): void; /** * Re-probe cells cooling on a 429 rung the relay GUESSED, so a recovered deployment is not * parked for the rest of its escalation step (owner, 2026-09-10: *"The relay should be polling * to see if things start working again anyway."*). Without this the catalog cadence would reach * such a cell only at its 24 h TTL, which is longer than three of the four rungs. * * Bounded twice: one probe per cell per `RATE_LIMIT_REPROBE_INTERVAL_MS`, and at most * `MAX_RATE_LIMIT_REPROBES_PER_TICK` per tick, soonest-lifting cells first. The probe goes * through `recordPing`, so a 200 ends the cooldown through the port and a 429 records an * ordinary failed probe — it never reaches the breaker's escalation ladder, which counts what * REAL traffic saw. Only openai-kind deployments are probed, as the catalog cadence does; a * model-less cell (a passthrough) has nothing to probe. Contained per cell. */ reprobeRateLimited(now?: number): Promise; /** * What `reprobeRateLimited` needs to probe one cooling cell, or null when the cell is not due * (probed within the interval), has no model, is not an openai-kind provider, or names no * enabled credential slot that allows the model and resolves to a key. */ private reprobeTarget; /** * Record a REAL SERVED REQUEST's latency against a deployment, with the output-token count when * the provider reported one. * * Owner decision 2026-08-30: the latency signal reads the PROBE dataset, and the probe dataset * carries request samples too, so a per-token rate can come from actual traffic rather than from * a one-token probe. * * WARNING: this must never look like a probe. `recordRequestSample` leaves every * probe-scheduling field alone, and this method deliberately does NOT touch `latestQuota` or * the entry status either. A served request already reported its quota headers through the * request path's own observer; re-recording them here would double-count one observation. * * WARNING: tokens ABSENT means unknown, never zero. A sample with no token count still measures * absolute latency; it simply cannot contribute to the per-token statistic. */ recordRequestLatency(providerKey: string, modelId: string, sample: { ms: number; tokens?: number; }): void; /** * Recent samples for a model — from memory, falling back to what previous runs persisted. * * ⚠ The fallback is the whole point. This used to read `pingHistory` alone, a Map built only * during the current process's life, while `recordProbeResult` wrote every probe to * `probe-cache.json` and nothing ever read it back. Every restart therefore reset every model * to `Pending` with `p95: -1`, so a proxy that restarts at all — a laptop that slept, an * upgrade, a crash — never accumulated latency history for anything. The disk is the long-term * record this is supposed to be keeping; memory is just the hot copy. * * ⚠ Since 2026-08-30 the window carries REQUEST samples beside probes (`source: "request"`, with * a token count). This is the seam `latency-demotion.ts` reads, which is exactly why the * fallback matters there too: without it the latency term would go inert after every restart. */ getModelPings(providerKey: string, modelId: string): PingRecord[]; private readPersisted; private probeCacheOpts; /** * Long-run uptime for a model across every probe ever recorded, or null if never probed. * * The rolling window answers "lately"; this answers "ever", and it is what keeps a model's * record from being erased by one bad afternoon inside the window. */ getLifetimeUptimePct(providerKey: string, modelId: string): number | null; /** * Returns quota only for its exact credential/model cell. A cold default cell may rehydrate * synthetic-probe observations from disk; another credential must never inherit that balance. */ getQuotaObservations(credentialId: CredentialId, modelId: string): QuotaObservation[]; private quotaKey; getModelSummary(providerKey: string, modelId: string): ModelHealthSummary; /** * Ask each provider for its stated spend headroom and feed the answer to `spend-headroom.ts`. * * Egress happens only where a provider actually publishes the figure — `fetchProviderQuota` * fetches for an OpenRouter base and no-ops for everyone else — so this walk costs nothing for * a config with no such provider. One ask per credential slot per SPEND_POLL_INTERVAL_MS; the * gate is stamped before the fetch so a failing endpoint is not re-asked every tick. A failed * or unparseable answer applies nothing in either direction (`unknown` has no effect), and any * throw is contained: a health poll must never break the ping loop. */ pollSpendHeadroom(now?: number): Promise; private pollSlotSpendHeadroom; tickOnce(scope?: "catalog" | "routable"): Promise; start(): void; stop(): void; }