import type { Config, OffloadRule } from "./config-types.js"; import type { CredentialSource } from "./authEnv.js"; import type { ModelCatalog } from "./catalog.js"; import type { PingLoop } from "./ping/cadence.js"; import { type StrengthBasis } from "./benchmarks.js"; import { type TierData } from "./tier-data.js"; import { type MetadataSource } from "./metadata.js"; import { type CircuitBreaker, type CooldownSource } from "./circuit-breaker.js"; import { type TelemetryData } from "./ping/runtime-telemetry.js"; import { type QuotaObservation } from "./quota-observation.js"; import { type LocalUsedReading, type RemainingResolution, type ResetsAtResolution } from "./availability.js"; /** * The §5.1/§5.2 ladders resolved for ONE (axis, period) bucket of one credential cell. * * Kept raw and un-blended like every other Candidate dimension: remaining, limit and localUsed * each travel with their own basis so a reader can tell a provider's stated figure from this * relay's own arithmetic. `routingEligible` reports whether the figure MAY gate routing under * spec M2 — it does not mean anything gates today. */ export interface CandidateAvailability { axis: "requests" | "tokens"; /** * ⚠ Includes `"unknown"` — the same spelling `dashboard-contract.ts` `QUOTA_PERIODS` has always * carried. A provider stating limit/remaining/reset on a header whose NAME omits the period * (groq, the `anthropic-ratelimit-*` family) produces exactly such a row, and hiding it here * while the dashboard shows it is how the two operator surfaces come to disagree about a cell. */ period: "minute" | "day" | "month" | "unknown"; limit: number | null; limitBasis: RemainingResolution["limitBasis"]; remaining: number | null; remainingBasis: RemainingResolution["basis"]; localUsed: number | null; localUsedBasis: LocalUsedReading["basis"]; resetsAt: number | null; resetsAtBasis: ResetsAtResolution["basis"]; /** True when the observation rung 1 used is still inside the CURRENT period. */ observedInCurrentPeriod: boolean; staleObservations: number; routingEligible: boolean; } /** * Everything known about one offload destination, kept as SEPARATE raw dimensions. * * Every raw dimension remains visible. Pool ordering additionally exposes one transparent * deployment-fitness scalar and its capability/operations/metadata components; it is necessary * to order candidates, not a claim that the underlying measurements are interchangeable. */ export interface Candidate { spec: string; provider: string; model?: string; /** The credential cell this row describes. */ credentialId: string; /** Non-secret credential-slot diagnostics; key material is never included. */ credential: { label: string; authEnv: string | null; enabled: boolean; models: readonly string[] | null; state: "not-declared" | "declared-present" | "declared-missing"; source: CredentialSource | null; /** Whether this slot's optional model allow-list includes the row's deployment. */ modelAllowed: boolean; }; /** Pools this spec belongs to, and subagent tiers currently pointing at it. */ pools: string[]; subagentTiers: string[]; /** This slot's auth env var is populated; intentional passthrough/keyless slots are usable. */ hasKey: boolean; /** In the provider's live /models catalog. null = not checkable (anthropic kind, or catalog down). */ listed: boolean | null; /** Which snapshot row `scores` came from, and how confidently. `fuzzy` = a similarly-named but * DIFFERENT model's row, so those numbers are indicative. Null = no row matched. */ capabilityMatch: { name: string; match: "exact" | "fuzzy"; } | null; health: { verdict: string; avgMs: number | null; p95Ms: number | null; jitterMs: number | null; uptimePct: number | null; lastPingCode: string | null; lastPingMs: number | null; } | null; /** Typed observations for this credential/model cell. Empty means not measured. */ quota: QuotaObservation[]; /** * The spec §5 ladders applied to this cell, one entry per (axis, period) bucket with any * evidence. Beside `quota`, never blended into it: `quota` is what providers SAID, * `availability` is what that leaves OVER after staleness and this proxy's own usage. * Display-only; nothing here reorders a candidate (Gap 12 owns any routing use). */ availability: CandidateAvailability[]; breaker: { open: boolean; consecutiveFailures: number; lastStatus: number | null; cooldownRemainingMs: number; cooldownSource: CooldownSource | null; unexplained429s: number; /** * Credential faults observed on real traffic, on their own axis. * * A member answering 401 on every call used to be indistinguishable here from a healthy * one — verdict `Pending`, breaker `closed`, no failure count — because a 401 is * deliberately not health data and so reached none of the fields above. It is still not * health data; it is reported as what it is. `credentialFault` true means the router is * currently DEMOTING this member (it is tried only after every other candidate fails), * and it expires, so a rotated key recovers without a restart. */ credentialFailures: number; lastCredentialStatus: number | null; credentialFault: boolean; }; /** * What this deployment (or its provider, or its group) has STATED about itself — the learned * facts from `target-facts.ts`, each with the scope it applies at. * * Separate from `breaker` on purpose, and for the same reason credential faults are: these are * not health measurements, they are things a backend said. The breaker fields describe how a * deployment has been behaving; these describe what it is entitled to. A member cooling on an * account-wide credit balance and one cooling on its own repeated timeouts look identical in * `breaker` alone, and they call for completely different responses — one is "wait or switch * provider", the other is "this deployment is sick". */ /** * What this deployment (or its provider, or its group) has STATED about itself — the learned * facts from `target-facts.ts`, each with the scope it applies at. The measurement kinds * (`context-limit`, `max-output`, `rate-limit-rpm|rpd|tpm|tpd`) also carry `value`: the ceiling * itself, as stated. Display-only — nothing here reorders or gates a candidate. */ facts: Array<{ kind: string; scope: string; expiresInMs: number; value?: number; }>; /** * G2's operator-set hard cap for this cell, as `evaluateHardCap` sees it RIGHT NOW — the same * resolver the request path refuses on, so this view can never disagree with enforcement. * Null when nothing is declared, the switch is off, or usage is unmeasured (unknown ⇒ no * refusal ⇒ no row). Beside `availability` rather than inside it, because a cap is an * OPERATOR instruction, not an observation about the deployment. */ hardCap: { axis: "requests" | "tokens"; period: "minute" | "day"; cap: number; used: number; /** The cap is an operator ASSERTION — labelled, so a machine consumer never reads it as measured. */ basis: "operator-declared"; /** Which declaration site supplied it, and so whose usage `used` counts (see `hard-cap.ts`). */ source: "provider" | "credential" | "provider-model" | "credential-model"; scope: "credential" | "deployment"; resetsAt: string; /** The reset is a UTC period boundary this relay derived, never a figure anyone published. */ resetsAtBasis: "derived-boundary"; } | null; /** Observed real traffic through this proxy (not synthetic probes). */ observed: { totalCalls: number; successCalls: number; avgLatencyMs: number | null; lastCalledAt: string | null; } | null; /** Provider-reported completion-token coverage; null means no usage was reported. */ completionTokens: { reported: number | null; reportedCalls: number; totalCalls: number; }; /** * Limits, each with its own provenance. `provider` = this provider published it about its own * deployment; `reference` = borrowed from another provider serving the same model id (different * deployment, so indicative only — `metadataReferenceFrom` names it). Null = nobody publishes it. * * `metadataReferenceFrom` is `openrouter` when the borrowed row is that exact model id, and * `openrouter:` when the snapshot row was only a FUZZY name match — i.e. the * number belongs to a different SKU. Read it before quoting a reference figure. */ contextLength: number | null; contextLengthSource: MetadataSource | null; maxOutputTokens: number | null; maxOutputTokensSource: MetadataSource | null; metadataReferenceFrom?: string; /** Per-million-token price, with the same provenance rules — a model free on one host and * metered on another must not report the other's rate. */ pricePerMTokIn: number | null; pricePerMTokOut: number | null; priceSource: MetadataSource | null; supportsTools: boolean | null; /** Which leaderboards published anything about this model. */ capabilitySources: string[]; /** Raw per-source capability values — kept separate, never collapsed into one another. */ scores: { /** Berkeley Function-Calling: tool-call accuracy, multi-turn, irrelevance detection. */ bfclOverall: number | null; bfclMultiTurn: number | null; bfclIrrelevance: number | null; /** Artificial Analysis indices, via OpenRouter. */ aaIntelligence: number | null; aaCoding: number | null; aaAgentic: number | null; /** Aider polyglot edit benchmark + edit-format compliance. */ aiderPassRate: number | null; aiderWellFormed: number | null; /** Design Arena Elo, averaged per arena (means, not measurements — see the `_mean` naming). */ designArenaAgentsEloMean: number | null; designArenaModelsEloMean: number | null; /** LMArena. */ arenaRating: number | null; arenaRank: number | null; }; /** What the proxy itself sorts by, with the component scores and provenance beside it. */ sortInputs: { /** Drives pool ordering: 75% capability, 20% operations, 5% task-fit metadata. */ fitness: number; capability: number; operational: number; metadata: number; /** Confidence-adjusted capability used in deployment-fitness ordering. */ strength: number; /** Raw dimension-balanced capability used for effort floors. */ rawStrength: number; /** 0-1 evidence confidence applied to rawStrength. */ strengthConfidence: number; /** snapshot | neutral. Operational telemetry never stands in for capability. */ strengthBasis: StrengthBasis; /** Direct capability signals behind the estimate. */ strengthSignals: string[]; /** Capability plus separate task-fit publications used by the admission evidence gate. */ publishedSignalCount: number; capabilityDimensions: Partial>; directDimensions: string[]; imputedDimensions: string[]; /** Specialized benchmark fit input, kept separate from raw capability. */ benchmarkTaskFit: number | null; /** * Drives circuit-breaker candidate ordering. **null when nothing has been measured** — * it used to report 100 for an untracked target, which made "never probed" and "proven * fast" the same number and the same sort position (INV-TS-7). */ breakerStability: number | null; }; } export interface CandidatesView { generated_at: string; /** True when at least one client-specific rule is enabled. */ offload_enabled: boolean; /** The per-originating-client rules; empty for the legacy boolean form. */ offload_clients: Record; note: string; candidates: Candidate[]; } /** Build the un-blended decision table for offload targets. */ export declare function buildCandidates(cfg: Config, opts?: { catalog?: ModelCatalog; pingLoop?: PingLoop; breaker?: CircuitBreaker; provider?: string; now?: string; nowMs?: number; /** Capability snapshot override. Injected the same way `breaker`/`nowMs` are, so the * match-quality behaviour can be exercised against a fixed row set instead of whatever * `npm run sync:tiers` last wrote. Omitted ⇒ the real snapshot. */ tierData?: TierData | null; /** Runtime telemetry override for deterministic views/tests. Omitted ⇒ the live store. */ telemetry?: TelemetryData; /** * The accounting store's in-memory window read (G2). Absent — the CLI against a remote * proxy, or a bare programmatic proxy — means no cell can show a reached cap, because * usage is unmeasured and unknown refuses nothing. */ accounting?: Pick | null; }): Promise;