/** * Backend protocol execution and response transformation pipeline. * * Charter & Invariants: * - Translates requests and responses across Anthropic, OpenAI, and Responses protocols. * - Handles streaming SSE transformations, tool-call dialect recovery, think-tag stripping, and usage accumulation. * - Classifies downstream HTTP and dialect errors into retriable failovers vs terminal refusals. * - Logs remain strictly metadata-only (no payload bodies or headers). */ import type { ResolvedAttempt } from "./resolved-attempt.js"; import { type DialectRefusalSignal } from "./tool-dialects.js"; import { type RecoveredOpenAiChatProcessor } from "./openai-dialect.js"; import { type UsageAccumulator } from "./usage-observer.js"; import type { ReasoningMode, ThoughtSignatureMode, ToolCallIdMode } from "./config-types.js"; /** * A tool-call envelope was present in the response text but could not be parsed — truncated, or a * dialect variant we do not model. Distinct from a mapper defect: the relay's translation is fine, * the SERVING HOST returned an unusable body. Carried as a retriable failure so the pool fails over * to a host that parses its models' dialect. */ export declare class DialectUnparseableError extends Error { readonly dialect: string; constructor(dialect: string); } /** * A tool call recovered from assistant TEXT names a tool on the operator's destructive list. * * Unlike `DialectUnparseableError` this is NOT a statement about the host: the envelope parsed * fine. It is a config decision, so it is carried as a LOCAL failure — the walk must not fail over * (re-asking N models to produce the same refused action) and the deployment must not be charged, * the same rule under which a hard cap "never registers on the breaker — it is config, not * health". See docs/history/dialect-rescue-destructive-refusal-2026-08-24.md. */ export declare class DialectDestructiveError extends Error { readonly dialect: string; readonly refused: string[]; constructor(dialect: string, refused: string[]); } /** * Response header stating who produced an error status: the provider, or this proxy. * * Every failure out of `fetchBackend` is a synthesized `Response` — a refused document, a * translation bug and a genuinely dead provider all arrived as a bare status code, so a * caller counting backend failures (the circuit breaker) charged our own local bugs to the * provider and failed over to a second provider that would have failed identically. The * marker is what makes them separable; `fetchBackend` states it, the caller decides. */ export declare const ERROR_ORIGIN_HEADER = "x-llm-relay-error-origin"; /** `upstream` = the provider answered with this status. `local` = the proxy produced it without asking. */ export type ErrorOrigin = "upstream" | "local"; export declare function errorOrigin(res: Response): ErrorOrigin | null; /** * Response header naming the deployment that actually answered — the (provider, model) left * standing after pool expansion, benchmark ranking, breaker demotion and failover. * * A pool request's most basic debugging question is "who served this?", and until now the only * way to answer it was to correlate timestamps against the proxy's own log. When every candidate * fails it carries the list that was tried instead, so an exhausted pool is self-describing. */ export declare const SERVED_BY_HEADER = "x-llm-relay-served-by"; /** * What the `auto` model name resolved to for this turn — ` ()`. * Announces both the concrete spec that served the turn and the ladder tier * that selected it. */ export declare const AUTO_HEADER = "x-llm-relay-auto"; /** * Request header specifying the dispatch ladder tier for `auto` model resolution * (`low` | `medium` | `high` | `xhigh`). Defaults to `medium` when omitted. */ export declare const AUTO_TIER_HEADER = "x-llm-relay-tier"; /** * WHY each of those candidates dropped out — `"13 tried, 0 served: 4x402, 5x429, 3x403, 1x400"`. * * `SERVED_BY_HEADER` answers "who was tried"; this answers "what happened to them", which is the * half that turns a pool exhaustion into a diagnosis. Without it a client holds one member's * error — a HuggingFace 402 pointing at a billing page — while the other twelve failed for three * unrelated reasons, and the correct action ("use another pool") is invisible. * * A header rather than a rewritten body, deliberately: the served body stays the last candidate's * real upstream error, because a true upstream error beats a synthesized one. */ export declare const POOL_ATTEMPTS_HEADER = "x-llm-relay-pool-attempts"; /** Response metadata headers for credential-aware pool diagnostics. */ export declare const CREDENTIAL_HEADER = "x-llm-relay-credential"; export declare const CREDENTIAL_ATTEMPTS_HEADER = "x-llm-relay-credential-attempts"; /** * How many refusals in this walk said something the relay could not interpret. * * The learned-eligibility store converges only as fast as somebody explains the messages it does * not recognise, and a pull-only queue is a backlog nobody works. This is the push half: the * caller — which is usually an agent that is about to report a pool failure to a human anyway — * finds out at the moment it matters that a NEW kind of refusal just appeared, and can run * `llm-relay eligibility` while the context is still in hand. * * ⚠ It is a COUNT, never the message. The message is untrusted text from an external service, and * putting it in a response header would be the relay handing an agent attacker-controlled prose in * a field agents tend to trust. The count says "go look"; `llm-relay eligibility` shows the text * with its provenance and the enum-constrained verdicts it may receive. */ export declare const UNKNOWN_REFUSAL_HEADER = "x-llm-relay-unknown-refusal"; type OnEgress = () => void; /** * This answer came from BELOW the effort band that was asked for. * * An effort pool falls back to lower-banded live members once its own band is exhausted, because a * band with nothing behind it turns "the strongest models are busy" into "no answer at all". That * fallback is automatic — but it must never be silent. A capability downgrade that reads as an * ordinary 200 is indistinguishable from having got what you asked for, which is the failure mode * that lets a caller build on a weaker answer without knowing it did. * * Names the deployment and the band it fell out of, e.g. `groq/llama-3.3-70b (below xhigh)`. */ export declare const DEGRADED_HEADER = "x-llm-relay-degraded"; /** * The walk's FIRST choice was demoted for a spent quota and the answer came from further down. * * Same maxim as `DEGRADED_HEADER` — automatic degradation is acceptable only because it is * announced — but a different fact: capability fell below the requested band there, while here a * quota figure said the first choice was spent until a known reset. Value is one bounded line, * e.g. `groq/llama-3.3-70b (requests/minute remaining 0, provider-stated)` — axis/period/remaining * plus the basis, no credential values, nothing secret. */ export declare const QUOTA_DEMOTED_HEADER = "x-llm-relay-quota-demoted"; /** * The walk's FIRST choice was demoted for SUSTAINED MEASURED LATENCY and the answer came from * further down. Third member of the same family as `DEGRADED_HEADER` and `QUOTA_DEMOTED_HEADER`, * and it exists for the same reason: an automatic reorder is acceptable only because it is * announced. * * ⚠ It is also the ONLY surface this demotion has. Quota demotion registers a breaker cooldown, so * `/candidates` and the dashboard Cooldowns panel can see it; latency states no reset, and this * relay never invents a cooldown duration, so no cooldown is registered and those panels show * nothing. See `src/latency-demotion.ts`. * * Value is one bounded line, e.g. `nim/deepseek-ai/deepseek-v4-flash (p95 70364ms > 30000ms over * 12 samples)` — the measured figure, the ceiling it crossed and the sample count behind it. * Nothing secret, no credential values. */ export declare const LATENCY_DEMOTED_HEADER = "x-llm-relay-latency-demoted"; /** * The serving candidate was placed by the PROBATION band: a free deployment with fewer than * `minSamples` served-request samples that the relay deliberately put first to gather data on * it (`routing.probation`, default ON). * * Fourth member of the `DEGRADED_HEADER` family, and it exists for the same reason: an * automatic reorder is acceptable only because it is announced. Unlike its three siblings it * does not state that the first choice was displaced — the probation member usually IS the * first choice — but that the answer came from a deployment with almost no served-traffic * evidence behind it. * * Value is one bounded line, e.g. `opencode/muse-spark-1.3-contributor-free (0 of 5 request * samples)` — the spec and the sample count behind the banding. Nothing secret. */ export declare const PROBATION_HEADER = "x-llm-relay-probation"; /** * The walk's FIRST choice was PACED: this relay's own attempts against it in the trailing window * had reached a ceiling the deployment stated (a quota header's `limit`, an operator `limits` * figure, or a `rate-limit-*` fact learned from a 429 body), so it stepped behind the live and * slow bands and another candidate led (`routing.pacing`, default ON, `src/pacing.ts`). * * Fifth member of the `DEGRADED_HEADER` family, and it exists for the same reason: an automatic * reorder is acceptable only because it is announced. Like `LATENCY_DEMOTED_HEADER` it is the * term's ONLY surface — pacing registers no breaker cooldown (it re-resolves from the start log * on every request and lifts by itself as the window drains), so `/candidates` and the dashboard * Cooldowns panel show nothing for it. * * Value is one bounded line, e.g. `groq/llama-3.3-70b (requests/minute 30 of 30 in the trailing * minute, learned)` — the spec, the count, the ceiling and who stated it. Nothing secret. */ export declare const PACED_HEADER = "x-llm-relay-paced"; /** * A HEDGE ran: a slow in-flight attempt had the next candidate started beside it, rather than * after it. Fourth member of the `DEGRADED_HEADER` family, and the announcement half of the * duplication bound. * * ⚠⚠ **This one announces something stronger than its three siblings, and the difference matters.** * They each state that the relay REORDERED the walk. This states that the relay sent the SAME * request to a SECOND deployment — the first behaviour here that does not merely reorder. The * `CLAUDE.md` invariant reads "Acting on counts is optional, always announced, and may only * reorder"; the owner amended it for hedging on 2026-08-30, and this header is one of the three * bounds that amendment rests on. The other two are `assessCost()`-free deployments only, and the * loser aborted the moment a winner commits. * * Value is one bounded line naming BOTH deployments and which one answered, e.g. * `nim/deepseek-ai/deepseek-v4-flash -> nim/nvidia/nemotron-3-ultra-550b-a55b (hedge won after * 3210ms, input-size 1180 tokens)` — the two specs, the winner, the delay that started the hedge * and the rung of evidence that set it. `input-size` (owner direction 2026-09-04) additionally * carries the estimated input-token count that decided the size-scaled floor, because a * `per-token`/`absolute` rung's own bare name is unaffected — it is a statement about the * DEPLOYMENT, not the request. Metadata only: no credential values, no content, no token text. * * ⚠ A hedge that starts and LOSES is announced too. The duplication happened either way, and a * header that appeared only when the hedge won would under-report exactly the case an operator * needs to see — a pool duplicating requests for no benefit. */ export declare const HEDGED_HEADER = "x-llm-relay-hedged"; /** * This request was refused by an OPERATOR-SET HARD CAP (G2) — not by a provider. * * Value is one line per capped credential cell, e.g. * `a/nim/z-ai/glm-5.2 requests/day 450/450` — label/deployment, axis/period, the inclusive * used/cap pair. Present only when EVERY walked candidate was capped: a partial walk serves * from whoever remained, and the pool-attempts header carries the `Nxcapped` tally beside the * other outcomes instead. Nothing secret: the label is the config slot name, the figures are * the operator's own declaration and this relay's own ledger reading. * * BOUNDED by cell count, not just by cell width: at most `MAX_CAPPED_HEADER_CELLS` (5) cells are * named and the rest are counted as `+K more`. A capped attempt costs no walk start budget, so a * fully capped 30-member dynamic pool reaches every one of them and an unbounded join would be a * ~1.2 KB header. The header says what stopped the request; it is not a roster. */ export declare const HARD_CAP_HEADER = "x-llm-relay-capped"; /** * This answer came from a deployment that is NOT free. * * Pools rank free capacity first but no longer exclude paid capacity, so a spent free lane is not * a dead end. That is only acceptable if spending is announced: an unflagged paid response is * indistinguishable from a free one, and the difference is money. Same reasoning as * `DEGRADED_HEADER` — automatic fallback is fine, silent fallback is not. * * Carries the deployment and how its cost was assessed, e.g. * `openrouter/anthropic/claude-sonnet-5 (paid, published-price)`. */ export declare const PAID_HEADER = "x-llm-relay-paid"; /** This response contains a tool call reconstructed from a recognized text dialect envelope. */ export declare const TOOL_DIALECT_HEADER = "x-llm-relay-tool-dialect"; export declare function dialectRefusalSignalOf(response: Response): DialectRefusalSignal | undefined; /** * `tool_use` ids in this response were MINTED by the relay because the host reused ones the * conversation already carried — value `" rewritten"`. * * Same maxim as `DEGRADED_HEADER`: an automatic fix is acceptable only because it is announced. * A count, never an id: the ids themselves are in the body the caller already has, and a header * is not the place to restate content. * * ⚠ Buffered responses only. On a stream the headers are written before the first * `content_block_start` exists, so a count there could only be a guess; the streaming pass * reports through `toolUseIdRewrites()` instead, which the server reads for its log record after * the stream drains. See `src/tool-use-ids.ts`. */ export declare const TOOL_USE_IDS_HEADER = "x-llm-relay-tool-use-ids"; /** * OUTBOUND tool-call ids were rewritten to this provider's stated id shape — value * `" rewritten"`. Today that is only mistral's `^[a-zA-Z0-9]{9}$` (`compat.toolCallIds: * "strict9"`; see `src/openai-request.ts` for the 400 that states the rule). * * The request-direction sibling of `TOOL_USE_IDS_HEADER`, and the same maxim: an automatic fix is * acceptable only because it is announced, a count and never an id. Unlike the response-direction * mint the figure is final BEFORE the request is even sent, so it rides a streamed response's * headers too. */ export declare const TOOL_CALL_IDS_HEADER = "x-llm-relay-tool-call-ids"; /** * The relay overrode a `"deepseek"` target's own natural thinking decision to OFF for this * request — value `"disabled: tool_choice or missing reasoning replay"` (F10/F11, 2026-09-10; see * `openai-request.ts` `deepSeekThinkingSpec`). Either a forced tool choice (DeepSeek accepts only * one of "forced tool" and "thinking") or a replayed assistant tool-call turn with no * `reasoning_content` this relay could carry forward. * * Announced like `TOOL_CALL_IDS_HEADER`, not silent like `thoughtSignatureSentinels`: unlike that * sentinel — vendor-protocol padding that alters nothing about the caller's data — this changes * real behaviour the caller (or the routed pool's effort band) asked for, so it is closer to a * tool-call-id rewrite than to padding. A count, from `onDeepSeekThinkingOverridden`, decides * whether the header is present at all; the reason text is fixed rather than threaded through the * callback, because the callback exists to COUNT (the `onThoughtSignatureSentinels` idiom), not to * explain. */ export declare const DEEPSEEK_THINKING_HEADER = "x-llm-relay-thinking-disabled"; /** * The provider's `Retry-After` in milliseconds, or null. * * Accepts both RFC 9110 forms — delta-seconds and an HTTP-date — because providers use both * (groq sends seconds, some CDNs in front of a provider send a date). A date in the past, a * negative delta or an unparseable value yields null rather than 0: "the provider said nothing * usable" and "the provider said retry immediately" call for different cooldowns, and treating * garbage as 0 would silently disable the backoff this exists to honour. */ export declare function parseRetryAfterMs(value: string | null | undefined, now?: number): number | null; /** * A fetch completed far enough to yield provider headers, but consuming its body failed. * * This is process-local metadata rather than a wire marker. Adapters sometimes have to consume * and rebuild an error response before the routing loop sees it; retaining this discriminant on * the original Response prevents that failure from being rewritten as an empty credential error. */ export interface PostHeaderBodyFailure { readonly kind: "post-header-body-failure"; readonly cause: unknown; } /** Read adapter-private post-header body failure metadata. */ export declare function postHeaderBodyFailure(response: Response): PostHeaderBodyFailure | undefined; /** Raw model id stated by the upstream response, before any relay translation. */ export declare function upstreamReportedModel(response: Response): string | undefined; /** * How many `tool_use` ids this response had minted, or `undefined` when none were. * * On a streamed response the figure is final only once the body has drained — read it where the * server reports end-of-stream facts, not before it writes headers. */ export declare function toolUseIdRewrites(response: Response): number | undefined; /** * How many OUTBOUND tool-call ids this request had rewritten to the provider's stated shape, or * `undefined` when none were. Final at request-mapping time, so it is readable as soon as the * Response exists. */ export declare function toolCallIdRewrites(response: Response): number | undefined; /** * How many replayed tool calls this request carried gemini's thought-signature sentinel on, or * `undefined` when none did. Final at request-mapping time, like `toolCallIdRewrites`. */ export declare function thoughtSignatureSentinels(response: Response): number | undefined; /** * Fetch the resolved provider target and return an ANTHROPIC-shaped `Response`, * regardless of the backend's native wire format. For kind="anthropic" this is a * passthrough. For kind="openai" (NIM/vLLM/OpenRouter/Gemini) the request is * translated Anthropic→OpenAI and the response translated back (streaming via * llm-bridge's SSE re-encoder, non-streaming via a direct mapper) — so the rest * of the proxy (validate/repair) always sees Anthropic Messages. */ export interface FetchBackendArgs { path: string; method: string; reqBuf: Buffer; reqJson: unknown; anthropicHeaders: Record; wantsStream: boolean; /** * The operator's configured destructive-tool set (`destructiveMatcher`). REQUIRED, not * optional: it reaches the four dialect-rescue commit points, and an optional field here would * let a caller silently disable the refusal — the failure mode that gap existed as. */ isDestructive: (name: string) => boolean; usage?: UsageAccumulator; signal: AbortSignal; onEgress?: OnEgress; } export interface AnthropicToOpenAiResponsesOptions { /** The resolved deployment's model id. Absent => no `model` key, same convention as `openai-request.ts`. */ model?: string | undefined; /** Whether THIS hop streams — a relay decision, not the caller's. Falls back to the body. */ stream?: boolean | undefined; /** Same resolved mode `anthropicRequestToOpenAi` receives; see that module for the provenance. */ toolCallIds?: ToolCallIdMode | undefined; onToolCallIdsRewritten?: ((count: number) => void) | undefined; thoughtSignature?: ThoughtSignatureMode | undefined; onThoughtSignatureSentinels?: ((count: number) => void) | undefined; } /** * Translate one Anthropic Messages request body into an OpenAI Responses request body. * * Item order is preserved exactly. Unknown top-level fields are not forwarded — a translation * between two contracts, not a passthrough, the same rule `anthropicRequestToOpenAi` states. * * @throws {RequestMappingError} for a block or declaration that cannot be represented. */ export declare function anthropicRequestToOpenAiResponses(reqJson: unknown, opts?: AnthropicToOpenAiResponsesOptions): Record; /** * Perform one upstream inference call against the resolved attempt. * * For an `anthropic`-kind target this forwards native Messages request bytes and parses native * Messages response bytes. For an `openai`-kind target it transcodes documents, maps the request to * Chat Completions (`openai-request.ts`), parses the completion and translates it back to Anthropic * Messages (`openAiResponseToAnthropic`). In both directions the rest of the proxy (validate/repair) * always sees Anthropic Messages. */ export declare function fetchBackend(attempt: ResolvedAttempt, args: FetchBackendArgs, fetchFn?: typeof fetch): Promise; /** * Map a non-streaming OpenAI chat completion into an Anthropic message. * * `schemas` enables recovery of a tool call the HOST failed to parse: some free hosts return the * model's native tool-call dialect as assistant TEXT instead of populating `tool_calls`, which * without this reaches the client as markup it treats as a final answer (see * docs/tool-call-dialect-leak.md). Omitting it keeps the pure translation behaviour. */ export declare function openAiResponseToAnthropic(j: Record, model: string, schemas: Map; }> | undefined, isDestructive: (name: string) => boolean, reasoning?: ReasoningMode | undefined): object; export type OpenAiFrontProtocol = "chat" | "responses"; /** * Translate an Anthropic non-streaming message into an OpenAI response. * * `protocol` selects the outer envelope: Chat Completions (`chat.completion`) vs Responses * (`response`). Used by the OpenAI front (`fetchOpenAiFront`) when an `anthropic`-kind target * answers a Codex or OpenAI-native request: llm-bridge models only Chat Completions, and its IR * loses tool calls, stop reasons and usage on the way back to the caller. */ export declare function anthropicMessageToOpenAi(body: Record, protocol: OpenAiFrontProtocol, fallbackModel?: string, /** * The RESOLVED `compat.reasoning` mode of the target that answered, handed in exactly as the * request mapper receives it — never sniffed from a provider identity. ONLY `"deepseek"` lets a * `thinking` block ride the Responses envelope as a `reasoning` item (the response-direction end * of that seam); every other target's bytes are the pre-2026-09-15 document, thinking dropped as * before. Absent ⇒ `"none"` ⇒ dropped, so the existing three-argument callers are unchanged. */ reasoning?: ReasoningMode | undefined): Record; /** * Coerce an upstream error body into the OpenAI error envelope — WITHOUT rewriting one that * already conforms. * * The OpenAI front promises "OpenAI in, OpenAI out", but on the error path it returned whatever * shape the provider chose. Gemini wraps its error in a JSON ARRAY (`[{"error":{…}}]`); a client * reading `response.choices[0]` gets `undefined` from that and reports a malformed completion, * so a plain 429 surfaces as "the model returned garbage" — which is exactly how a rate limit * cost two days of debugging on the caller's side. * * Rules, in order: * - already `{error:{…}}` → returned BYTE-EXACT. A conforming provider's message, code and * type are its own to state, and rewriting them would lose detail. * - `[{error:{…}}, …]` → unwrapped to the element. Same fields, now at the top level. * - anything else (HTML, text, * a bare string, empty) → wrapped, with the original preserved as the message. * * Returns null when the body is already conforming, so the caller can stream the original bytes * rather than re-serialize them. */ export declare function normalizeOpenAiErrorBody(body: string, status: number): string | null; /** * OpenAI-compatible FRONT: an OpenAI Chat Completions or Responses request comes in, its * `model` has already been resolved to a provider target by namespace/tier routing. * * The common case remains a byte-transparent OpenAI→OpenAI Chat Completions proxy. The other * combinations use the same Anthropic-shaped internal seam as the Messages front: * OpenAI request → Anthropic request → resolved backend → Anthropic response → OpenAI response. * That makes an Anthropic passthrough usable from Codex and OpenAI-native IDEs without changing * the existing Claude client path. */ export interface FetchOpenAiFrontArgs { reqJson: unknown; wantsStream: boolean; signal: AbortSignal; protocol?: OpenAiFrontProtocol; anthropicHeaders?: Record; /** See `fetchBackend`'s field of the same name: required so no caller can silently opt out. */ isDestructive: (name: string) => boolean; processRecoveredChat?: RecoveredOpenAiChatProcessor; usage?: UsageAccumulator; onEgress?: OnEgress; /** * Catalog max-output lookup for the Responses front's `anthropic`-kind fallback (2026-09-09). * Anthropic's Messages API REQUIRES `max_tokens`, and `openaiResponsesRequestToAnthropic` no * longer invents a value when the caller stated none (see that module's header) — so a * passthrough target still needs one resolved, and the catalog's published figure is the * second-strongest evidence after the deployment's own learned `max-output` fact. Optional: * absent (or a provider/model the catalog holds nothing for) falls straight through to * `DEFAULT_RESPONSES_MAX_TOKENS`. */ catalogLimits?: (provider: string, model: string) => { maxOutputTokens: number | null; } | null | undefined; } /** * OpenAI-compatible FRONT: an OpenAI Chat Completions or Responses request comes in, its * `model` has already been resolved to a provider target by namespace/tier routing. * * The common case remains a byte-transparent OpenAI→OpenAI Chat Completions proxy. The other * combinations use the same Anthropic-shaped internal seam as the Messages front: * OpenAI request → Anthropic request → resolved backend → Anthropic response → OpenAI response. * That makes an Anthropic passthrough usable from Codex and OpenAI-native IDEs without changing * the existing Claude client path. */ export declare function fetchOpenAiFront(attempt: ResolvedAttempt, args: FetchOpenAiFrontArgs, fetchFn?: typeof fetch): Promise; export {};