/** * Multi-provider LLM caller with priority failover — the policy layer the * Vercel AI SDK deliberately doesn't ship. * * The AI SDK owns the transport: HTTP, SSE parsing, provider dialects * (OpenAI-compatible, Anthropic-messages, OpenRouter), tool calling, * streaming. This module owns the policy on top: * * - **Priority failover across providers AND keys.** Several declarations * may share one `id` to register multiple API keys for the same provider; * on failure the loop prefers another healthy key of the same `id` before * moving to the next `id` by priority. * - **Per-key circuit breaker.** Failures cool a key down with exponential * backoff (rate-limit / auth / transient classed separately), persisted in * an optional {@link HealthStore} (Cloudflare KV satisfies it * structurally). Without a store the loop degrades to per-request * failover. * - **Reset semantics for live previews.** A provider that dies mid-stream * forfeits its output: the consumer gets `{kind: 'reset'}` (drop * everything rendered so far) and the next candidate regenerates from * scratch. Telegram draft previews repaint per frame, so a reset is one * cheap frame. * - **Usage accounting that feeds a ledger.** Tokens are summed across every * billed attempt (a failover means more than one billed request). * `costUsd` is the ACTUAL charge when the gateway reports one — * OpenRouter's usage accounting (`providerMetadata.openrouter.usage.cost`) * or a `cost` field on the provider's raw usage frame — never a price * table estimate. * * Worker-safe: fetch/WebStreams only, no `node:*`. Peer deps: `ai`, * `@ai-sdk/openai-compatible`, `@ai-sdk/anthropic`, * `@openrouter/ai-sdk-provider` (install all four; they are small and the * provider used is picked per {@link ProviderConfig.type} at runtime). * * @example one-shot with failover * import { createLlm } from '@adriangalilea/utils/llm' * * const llm = createLlm({ * providers: [ * { id: 'openrouter', type: 'openrouter', apiKey, defaultModel: 'deepseek/deepseek-chat', priority: 0 }, * { id: 'deepseek', type: 'openai', baseUrl: 'https://api.deepseek.com', apiKey: key2, defaultModel: 'deepseek-chat', priority: 1 }, * ], * health: env.KEY_HEALTH, // optional KV namespace * }) * const { text, usage } = await llm.complete({ prompt }) * * @example streaming with tools * import { tool } from '@adriangalilea/utils/llm' * import { z } from 'zod' * * const events = llm.stream({ * prompt, * tools: { not_covered: tool({ description: '…', inputSchema: z.object({}) }) }, * }) * for await (const e of events) { * if (e.kind === 'delta') draft.append(e.text) * else if (e.kind === 'reset') draft.clear() * else if (e.kind === 'tool-call') handle(e.toolName, e.input) * } */ import { type ModelMessage, type ToolChoice, type ToolSet } from "ai"; export { jsonSchema, type ModelMessage, type Tool, type ToolSet, tool, } from "ai"; /** * One provider declaration. Multiple declarations MAY share an `id` to * register several API keys for the same provider; key health is tracked per * key (fingerprinted, never the secret), so one dead key never sidelines its * siblings. */ export interface ProviderConfig { id: string; /** * Wire dialect. `openai` = any OpenAI-compatible endpoint (DeepSeek, * vllm, mlx-lm, gateways). `anthropic` = the Anthropic messages API and * compatible gateways (GLM/Zhipu, Moonshot). `openrouter` = OpenRouter * with typed usage accounting (actual billed cost). An `openai` entry * whose baseUrl points at openrouter.ai is upgraded to `openrouter` * automatically so pre-existing configs keep their cost reporting. */ type: "openai" | "anthropic" | "openrouter"; /** Base URL. Defaults: OpenAI-compat has none (required), anthropic → api.anthropic.com, openrouter → openrouter.ai/api/v1. */ baseUrl?: string; /** Omit for auth-less endpoints (a local vllm/mlx-lm/llama.cpp server). */ apiKey?: string; /** Optional allow-list of model names; when present, defaultModel must be in it. */ models?: string[]; defaultModel: string; /** Lower tries first. Default 99. */ priority?: number; /** Upper bound on output tokens for this provider; caps the per-request ask. */ maxTokens?: number; /** Per-model temperature overrides, keyed by model name. */ temperatures?: Record; /** Fallback temperature for models not in `temperatures`. */ temperature?: number; /** Operator kill switch: skip this declaration entirely. */ disabled?: boolean; /** * Suppress reasoning/thinking output. Reasoning models otherwise burn the * output budget on thinking. anthropic → `thinking: {type: 'disabled'}`; * openrouter → `reasoning: {enabled: false}`; openai → a raw * `thinking: {type: 'disabled'}` body field (the DeepSeek-style flag; * plain OpenAI ignores unknown fields). */ disableThinking?: boolean; /** * Transport override for this provider: every request to it goes through * this function instead of the global fetch. How a Worker reaches an * endpoint it has no route to (a self-hosted engine behind a relay * socket, a tunnel, a test double). Not serializable, so it can only be * set by code, never by a config file. Honored by every dialect. */ fetch?: typeof globalThis.fetch; } /** * Minimal persistence for the circuit breaker. Structurally satisfied by * Cloudflare's `KVNamespace`. Absent → per-request failover only. */ export interface HealthStore { get(key: string): Promise; put(key: string, value: string, options?: { expirationTtl?: number; }): Promise; delete(key: string): Promise; } export interface LlmOptions { providers: ProviderConfig[]; health?: HealthStore; /** Per-attempt hard timeout in ms. Default 120_000. */ requestTimeoutMs?: number; } /** * Token accounting for one completed call: tokens summed across every billed * attempt (failovers included); `costUsd` the gateway-reported actual charge * when available. `provider`/`model` name the attempt that finished the * response. */ export interface LlmUsage { promptTokens: number; /** Prompt tokens the provider served from its KV/prompt cache, when it says (OpenAI `cached_tokens`). */ cachedTokens?: number; completionTokens: number; costUsd?: number; provider: string; model: string; } /** * One event of a streamed call. `reset` orders the consumer to discard * everything streamed so far — a fresh full response follows from another * provider. `usage` is absent when no provider reported it. */ export type LlmStreamEvent = { kind: "delta"; text: string; } | { kind: "reasoning"; text: string; } | { kind: "reset"; } /** A tool call's arguments arriving as text; `tool-call` follows with the parsed whole. */ | { kind: "tool-input-delta"; toolCallId: string; delta: string; } | { kind: "tool-call"; toolCallId: string; toolName: string; input: unknown; } | { kind: "end"; usage: LlmUsage | null; }; export interface LlmToolCall { /** The provider's id for this call: what a later turn must echo back to replay it (ds4 keys its exact-DSML replay on it). */ toolCallId: string; toolName: string; input: unknown; } export interface LlmResult { text: string; reasoning: string; toolCalls: LlmToolCall[]; usage: LlmUsage | null; } export declare class LlmError extends Error { readonly status?: number | undefined; constructor(message: string, status?: number | undefined); } export interface ChatRequest { /** Single user message. Exactly one of `prompt` / `messages`. */ prompt?: string; /** Full conversation. Exactly one of `prompt` / `messages`. */ messages?: ModelMessage[]; /** System instructions. */ instructions?: string; /** Output-token budget for this request; provider `maxTokens` caps it. Default 4096. */ maxTokens?: number; tools?: ToolSet; toolChoice?: ToolChoice; abortSignal?: AbortSignal; /** * OpenAI-dialect only: raw request-body fields spread into the JSON body * (`seed`, `enable_thinking`, `chat_template_kwargs`, …) — the escape hatch * for self-hosted endpoints with bespoke knobs. Applied to `openai` and * `openrouter` attempts; `anthropic` attempts ignore it (that API validates * its body — no arbitrary fields). */ extraBody?: Record; } export declare function createLlm(opts: LlmOptions): Llm; export declare class Llm { private readonly opts; constructor(opts: LlmOptions); /** One-shot: run {@link stream} to completion and collect the pieces. */ complete(req: ChatRequest): Promise; /** * Stream one response with failover. Attempts candidates in priority order * (healthy keys first, same-`id` siblings preferred after a failure); an * attempt that fails after emitting output yields `reset` so the consumer * starts over. Throws {@link LlmError} when every candidate is exhausted. * An attempt that finishes with no text and no tool calls counts as failed * — an empty completion is a provider bug, not an answer. */ stream(req: ChatRequest): AsyncGenerator; private runAttempt; private modelFor; private providerOptionsFor; private buildQueue; private getHealth; private recordFailure; private clearHealth; } //# sourceMappingURL=index.d.ts.map