import type { IncomingMessage } from "node:http"; import { type RouteKitPlatform } from "@velum-labs/routekit-runtime/effect"; import { Effect } from "effect"; import type { AnthropicRequest } from "./adapters/anthropic-wire.js"; import type { ResponsesRequest } from "./adapters/responses-wire.js"; import type { ProvenanceSink } from "./observability/provenance.js"; import { type Backend, type BackendRequest, type BackendRequestOptions } from "./providers/backend.js"; import type { CompositionalRoutingRuntime } from "./routing/eval-policy.js"; /** * The local-model gateway HTTP server. It fronts a single OpenAI Chat * Completions backend (the owned mlx fork by default) and exposes the wire * dialects each agent harness needs: OpenAI chat, Anthropic Messages, OpenAI * Responses, and Cursor's Responses-hybrid BYOK shape. */ export type GatewayOptions = { backend: Backend; /** Whether this gateway closes the backend; defaults to `owned`. */ backendOwnership?: "owned" | "borrowed"; /** Bind host; defaults to loopback. */ host?: string; /** Bind port; defaults to an ephemeral free port. */ port?: number; /** When set, require this bearer token (or matching `x-api-key`). */ authToken?: string; /** Optional observation sink for model calls. */ provenance?: ProvenanceSink; /** Dimension-decomposition and evidence-matrix runtime used by `model: "auto"`. */ compositionalRouting?: CompositionalRoutingRuntime; /** Optional client-authenticated Responses relay. */ codexRelay?: ProviderRelayPorts; /** Provider-native relays sharing this HTTP boundary. */ providerRelays?: Partial>; /** Whether this gateway closes supplied relay lifecycles; defaults to `owned`. */ relayOwnership?: "owned" | "borrowed"; /** Optional provider usage payload for `GET /usage`. */ usage?: () => Effect.Effect; }; export type ProviderRelayDialect = "anthropic" | "codex"; export type RequestRelay = { readonly kind: "request"; readonly dialect: ProviderRelayDialect; shouldRelay(headers: IncomingMessage["headers"], model: string | undefined, servesLocally: (model: string) => boolean): boolean; relay(headers: IncomingMessage["headers"], body: AnthropicRequest | ResponsesRequest, signal?: AbortSignal, options?: Pick): BackendRequest; }; export type ModelCatalogRelay = { readonly kind: "models"; readonly dialect: "anthropic"; models(headers: IncomingMessage["headers"], search: string, signal?: AbortSignal): BackendRequest; } | { readonly kind: "merged-models"; readonly dialect: "codex"; mergedCatalog(headers: IncomingMessage["headers"], search: string): Effect.Effect<{ models: Array>; etag?: string; } | undefined, Error, RouteKitPlatform>; mergeDataIds(data: Array<{ id: string; } & Record>, models: readonly Record[]): Array<{ id: string; } & Record>; }; export type TokenCountRelay = { readonly kind: "token-count"; readonly dialect: "anthropic"; countTokens(headers: IncomingMessage["headers"], body: AnthropicRequest, signal?: AbortSignal): BackendRequest; }; export type RelayLifecycle = { readonly kind: "lifecycle"; readonly close: Effect.Effect; }; export type ProviderRelayPorts = Readonly<{ request: RequestRelay; catalog?: ModelCatalogRelay; tokenCount?: TokenCountRelay; lifecycle?: RelayLifecycle; }>; export type Gateway = { /** Base URL clients should target (without the `/v1` suffix). */ url(): string; port(): number; /** * Graceful drain: flip `/health` to 503 and reject new model calls while * letting in-flight requests (long-lived LLM streams) finish, bounded by * `graceMs`; then close the listener and sever whatever remains. Does not * release the backend — follow with {@link close}. */ drain(graceMs?: number): Effect.Effect; /** Immediate close: equivalent to `drain(0)` plus backend/relay teardown. */ readonly close: Effect.Effect; }; export declare function startGatewayEffect(options: GatewayOptions): Effect.Effect; /** Promise adapter for hosts that create a standalone gateway. */ export declare function startGateway(options: GatewayOptions): Promise;