/** * kosha-discovery — Shared LiteLLM catalog loader. * * Single source of truth for the community-maintained LiteLLM model catalog * used by both the discovery seed (origin providers without an API key) and * the post-discovery pricing enricher. * * Hardening: * - HTTPS-only, fixed source URL — no caller-supplied URLs. * - Promise-deduplicated singleton load — concurrent callers share one fetch. * - AbortController-bounded fetch with explicit timeout. * - Response size cap before parsing to prevent memory bombs. * - {@link quarantineEntries} runs before any field is read, dropping an * unclean model entry rather than the whole catalog. * - Entry-count cap after parse to bound downstream work. * * The catalog source is the same one already trusted by the existing * enricher; this module just centralises the load. * @module */ import { type QuarantinedEntry } from "../security.js"; /** Pinned upstream catalog URL — HTTPS only, no user override. */ export declare const LITELLM_CATALOG_URL = "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json"; /** * Shape of a single entry in the LiteLLM pricing JSON. * Fields are all optional because the upstream schema evolves over time; * consumers must defensively handle missing fields. */ export interface LiteLLMModelEntry { max_tokens?: number; max_input_tokens?: number; max_output_tokens?: number; input_cost_per_token?: number; output_cost_per_token?: number; input_cost_per_reasoning_token?: number; output_cost_per_reasoning_token?: number; reasoning_input_cost_per_token?: number; reasoning_output_cost_per_token?: number; cache_read_input_token_cost?: number; cache_creation_input_token_cost?: number; input_cost_per_token_batches?: number; output_cost_per_token_batches?: number; input_cost_per_image?: number; output_cost_per_image?: number; input_cost_per_audio_token?: number; output_cost_per_audio_token?: number; input_cost_per_audio_per_second?: number; output_cost_per_audio_per_second?: number; input_cost_per_video_per_second?: number; output_cost_per_video_per_second?: number; input_cost_per_video_token?: number; input_cost_per_character?: number; output_cost_per_character?: number; input_cost_per_token_above_128k_tokens?: number; output_cost_per_token_above_128k_tokens?: number; input_cost_per_token_above_200k_tokens?: number; output_cost_per_token_above_200k_tokens?: number; deprecation_date?: string; litellm_provider?: string; mode?: string; supports_function_calling?: boolean; supports_parallel_function_calling?: boolean; supports_tool_choice?: boolean; supports_response_schema?: boolean; supports_system_messages?: boolean; supports_vision?: boolean; supports_audio_input?: boolean; supports_audio_output?: boolean; supports_video_input?: boolean; supports_prompt_caching?: boolean; supports_reasoning?: boolean; output_vector_size?: number; } /** * Model entries the threat scan dropped from the last successful load. * * A quarantined entry is a supply-chain signal — a base64 credential or a * script payload someone committed to the upstream catalog. Dropping the row * keeps the other few thousand models usable, but the drop has to be * reportable rather than silent, which is what this exposes. */ export declare function liteLLMQuarantined(): QuarantinedEntry[]; /** * Fetch the LiteLLM catalog with full hardening, returning a defensively * filtered map. Concurrent callers share the same in-flight promise. */ export declare function loadLiteLLMCatalog(): Promise>; /** * Reset the cached catalog. Test-only helper — callers in production code * should rely on the process-lifetime cache. */ export declare function resetLiteLLMCatalogCache(): void; //# sourceMappingURL=litellm-catalog.d.ts.map