import { EmbeddingsInterface } from "@langchain/core/embeddings"; import { BaseChatModel } from "@langchain/core/language_models/chat_models"; import { OnModuleInit } from "@nestjs/common"; import { ConfigService } from "@nestjs/config"; import { ClsService } from "nestjs-cls"; import OpenAI, { AzureOpenAI } from "openai"; import { BaseConfigInterface, ConfigAiInterface } from "../../../config/interfaces"; import { AppLoggingService } from "../../logging/services/logging.service"; import { ModelWeight } from "../enums/model.weight"; import { ReasoningEffort } from "../enums/reasoning.effort"; import { AiConnectionType, ResolvedAiCandidate } from "../interfaces/ai-candidate.interface"; import { AiConnectionResolverService } from "./ai-connection-resolver.service"; import { EmbedderTokenBucketService } from "./embedder-token-bucket.service"; /** * Securely materialises Google Vertex credentials to a temp file. * * Security properties (Wave 4 hardening): * - UUID-unique filename — no predictable path another process can pre-create * or read by guessing. * - mode 0o600 — owner read/write only. * - registers a single best-effort `exit` cleanup that unlinks every file we * wrote, so secrets do not linger in the OS temp dir. * * @param decodedCredentials - The DECODED credentials JSON text to write. The * caller already has the decoded JSON in scope (`credentialsJson`), so the * helper writes it verbatim — it does NOT base64-decode (avoids double-decode). * @param tag - A modality tag used only to make the filename human-readable. * @returns The absolute path of the written credentials file. */ export declare function writeGcpCredentials(decodedCredentials: string, tag: string): string; /** * Validates an LLM endpoint URL before an API key is sent to it. * * Security properties (Wave 4 hardening): * - Refuses an empty / missing URL for providers that require one. * - Refuses a malformed URL. * - Refuses plaintext HTTP to a non-local host (would leak the API key on the * wire). localhost / 127.0.0.1 / ::1 / *.local are exempt (dev loopback). * - Optionally enforces an `AI_URL_ALLOWLIST` (comma-separated host suffixes). * * @throws {Error} if the URL fails any check. */ export declare function validateAiUrl(url: string, provider: string): void; /** * Narrows a CONFIG-sourced reasoning effort to the allowed set. * * `AI_REASONING_EFFORT` reaches this code as a free-form string, and a typo * (`AI_REASONING_EFFORT=lwo`) would otherwise be cast straight onto the wire. The * provider answers 400 `unsupported_value`, which `unsupportedParamFetch` then * remembers for that (parameter, value) pair — one wasted round-trip per distinct * typo, and an effort silently degraded to the provider default for the rest of the * process. Rejecting it here keeps the misconfiguration visible and costs nothing. * * Per-call values are NOT routed through this: those are typed `ReasoningEffort` at * the call site, so the compiler already guarantees them. * * @returns the effort when recognised, otherwise undefined (tier default unset). */ export declare function normaliseConfiguredReasoningEffort(value: unknown): ReasoningEffort | undefined; /** Gemini's own thinking-level vocabulary (`GoogleThinkingLevel` in @langchain/google-common). */ export type GeminiThinkingLevel = "MINIMAL" | "LOW" | "MEDIUM" | "HIGH"; /** * Is `model` a Gemini 3-or-later model, i.e. one that understands `thinkingLevel`? * * Gemini introduced `thinking_level` with the 3.x family, and sending it to an * EARLIER model is a hard error rather than a no-op — so this gate is what keeps * a `gemini-2.5-*` tier working unchanged when a reasoning effort is configured. * * Matched on the major version rather than the literal string "gemini-3" so 3.1, * 3.5 and a future 4.x are all covered without another edit. The name is searched * anywhere in the string because providers qualify it differently * ("gemini-3.1-flash-lite", "google/gemini-3.1-flash-lite"). */ export declare function supportsGeminiThinkingLevel(model?: string): boolean; /** * Maps this project's reasoning-effort vocabulary onto Gemini's thinking levels. * * `none` becomes `MINIMAL` because Gemini 3 has no zero: MINIMAL is documented as * "as close as possible to a zero budget for thinking, but still requires thought * signatures". Reporting the nearest honest equivalent beats refusing to send * anything and silently inheriting the model default. * * NOTE this deliberately does NOT go through `reasoningEffort` on the Vertex * client: @langchain/google-common maps THAT to `maxReasoningTokens`, a token * BUDGET, and Gemini 3 rejects a request carrying both a thinking budget and a * thinking level. The level is the parameter we want; the budget is a different * knob wearing a similar name. */ export declare function toGeminiThinkingLevel(effort: ReasoningEffort): GeminiThinkingLevel; export declare class ModelService implements OnModuleInit { private readonly clsService; private readonly configService; private readonly bucket?; private readonly logger?; private readonly aiConnectionResolver?; private cachedEmbedder?; /** * Connection id the {@link cachedEmbedder} was built from. The embedder is * memoised for the lifetime of the process (it owns the shared rate-limit * bucket), so without this a DB-side embedder change would stay invisible * until a restart. */ private cachedEmbedderConnectionId?; /** * Chat models built by {@link getLLM}, keyed on every parameter that is BAKED * INTO the instance (see the key built there). * * Why: `getLLM` used to construct a fresh `ChatOpenAI` — and with it a fresh * OpenAI SDK client, its agent and its fetch middleware — on every single LLM * call. Under a worker load that is thousands of short-lived clients, which is * heap the process never gets back quickly enough. * * Two constructions are deliberately NOT cached (both handled in `getLLM`): * MOCK_AI's `FakeListChatModel` (stateful — it walks a response list, so * sharing one across calls changes what tests see), and an OpenRouter tier * with a pinned `region` (its `openRouterEscalatingFetch` closure is per-call * BY DESIGN: attempt 1 hard-pins, retries allow fallbacks — sharing it would * leave every later call permanently escalated). * * `unsupportedParamFetch` is share-safe: its learned verdicts live in a * module-level map keyed by deployment, with no per-call state. */ private readonly llmCache; constructor(clsService: ClsService, configService: ConfigService, bucket?: EmbedderTokenBucketService, logger?: AppLoggingService, aiConnectionResolver?: AiConnectionResolverService); /** * Fail-closed MOCK_AI safety gate. MOCK_AI returns synthetic data (no provider * call) for every model/embedder/structured call — invaluable for local dev and * tests, catastrophic in production (it would write fake AI data into the graph). * This refuses to start when MOCK_AI is on AND the environment is production. * Reads the module-level `baseConfig` rather than the injected ConfigService on * purpose: a fail-closed safety gate must not depend on DI wiring being correct, * and `baseConfig` is a plain constant resolved at import time. */ onModuleInit(): void; private get aiConfig(); private get visionConfig(); private get audioConfig(); /** * Resolves the `.env` AI config block for a model weight. * Undefined / Normal → `ai`; Lite → `aiLite`; Large → `aiLarge`. * * This is the FINAL fallback of every chat chain — {@link getResolvedConfig} * layers the resolved (DB-first) candidate over it. */ private envTierConfig; /** Chat tier → AI connection type. Undefined / Normal → `ai`. */ private weightToType; /** * Ordered fallback candidates for a chat tier, healthiest first. * * Without a resolver (tests, minimal harnesses, an app that never registered * the AI-connection feature) this is exactly one `.env` candidate — today's * behaviour. */ getCandidates(weight?: ModelWeight): ResolvedAiCandidate[]; /** * Ordered fallback candidates for any AI connection type. * * Never throws: a resolver failure degrades to the `.env` candidate rather * than to "no AI" (spec § 5), because this sits on the hot path of every * single model construction. */ getCandidatesForType(type: AiConnectionType): ResolvedAiCandidate[]; /** Reports a transient failure so the resolver cools that connection down. */ notifyCandidateFailure(candidate: ResolvedAiCandidate): void; /** * Picks one link out of a chain. Out-of-range indexes clamp to the last * candidate, so a retry loop that outruns the chain simply keeps hammering * the final (`.env`) link instead of crashing. */ private pickCandidate; /** * The `.env` block of one connection type, normalised to a candidate. * * Used when no resolver is wired. A resolver appends its own env candidate as * the last link of every chain, so this is the no-resolver twin of that entry * and MUST stay field-for-field identical to it. * * Tolerates missing config blocks (a harness that only configures the chat * tiers must not explode when something asks for the vision chain). */ private envCandidate; /** * Resolves the effective AI config block for a model weight: the first * healthy candidate of that chat tier, expressed in the `.env` block's shape. * * The signature is deliberately unchanged, so every existing caller * (`LLMService` cost lookups, {@link supportsStrictStructuredOutput}, …) * transparently reads the DB-first configuration. Only DEFINED candidate * fields are layered over the env block, so fields the candidate shape does * not carry (`secret`) survive. */ getResolvedConfig(weight?: ModelWeight): ConfigAiInterface["ai"]; /** * Whether this tier's provider honours OpenAI's STRICT structured-output mode. * * Only the OpenAI-compatible chat-completions family enforces `strict`. * `ChatGoogleBase.withStructuredOutput` ignores the flag outright, so on Vertex * a strict-shaped schema costs the model an explicit null for every optional * field and returns no guarantee in exchange — verified against a live * gemini-2.5-flash-lite deployment, which accepts both shapes. * * Answered from the tier's PROVIDER, which this service already owns, rather * than `instanceof ChatOpenAI`: `instanceof` fails silently when two copies of * `@langchain/openai` resolve, which is exactly the dual-instance hazard the * July 2026 dependency sweep removed. */ supportsStrictStructuredOutput(weight?: ModelWeight): boolean; /** * Gets a configured LLM instance based on the current config. * * Supports multiple providers: * - `llamacpp`/`local`: Local llama.cpp server (OpenAI-compatible API) * - `openrouter`: OpenRouter cloud service * - `requesty`: Requesty proxy service * - `vertex`: Google Vertex AI (Gemini models) * - `azure`: Azure OpenAI Service — chat branch speaks the Responses API on * the GA v1 surface ({instance}.openai.azure.com/openai/v1), not chat-completions * - any other provider name: generic OpenAI-compatible endpoint (requires `url`) * * Each model weight resolves its own full config block (provider, apiKey, * url, model, …), so different tiers can live on different providers. * * Instances are CACHED per resolved parameter set (see {@link llmCache}) — * identical parameters return the same client instead of building a new SDK * client per call. MOCK_AI and region-pinned OpenRouter tiers always get a * fresh instance; see the cache's docblock for why. * * @param params - Optional parameters * @param params.temperature - Temperature for text generation (0-2, default: 0.2) * Lower = more deterministic, Higher = more creative * @param params.maxOutputTokens - Maximum output tokens (default from config) * @param params.modelWeight - Which AI tier to use (undefined → Normal) * @param params.candidateIndex - Which link of the tier's fallback chain to * build (default 0 = first healthy candidate) * @returns Configured BaseChatModel instance from LangChain * @throws {Error} If the configured LLM type is not supported */ getLLM(params?: { temperature?: number; maxOutputTokens?: number; frequencyPenalty?: number; modelWeight?: ModelWeight; /** * @deprecated Use `reasoningEffort: "none"` instead. Kept as a working alias * for backward compatibility — a published-library API is never removed. */ disableThinking?: boolean; reasoningEffort?: ReasoningEffort; /** Per-attempt request budget in ms. Defaults to `ai.requestTimeoutMs`. */ timeoutMs?: number; /** * Which link of this tier's fallback chain to build. 0 (default) is the * first healthy candidate; a retry loop advances it on a transient failure. * Out-of-range values clamp to the last candidate (the `.env` block). */ candidateIndex?: number; }): BaseChatModel; /** * Gets a configured LLM instance for vision operations based on the current config. * * Supports multiple providers: * - `llamacpp`/`local`: Local llama.cpp server (OpenAI-compatible API) * - `openrouter`: OpenRouter cloud service * - `requesty`: Requesty service * - `vertex`: Google Vertex AI (Gemini models) * - `azure`: Azure OpenAI Service — chat branch speaks the Responses API on * the GA v1 surface ({instance}.openai.azure.com/openai/v1), not chat-completions * * @param params - Optional parameters * @param params.temperature - Temperature for text generation (0-2, default: 0.1) * @param params.candidateIndex - Which link of the vision fallback chain to * build (default 0 = first healthy candidate) * @returns Configured BaseChatModel instance from LangChain * @throws {Error} If the configured LLM type is not supported */ getVisionLLM(params?: { temperature?: number; candidateIndex?: number; }): BaseChatModel; /** * Gets a configured LLM instance for audio operations based on the current config. * * Supports the same providers as getVisionLLM(): llamacpp/local, openrouter, * requesty, vertex, azure. The configured chat model must accept multi-modal * `input_audio` content parts in HumanMessage payloads (Gemini 2.5 family, * GPT-4o-audio, etc.). * * @param params.temperature - default 0.1 (deterministic transcription) */ getAudioLLM(params?: { temperature?: number; }): BaseChatModel; /** * Builds a LangChain chat model from a resolved config block. * Single source of truth for the provider switch shared by getLLM / * getVisionLLM / getAudioLLM. `credentialFileTag` keeps the per-modality * Vertex temp-credential filenames distinct. */ private buildChatModel; /** * Returns the embedder used for vectorisation. Three layers, all additive over * the raw provider embedder: * 1. MOCK_AI → a zero-vector embedder (no provider call), sized to * `embedder.dimensions` so downstream vector writes still have the right shape. * 2. When `embedder.rateLimit` is configured AND the token bucket is wired, * the provider embedder is wrapped in a RateLimitedEmbedder (distributed * token bucket + local concurrency gate + 429 handling) and CACHED on the * instance, so every caller shares one bucket/gate. * 3. Otherwise the raw provider embedder is returned unchanged. */ getEmbedder(): EmbeddingsInterface; /** * Builds the raw provider embedder for one resolved candidate. `dimensions` * falls back to the `.env` embedder block when the candidate does not carry * one, so a DB connection that omits it keeps today's vector size. */ private buildInnerEmbedder; getEmbedderDimensions(): number; /** * Whether this installation can actually run the chunking pipeline. * * Resolution goes through `pickCandidate`, the path the runtime itself uses, * so an installation whose AI lives in DB-backed AiConnection rows is judged * on what it will really use rather than on what `.env` happens to hold. * * An apiKey is deliberately NOT required: a local or gateway provider * legitimately has none, and demanding one would refuse installations that * work. Under MOCK_AI only the embedder width matters — the mock embedder * returns zero vectors of that size and calls no provider. */ isAiConfigured(): boolean; /** * Builds an OpenAI / Azure OpenAI SDK client for audio transcription. This is * the SDK-based path (`audio.transcriptions.create`), distinct from * AudioLLMService (chat-LLM / OpenAI-style /audio/transcriptions HTTP). Driven * by the `transcriber` config block (TRANSCRIBER_* env vars). */ getTranscriber(): OpenAI | AzureOpenAI; transcribeAudio(params: { filePath: string; prompt: string; language?: string; }): Promise; } //# sourceMappingURL=model.service.d.ts.map