import { AIMessage, BaseMessage } from "@langchain/core/messages"; import { DynamicStructuredTool } from "@langchain/core/tools"; import { ConfigService } from "@nestjs/config"; import { ZodType } from "zod"; import { AgentMessageType } from "../../../common/enums/agentmessage.type"; import { TokenUsageRecorderInterface } from "../../../common/tokens"; import { BaseConfigInterface } from "../../../config/interfaces"; import { TokenUsageService } from "../../../foundations/tokenusage/services/tokenusage.service"; import { ModelWeight } from "../enums/model.weight"; import { ReasoningEffort } from "../enums/reasoning.effort"; import { LLMCacheService } from "./llm-cache.service"; import { ModelService } from "../../llm/services/model.service"; import { LLMCallDumper } from "./llm-call-dumper.service"; export { injectOpenRouterProvider } from "./openrouter-fetch"; /** * True for the abort a request timeout raises, whichever layer raised it — the * OpenAI SDK's `APIConnectionTimeoutError`, undici's `TimeoutError`/ * `AbortError` DOMException, or a LangChain wrapper around either. Matched on * name and message because the concrete class differs per transport. */ export declare function isTimeoutError(error: unknown): boolean; /** * True for a failure that means the request never reached a working provider, * and is therefore worth another attempt: a DNS/socket-level error code, an * HTTP 429 or 5xx, or the message text either of those arrives as. * * All three are checked because the same failure wears different clothes per * transport: undici puts the code on `error.cause.code` behind a bare * `TypeError: fetch failed`, the OpenAI SDK puts the status on the error and * the code in the message, and LangChain re-wraps both in a plain `Error`. * * Deliberately NOT transient: a stall that burned its whole deadline * ({@link LLMTimeoutError}), a refusal (402/403), a malformed request (400) or * a parse failure. Those either already had their retry (`call()` re-issues a * timed-out attempt once, escalating the OpenRouter pin) or will fail * identically forever. The 429 vocabulary matches the one * `VisionLLMService.isRateLimitError` retries on, so the two agree on what a * rate limit looks like. * * Exported as a standalone function (with {@link LLMService} keeping a private * method that delegates to it) so the other modality services — which have no * `LLMService` dependency and must not grow one — classify a failure by exactly * the same rule before failing a connection over to the next candidate. */ export declare function isTransientNetworkError(err: unknown): boolean; /** * Drop array entries that do not satisfy their element schema, leaving the rest intact. * * Zod validates a payload all-or-nothing, which is the wrong trade when a response is * mostly good: one corrupt entry from the model, or one half-written entry left by a * truncated stream, throws away every complete sibling alongside it. Filtering first * keeps what parsed and loses only what did not. * * Shared by the lenient `tool_calls` rung and the truncation-repair rung of the salvage * ladder. Returns the cleaned object and how many entries were dropped per field, so the * caller can say what it discarded rather than silently shrinking the result. */ export declare function filterInvalidArrayEntries(outputSchema: unknown, args: Record): { cleaned: Record; dropped: Record; }; /** * Thrown when a provider call burns its whole deadline without settling. * Distinguishable from a provider error so callers can treat a stall * differently from a refusal if they choose — both are retryable. */ export declare class LLMTimeoutError extends Error { readonly label: string; readonly deadlineMs: number; constructor(label: string, deadlineMs: number); } /** * Parameters for LLM service calls */ interface LLMCallParams { inputParams: Record; inputSchema?: ZodType; outputSchema: ZodType; systemPrompts: string[]; instructions?: string; temperature?: number; history?: Array<{ role: AgentMessageType; content: string; }>; maxTokens?: number; timeout?: number; metadata?: Record; stopSequences?: string[]; maxHistoryMessages?: number; validateInput?: boolean; tools?: DynamicStructuredTool[]; maxToolIterations?: number; modelWeight?: ModelWeight; tokenUsageType?: string; relationshipId?: string; relationshipType?: string; cacheable?: boolean; disableThinking?: boolean; reasoningEffort?: ReasoningEffort; } export declare class LLMService { private readonly modelService; private readonly config; private readonly dumper; private readonly tokenUsageService; private readonly cache?; private readonly tokenUsageRecorder?; private readonly logger; /** * LangChain's ChatModel.invoke crashes with `TypeError: Cannot read * properties of undefined (reading 'message')` when the provider returns a * response with no candidates (Gemini does this on malformed function calls * and safety blocks). Convert that cryptic internal crash into a clear, * retryable error; every other error passes through untouched. */ private static normaliseEmptyResponseError; constructor(modelService: ModelService, config: ConfigService, dumper: LLMCallDumper, tokenUsageService: TokenUsageService, cache?: LLMCacheService, tokenUsageRecorder?: TokenUsageRecorderInterface); /** * The per-ATTEMPT budget for one provider request. `params.timeout` wins; * otherwise the configured default (`AI_REQUEST_TIMEOUT_MS`). */ private attemptTimeoutMs; /** * Reads a stream's `usage` promise WITHOUT ever hanging on it. * * Used only from the streaming error paths. `streamObject`/`streamText` * usually settle `usage` even when `object`/`text` reject — a schema-invalid * or aborted generation is still billed — so it is worth awaiting. But the * `result` promise those catches belong to has no outer deadline (unlike * `call()`, which runs under `runBounded`), so if the SDK ever rejected the * content promise while leaving `usage` pending, an unbounded await would turn * a prompt rejection into a caller that waits forever. Rejection is handled by * `.catch`; PENDENCY is handled by the race. The loser's timer is cleared, so * a settled call leaves no timer behind. * * @returns the usage object, or undefined if it rejected or did not settle in time */ private readUsageBounded; /** * Runs one provider call under a WATCHDOG and an ABSOLUTE DEADLINE. * * Why this exists (game e51493e4 r002, 2026-07-26): a plotter request was * accepted by the provider and never answered. Nothing in the stack could * interrupt it — no timeout was configured anywhere, the dump file is only * written when the call closes, and the callers' retry/fallback wrappers catch * ERRORS, not promises that never settle. The round froze for 639 seconds with * no log, no dump and no error, and only unfroze because the OpenAI SDK's own * 600s default finally fired. Three such stalls are on record (547s, 514s, * 639s) across two tiers and two models, so this is a property of talking to a * provider, not of one model. * * Two independent guarantees, because the first one can be ignored by an * adapter that owns its own transport: * - the WATCHDOG logs every `requestWatchdogMs` while the call is pending, so * a stall is visible AS IT HAPPENS instead of after it resolves; * - the DEADLINE aborts the signal and rejects with {@link LLMTimeoutError} * once the call has burned its whole attempt budget, so the promise ALWAYS * settles and the caller's existing retry/fallback path gets to run. * * The deadline is deliberately the LAST line of defence: it budgets * `requestDeadlineAttempts` attempts, so in normal operation the per-attempt * timeout fires first and the retry escalates the OpenRouter pin onto a * healthy provider — which is the outcome we actually want. Reaching the * deadline means the adapter never honoured its own timeout. */ private runBounded; /** * See the module-level {@link isTransientNetworkError} for the classification * rule and why it lives outside the class. Kept as a method because it is part * of this service's shape (spec harnesses stub and assert on it) — the * unqualified call below resolves to the module function, not to itself. */ private isTransientNetworkError; /** * Ordered failover candidates for a chat tier, newest health state applied. * * EMPTY means "no candidate machinery available" (a `ModelService` double * without `getCandidates`, or a resolution failure) — every caller then * behaves exactly as it did before failover existed. A resolution failure is * logged and swallowed rather than thrown: losing the chain must degrade to * the configured tier, never break the call. */ private resolveCandidates; /** * Puts the candidate that just failed transiently into its cooldown window, so * the next resolution skips it. Best-effort in every direction: no candidate, * no candidate-aware `ModelService`, or a throwing registry all leave the call * itself untouched — bookkeeping must never turn a retryable failure into a * hard one. */ private markCandidateFailure; /** * The cost rates to bill this call at, or undefined to keep the config-block * rates. * * ONLY a DB-backed connection supplies rates: an `.env` candidate is the very * config block `computeCost` already reads, so passing its numbers back in * would be a no-op at best and a rounding difference at worst. A DB connection * that prices nothing also returns undefined, so it bills at the tier rate * instead of silently costing zero. */ private ratesForCandidate; /** * Sleeps the jittered backoff for one transient retry and says so in the log. * ±20% jitter so a fleet of workers knocked out by the same DNS blip does not * come back in lockstep and knock it out again. */ private waitBeforeTransientRetry; /** * Runs a provider call under {@link runBounded}, retrying it when — and only * when — it failed for a transient network reason * ({@link isTransientNetworkError}). Two extra attempts by default, with the * long jittered waits {@link TRANSIENT_RETRY_WAITS_MS} explains. * * Owns a FRESH AbortController per attempt: a controller that has already * aborted stays aborted forever, so re-using one would make every retry abort * before it sent a byte. `work` therefore receives the signal rather than * capturing one from the caller. * * `work` also receives the ATTEMPT INDEX, which is what turns this retry into * a failover: `call()` uses it to pick the n-th connection of the chain, so * attempt 2 of a 429 storm talks to a different provider instead of knocking * on the same closed door. Callers that pass no `options` keep today's budget * (three attempts, the two configured waits) exactly. */ private runWithTransientRetry; /** * Records token usage for cost/observability attribution. Never throws — * a persistence failure logs a warning and the LLM call continues, so * observability problems can't break the primary request path. * * Writes through the application-provided `TOKEN_USAGE_RECORDER` when one is * bound, falling back to the module-local `TokenUsageService` otherwise. This * is the ONLY token-usage write inside the package; any future package caller * MUST use the same token rather than injecting `TokenUsageService` directly * (see the token's docblock for why). * * ZERO-TOKEN SUCCESS RULE: a call that SUCCEEDED but reported no usage at all * (the provider omitted `usage_metadata`) is still recorded — the call really * happened and must stay visible — but with `applyMinimum: false`, so it * costs 0 credits instead of being floored to `minCreditsPerRecord`. Flooring * exists to stop sub-cent REAL usage rounding to nothing, not to invent a * charge for tokens nobody measured. A success carrying real counts keeps the * floor exactly as before. * * This is deliberately NOT the same rule as the zero-token FAILURE rule in * {@link persistUsageOnFailure}, which writes nothing at all: there the * provider was never reached, so there is no call to make visible. */ private persistUsage; /** * Records what a FAILED call already burned. A failure is not a free call: * the provider bills every round it served, so a tool loop that dies on its * final structured invocation has already been charged for six figures of * input tokens. Billing only successful calls understates real spend. * * ZERO-TOKEN RULE: a failure that consumed nothing (the provider was never * reached — input validation, an unreachable host, an immediate abort) is NOT * recorded. `recordTokenUsage` floors every record at `minCreditsPerRecord`, * so writing a 0/0 row would invent a charge for tokens nobody spent; the * floor exists to stop sub-cent REAL usage rounding to nothing, not to price * a call that never happened. Such failures remain fully visible through the * dump session, which closes with `finalStatus: "error"`. * * Never throws (delegates to {@link persistUsage}), so it can sit in a catch * block without masking the original error. */ private persistUsageOnFailure; /** * Converts AgentMessageType to LangChain BaseMessage */ private _convertToBaseMessage; /** * Trims history to prevent context overflow */ private _trimHistory; /** * Auto-generates instructions from input parameters * * Formats parameters as "key: value" pairs separated by double newlines. * Handles primitives, objects, and arrays intelligently. * * IMPORTANT: For objects/arrays, curly braces are escaped with double braces * ({{ and }}) to prevent ChatPromptTemplate from treating them as template * variables. ChatPromptTemplate will render {{ as literal { in the final prompt. * * @param inputParams - Parameters to format * @returns Formatted instruction string with escaped braces, or empty string if no params */ private _autoGenerateInstructions; /** * Generates schema-guided instructions with inline descriptions * * This method enhances auto-generated instructions by including field descriptions * from the input schema. This provides the LLM with semantic context about each * input parameter, improving understanding and adherence to constraints. * * Benefits: * - LLM understands field purposes (e.g., "use likes to SUBTLY influence tone") * - LLM receives explicit constraints (e.g., "FORBIDDEN - never repeat these") * - Reduces need for redundant explanations in system prompts * - Single source of truth for input semantics * * Format: "fieldName (description): value" * * @param inputParams - Parameters to format (actual values) * @param inputSchema - Optional Zod schema with descriptions * @returns Formatted instruction string with inline descriptions * * @example Without schema (fallback to auto-generation): * ```typescript * _generateSchemaGuidedInstructions({ name: "Alice" }) * // Returns: "name: Alice" * ``` * * @example With schema (includes descriptions): * ```typescript * const schema = z.object({ * name: z.string().describe("The user's name"), * recentActions: z.array(z.string()).describe("FORBIDDEN - never repeat") * }); * _generateSchemaGuidedInstructions({ name: "Alice", recentActions: ["wave"] }, schema) * // Returns: * // "name (The user's name): Alice * // * // recentActions (FORBIDDEN - never repeat): ["wave"]" * ``` */ private _generateSchemaGuidedInstructions; /** * Creates message array for ChatPromptTemplate using MessagesPlaceholder pattern */ private _createMessages; /** * Calls the LLM with structured input/output using LangChain. * * This method: * 1. Builds a chat prompt from system prompts, history, and user instructions * 2. Auto-generates instructions from inputParams if not provided * 3. Trims history if maxHistoryMessages is specified (prevents context overflow) * 4. Substitutes {placeholders} in instructions with values from inputParams * 5. Calls the LLM with structured output enforcement (via function calling) * 6. Implements automatic retry logic with exponential backoff * 7. Returns the parsed response with token usage metadata * 8. Tracks session-level token usage * * @template T - The expected output type (inferred from outputSchema) * * @param params - Call parameters * @param params.inputParams - Variables to substitute in instruction template, or to auto-generate * Keys match {placeholders} in instructions (if provided) * Example: {character: {...}, userMessage: "Hello"} * @param params.inputSchema - Optional Zod schema for input validation and context injection * Field descriptions are extracted and included in prompts * @param params.outputSchema - Zod schema defining expected LLM response structure * @param params.systemPrompts - Array of system prompts to set context/behavior * @param params.instructions - Optional user instructions template with {placeholders} * If omitted, auto-generates from inputParams (with schema descriptions if provided) * Example: "Character: {character}\nUser says: {userMessage}" * @param params.temperature - Optional temperature override (0-2, default from config) * @param params.history - Optional conversation history as role/content pairs * @param params.maxHistoryMessages - Optional limit on history size (default: unlimited) * @param params.maxTokens - Optional max tokens for response * @param params.timeout - Optional timeout in milliseconds * @param params.metadata - Optional metadata for LangSmith tracking * @param params.stopSequences - Optional stop sequences * @param params.validateInput - Optional flag to enable input validation (default: false) * @param params.tools - Optional array of tools to bind to the LLM * @param params.maxToolIterations - Optional max tool call iterations (default: 5) * * @returns Promise resolving to parsed output + token usage metadata * @throws {Error} If LLM call fails or returns invalid structured output * * @example Simple case (auto-generated instructions): * ```typescript * const response = await llm.call({ * inputParams: { character: {...}, userMessage: "Hello" }, * outputSchema: z.object({ response: z.string() }), * systemPrompts: ["You are a helpful assistant"], * // No instructions - auto-generates: "character: {...}\n\nuserMessage: Hello" * }); * ``` * * @example Custom instructions with placeholders: * ```typescript * const response = await llm.call({ * inputParams: { * character: { name: "Zoe", description: "..." }, * userMessage: "Hello" * }, * outputSchema: z.object({ response: z.string() }), * systemPrompts: ["You are a helpful assistant"], * instructions: "Character: {character}\nUser says: {userMessage}\nRespond in character:", * temperature: 0.7, * maxHistoryMessages: 20, * metadata: { node_type: "character" }, * history: [ * { role: AgentMessageType.User, content: "Previous message" }, * { role: AgentMessageType.Assistant, content: "Previous response" } * ] * }); * ``` */ call(params: LLMCallParams): Promise; private _invokeOriginal; /** * Streaming variant of {@link call}. Same structured-input/structured-output * contract — uses `outputSchema` (Zod) for enforced structured output and * `inputSchema` for schema-guided instructions — but yields the LLM's * response progressively as it generates. * * Returns two handles: * - `textStream` — raw JSON-text fragments as the LLM builds the object. * Useful for forwarding to clients that incrementally parse JSON (e.g. * BlockNote AI's UIMessageStream consumer). MUST be consumed (even if * just to drain) for `result` to resolve — `streamObject` only commits * the final object after the source stream is fully read. * - `result` — Promise that resolves to the final fully-parsed structured * output + token usage, once the stream completes. Equivalent to what * `call` returns. * * Returns `partialObjectStream` (not `textStream`): each iteration yields * the cumulative best-effort parse of the output object as it grows. For a * schema like `z.object({ paragraph: z.string() })`, consumers see * `{paragraph: undefined}` → `{paragraph: "Jam"}` → `{paragraph: "James"}` * etc., letting them extract field-value deltas without parsing raw JSON * tokens themselves. * * Note: only one stream view is exposed because `streamObject`'s * `textStream` / `partialObjectStream` / `fullStream` getters all consume * the same underlying source — accessing two of them locks the source on * the first and throws on the second. For consumers that want raw JSON * text fragments instead, add a separate variant. * * Implementation note: uses Vercel AI SDK's `streamObject` under the hood * because LangChain's `withStructuredOutput().stream()` only yields parsed * partials — not the raw JSON text fragments that downstream UI-message * consumers (BlockNote AI, the AI SDK's own React hooks, etc.) need to * apply changes incrementally. `call` stays on LangChain for non-streaming * structured output — both paths share `_generateSchemaGuidedInstructions` * for input formatting and the `LLMCallDumper` session for cost tracking. * * Provider support: currently OpenAI-compatible only (llamacpp, openrouter, * requesty, plus any other provider exposed via an OpenAI-compatible URL). * Throws for vertex/azure with native (non-OpenAI-compat) configurations. * Extend via new `@ai-sdk/*` provider adapters when needed. */ streamCall>(params: LLMCallParams): Promise<{ partialObjectStream: AsyncIterable>; result: Promise; }>; /** * Plain-text streaming variant of {@link streamCall}. Unlike `streamCall` * (which enforces a Zod `outputSchema` via `streamObject` and therefore * requires the model/provider to support structured/JSON output), this method * streams free-form text via the Vercel AI SDK's `streamText`. * * Use this when: * - The desired output is just prose (no structured object), AND/OR * - The provider/model does not support JSON response formats * (e.g. Ollama with many local models), where `streamObject` fails with * `NoObjectGeneratedError` because the model returns prose, not JSON. * * Returns two handles: * - `fullStream` — normalized incremental parts as the model generates them: * `{ type: "text", delta }` for answer content and `{ type: "reasoning", * delta }` for a reasoning/"thinking" trace (emitted by reasoning-capable * models — e.g. Ollama surfaces `delta.reasoning` over its OpenAI-compatible * endpoint). Non-reasoning models simply never yield `reasoning` parts. * MUST be consumed for `result` to resolve. * - `result` — Promise resolving to the final concatenated answer `text`, the * full `reasoning` trace (empty string when the model emits none), and token * usage once the stream completes. * * Provider support mirrors `streamCall`: OpenAI-compatible only (llamacpp, * local, openrouter, requesty, ollama, plus any provider exposed via an * OpenAI-compatible URL). */ streamText(params: { systemPrompts: string[]; prompt: string; temperature?: number; maxTokens?: number; timeout?: number; modelWeight?: ModelWeight; metadata?: Record; tokenUsageType?: string; relationshipId?: string; relationshipType?: string; }): Promise<{ fullStream: AsyncIterable<{ type: "text" | "reasoning"; delta: string; }>; result: Promise<{ text: string; reasoning: string; tokenUsage: { input: number; output: number; }; modelWeight: ModelWeight; }>; }>; /** * Structured extraction via FORCED tool calling — the gemma/Ollama-reliable * counterpart to `streamCall`/`call()`'s `withStructuredOutput`, which fails on * models that don't support `response_format` json_schema. Forces the model to * call a single tool and returns its (Zod-validated) arguments. */ extractViaTool(params: { systemPrompts: string[]; prompt: string; tool: { name: string; description: string; schema: ZodType; }; modelWeight?: ModelWeight; metadata?: Record; tokenUsageType?: string; relationshipId?: string; relationshipType?: string; disableThinking?: boolean; /** Optional: how much hidden reasoning the model may spend. Overrides * `disableThinking` and the tier default. Unset = provider default. */ reasoningEffort?: ReasoningEffort; maxOutputTokens?: number; frequencyPenalty?: number; /** Sampling temperature. Defaults to getLLM's 0.2 (near-greedy, good for * deterministic extraction). Raise it for creative generation (e.g. game * creation) where greedy decoding collapses onto the model's prior names. */ temperature?: number; /** Opt-in Redis caching keyed on generic params (modelWeight/temperature/ * systemPrompts/prompt). A hit returns early WITHOUT invoking the provider, * so it costs no tokens. Default: false. */ cacheable?: boolean; /** Per-attempt request budget in ms. Defaults to `ai.requestTimeoutMs`. */ timeout?: number; }): Promise; /** * Single-step model invocation with tools bound — the durable-checkpointing * counterpart to {@link call}'s internal tool loop. Performs exactly ONE * model STEP (no tool execution, no loop, no structured output) and returns * the raw AIMessage with any `tool_calls` untouched, so the caller (e.g. the * operator agent) can checkpoint state and execute the tool calls itself. * * "One step" is not "one socket": the step is bounded like every other call * (per-attempt budget, watchdog, deadline) and re-issued on a transient * network failure. What it never does is re-run the AGENT — the caller's * checkpointed state is untouched either way. * * Reuses {@link call}'s model construction (`modelService.getLLM` + * `bindTools`) and `LLMCallDumper` hooks. The step's token usage is still * RETURNED to the caller (so an agent can keep its own running total), and — * when `relationshipId`/`relationshipType` are supplied — is now also * persisted here, exactly like every other provider call in this service. * A caller that omits the attribution gets the previous behaviour: nothing is * written. Each step is billed once, by whichever path completes it. * * @param params.systemPrompts - System prompts, prepended (in order) as * SystemMessages before `messages` * @param params.messages - Conversation messages sent verbatim to the model * @param params.tools - Tools to bind (NOT executed by this method) * @param params.temperature - Optional temperature override * @param params.metadata - Optional metadata for dump-session tracking * @param params.tokenUsageType - Optional usage type for the recorded row * @param params.relationshipId - Optional entity this usage is attributed to * @param params.relationshipType - Optional entity type for the attribution * * @returns The raw AIMessage (tool_calls intact) plus this call's token usage */ callStep(params: { systemPrompts: string[]; messages: BaseMessage[]; tools: DynamicStructuredTool[]; temperature?: number; metadata?: Record; tokenUsageType?: string; relationshipId?: string; relationshipType?: string; }): Promise<{ message: AIMessage; tokenUsage: { input: number; output: number; }; }>; } //# sourceMappingURL=llm.service.d.ts.map