import { isThinkingByName } from './model-capabilities.js'; export interface LLMConfig { apiUrl: string; apiKey: string; defaultModel: string; timeout?: number; /** * Default `reasoning_effort` applied to detected reasoning models (see * isReasoningModel). Defaults to 'low' — the level that reliably keeps gpt-oss * from spending its whole budget in the thinking channel. From LLM_REASONING_EFFORT. */ reasoningEffort?: ReasoningEffort; } export interface ChatMessage { role: 'system' | 'user' | 'assistant'; content: string; /** * Some OpenAI-compatible endpoints (notably Ollama Cloud for reasoning * models like gpt-oss, kimi-thinking, deepseek-r, and the OpenAI o-series * itself) return the chain-of-thought separately from the final content. * ClarifyPrompt never returns this as the optimized prompt — it's thinking, * not answer — but it's useful for diagnostics when `content` is empty. * * Providers disagree on the field name. We read all three: * - `reasoning` — legacy / several OpenAI-compatible gateways * - `thinking` — Ollama (gpt-oss, qwen3-thinking, …) * - `reasoning_content` — DeepSeek and some gateways */ reasoning?: string; thinking?: string; reasoning_content?: string; } /** * Recover the assistant's final answer and its (optional) chain-of-thought from * a completion message, tolerant of the three field names providers use for the * thinking channel. The thinking trace is NEVER returned as content — it's * diagnostics only (see the gpt-oss empty-content failure mode, issue #3). */ export declare function extractAssistantContent(message: ChatMessage | undefined): { content: string; reasoning: string; }; export type ReasoningEffort = 'low' | 'medium' | 'high'; export declare const isReasoningModel: typeof isThinkingByName; export interface ChatCompletionRequest { model: string; messages: ChatMessage[]; stream?: boolean; temperature?: number; max_tokens?: number; /** * Optional caller cancellation signal (1.10.0). Combined with the per-call * timeout so a client cancel (MCP `notifications/cancelled` → the tool * handler's `extra.signal`) aborts the in-flight HTTP request immediately, * not just on timeout. Model-agnostic: the signal reaches `fetch` regardless * of provider or model. */ signal?: AbortSignal; /** * OpenAI `reasoning_effort` ('low' | 'medium' | 'high'). When omitted, the * client auto-applies its configured default to detected reasoning models * (see isReasoningModel) and sends nothing for everyone else. */ reasoning_effort?: ReasoningEffort; } export interface ChatCompletionResponse { id: string; model: string; choices: Array<{ index: number; message: ChatMessage; finish_reason: string; }>; usage: { prompt_tokens: number; completion_tokens: number; total_tokens: number; }; } export interface StreamChunk { id: string; model: string; choices: Array<{ index: number; delta: Partial; finish_reason: string | null; }>; } export declare class LLMClient { private config; private isAnthropic; constructor(config?: Partial); getModelName(): string; /** * The `reasoning_effort` to send: an explicit per-request value wins; otherwise * the configured default is applied ONLY to detected reasoning models, so the * common (non-reasoning) path stays byte-identical. Returns undefined when * nothing should be sent. */ private resolveReasoningEffort; /** Build the OpenAI-compatible chat body, adding reasoning_effort when applicable. */ private buildOpenAIBody; /** * Combine the per-call timeout with an optional caller cancellation signal so * `fetch` aborts on whichever fires first. `AbortSignal.any` is available on * Node ≥18.17 (our floor is >=18; CI covers 18/20/22/24). */ private withTimeout; chat(request: Omit & { model?: string; }): Promise; private chatOpenAI; private chatAnthropic; chatStream(request: Omit & { model?: string; }): AsyncGenerator; simpleGenerate(systemPrompt: string, userPrompt: string, options?: { model?: string; temperature?: number; maxTokens?: number; reasoningEffort?: ReasoningEffort; signal?: AbortSignal; }): Promise<{ content: string; tokensUsed: number; }>; private warnedReasoningEmptyContent; } export declare class LLMError extends Error { statusCode: number; details: string; constructor(message: string, statusCode: number, details: string); } export declare function getLLMClient(): LLMClient;