/** * Type-safe MLLM (Multimodal Large Language Model) vendor classes. * * MLLM vendors handle real-time audio end-to-end, bypassing the standard * ASR → LLM → TTS pipeline. Calling `agent.withMllm(vendor)` automatically * sets `mllm.enable: true`. */ import type { MllmConfig, MllmTurnDetectionConfig } from "../types.js"; import { BaseCNMLLM, BaseMLLM, type BaseMllmOptions } from "./base.js"; export declare const GeminiLiveModels: { /** Public ID for the low-latency voice model. */ readonly Live38: "models/gemini-3.8-live"; /** Public ID for the reasoning voice model. */ readonly Live38ExtendedThinking: "models/gemini-3.8-live-extended-thinking"; }; /** The model Gemini Live sends by default. */ export declare const GEMINI_MLLM_DEFAULT_MODEL: string; /** Default Gemini Developer API endpoint for Gemini 3.8 models. */ export declare const GEMINI_MLLM_URL = "https://generativelanguage.googleapis.com"; /** @deprecated Use {@link GEMINI_MLLM_URL}. */ export declare const GEMINI_PREVIEW_MLLM_URL: string; /** A Gemini Live model. Known models autocomplete while future model IDs remain accepted. */ export type GeminiLiveModel = (typeof GeminiLiveModels)[keyof typeof GeminiLiveModels] | (string & {}); /** Reasoning budgets accepted by Gemini Extended Thinking. */ export declare const GeminiThinkingLevels: readonly ["low", "medium", "high"]; export type GeminiThinkingLevel = (typeof GeminiThinkingLevels)[number]; /** Known Gemini output voices; future voice names remain accepted. */ export type GeminiLiveVoice = "Puck" | "Charon" | "Kore" | "Fenrir" | "Aoede" | "Leda" | "Orus" | "Zephyr" | (string & {}); /** @deprecated Use {@link GeminiLiveVoice}. */ export type GeminiPreviewVoice = GeminiLiveVoice; export declare const OPENAI_GPT_LIVE_VENDOR: "openai_gpt_live"; export declare function isOpenAIGPTLiveConfig(config: unknown): boolean; /** GPT Live v3 options. */ export interface OpenAIGPTLiveOptions extends BaseMllmOptions { apiKey: string; /** Full WebSocket URL, used verbatim except OpenAI's legacy /v1/live route. */ url?: string; /** @deprecated Use prompt. Serialized as prompt; an explicit prompt wins. */ instructions?: string; greeting?: string; failureMessage?: string; inputModalities?: string[]; outputModalities?: string[]; messages?: Record[]; /** Additional provider fields; explicit options take precedence. */ params?: Record; /** @deprecated Unsupported in v3; setting this raises an error. */ inputAudioTranscription?: Record; /** @deprecated Ignored with a warning; v3 performs endpointing internally. */ turnDetection?: MllmTurnDetectionConfig; /** Defaults to gpt-live-1. */ model?: string; /** Output voice; provider default marin. Custom voice objects require PR #1522; use params after rollout. */ voice?: string; /** Session instructions. */ prompt?: string; /** Host when url is omitted; default wss://api.openai.com. */ baseUrl?: string; /** WebSocket path; default /v1/live/sessions. */ path?: string; /** Optional OpenAI-Alpha selector for preview contracts. Omitted by default. */ alphaSelector?: string; /** Extra provider request headers as a JSON string; protocol headers win. */ headers?: string; /** Assistant silence boundary in ms; provider default 600. Zero disables inference. */ outputIdleEndMs?: number; /** Caller silence boundary in ms; provider default 1500. */ inputIdleEndMs?: number; /** Speech amplitude threshold on the 16-bit scale; provider default 50. */ outputSilencePeak?: number; /** Graph PCM sample rate; provider default 24000. */ outputSampleRate?: number; /** Initial audio cushion; provider default 0. Negative disables pacing. */ outputBufferMs?: number; /** Mic append batching in ms. Join default 0; extension class default 100. */ inputBatchMs?: number; /** Advertise graph tools; provider default false. Does not control delegate built-ins. */ toolEnabled?: boolean; /** Tool delegation mode; provider default responses. Fixed for the session. */ delegation?: "client" | "responses"; /** Tool delegate model; provider default gpt-5.6-sol. */ responsesModel?: string; /** Interrupt playback on caller speech; provider default false. */ interruptOnUserTurn?: boolean; /** Unmodelled v3 session fields. Cannot override model, delegation, audio, instructions or input. */ sessionParams?: Record; } export declare class OpenAIGPTLive extends BaseMLLM { private readonly options; constructor(options: OpenAIGPTLiveOptions); toConfig(): MllmConfig; } /** * Constructor options for OpenAI Realtime API. */ export interface OpenAIRealtimeOptions extends BaseMllmOptions { /** OpenAI API key */ apiKey: string; /** Model name (e.g., 'gpt-4o-realtime-preview') */ model?: string; /** Voice identifier for audio output */ voice?: string; /** System instructions that define agent behavior */ instructions?: string; /** Audio transcription settings */ inputAudioTranscription?: Record; /** WebSocket URL. Defaults to `wss://api.openai.com/v1/realtime` when omitted or empty. */ url?: string; /** Agent greeting message */ greetingMessage?: string; /** Input modalities (e.g., ['audio'], ['audio', 'text']) */ inputModalities?: string[]; /** Output modalities (e.g., ['text', 'audio']) */ outputModalities?: string[]; /** Conversation messages for short-term memory */ messages?: Record[]; /** Additional MLLM parameters */ params?: Record; /** MLLM turn detection configuration. Overrides top-level turn_detection. */ turnDetection?: MllmTurnDetectionConfig; /** Message played on failure */ failureMessage?: string; } /** * OpenAI Realtime API MLLM vendor. * * @example * ```typescript * const client = new AgoraClient({ area: Area.US, appId: '...', appCertificate: '...' }); * const agent = new Agent({ client }) * .withMllm(new OpenAIRealtime({ * apiKey: process.env.OPENAI_API_KEY, * greetingMessage: 'Hello! How can I help you?', * })); * ``` */ export declare class OpenAIRealtime extends BaseMLLM { private readonly options; constructor(options: OpenAIRealtimeOptions); toConfig(): MllmConfig; } /** Parameters accepted by Azure OpenAI Realtime. */ export interface AzureOpenAIRealtimeParams { /** System instructions that define agent behavior */ instructions?: string; /** Model or deployment model identifier */ model?: string; /** Voice identifier for audio output */ voice?: string; } /** Constructor options for Azure OpenAI Realtime API. */ export interface AzureOpenAIRealtimeOptions extends BaseMllmOptions { /** Azure OpenAI API key */ apiKey: string; /** Azure OpenAI Realtime WebSocket URL, including deployment routing when required */ url: string; /** Model or deployment model identifier */ model?: string; /** Voice identifier for audio output */ voice?: string; /** System instructions that define agent behavior */ instructions?: string; /** Number of conversation history messages cached by the MLLM */ maxHistory?: number; /** Agent greeting message */ greetingMessage?: string; /** Output modalities (e.g., ['text', 'audio']) */ outputModalities?: string[]; /** Conversation messages for short-term memory */ messages?: Record[]; /** Azure Realtime model parameters */ params?: AzureOpenAIRealtimeParams; /** Required MLLM turn detection configuration. Overrides top-level turn_detection. */ turnDetection: MllmTurnDetectionConfig; } /** * Azure OpenAI Realtime MLLM vendor for global deployments. * * @example * ```typescript * const agent = new Agent({ client }).withMllm(new AzureOpenAIRealtime({ * apiKey: process.env.AZURE_OPENAI_API_KEY, * url: 'wss://example.openai.azure.com/openai/realtime', * model: 'gpt-4o-realtime-preview', * maxHistory: 32, * turnDetection: { mode: 'server_vad' }, * })); * ``` */ export declare class AzureOpenAIRealtime extends BaseMLLM { private readonly options; constructor(options: AzureOpenAIRealtimeOptions); toConfig(): MllmConfig; } /** * Constructor options for Google Gemini Live (direct API, non-Vertex AI). */ export interface GeminiLiveOptions extends BaseMllmOptions { /** Google API key */ apiKey: string; /** Model name (e.g., 'gemini-live-2.5-flash') */ model?: GeminiLiveModel; /** Sent only for models/gemini-3.8-live-extended-thinking. */ thinkingLevel?: GeminiThinkingLevel; /** Languages for Gemini 3.8, sent as params.language_codes. */ languageCodes?: readonly string[]; /** Endpoint override; Gemini 3.8 defaults to the Developer API host. */ url?: string; /** System instructions for the model */ instructions?: string; /** Voice name (e.g., 'Aoede', 'Charon') */ voice?: GeminiLiveVoice; affectiveDialog?: boolean; proactiveAudio?: boolean; transcribeAgent?: boolean; transcribeUser?: boolean; httpOptions?: Record; /** Agent greeting message */ greetingMessage?: string; /** Input modalities (e.g., ['audio'], ['audio', 'text']) */ inputModalities?: string[]; /** Output modalities (e.g., ['text', 'audio']) */ outputModalities?: string[]; /** Conversation messages for short-term memory */ messages?: Record[]; /** Additional MLLM parameters passed directly to the model */ additionalParams?: Record; /** MLLM turn detection configuration. Overrides top-level turn_detection. */ turnDetection?: MllmTurnDetectionConfig; /** Message played on failure */ failureMessage?: string; } /** * Google Gemini Live MLLM vendor (direct API, non-Vertex AI). * * Uses a Google API key. For Vertex AI / ADC credentials use {@link VertexAI} instead. * * @example * ```typescript * const client = new AgoraClient({ area: Area.US, appId: '...', appCertificate: '...' }); * const agent = new Agent({ client }) * .withMllm(new GeminiLive({ * apiKey: process.env.GOOGLE_API_KEY, * model: 'gemini-live-2.5-flash', * greetingMessage: 'Hello! Gemini is listening.', * })); * ``` */ export declare class GeminiLive extends BaseMLLM { private readonly options; constructor(options: GeminiLiveOptions); toConfig(): MllmConfig; } /** * Constructor options for Google Gemini Live (Vertex AI). */ export interface VertexAIOptions extends BaseMllmOptions { /** Model name (e.g., 'gemini-live-2.5-flash-preview-native-audio-09-2025') */ model: string; /** WebSocket URL for real-time communication */ url?: string; /** Google Cloud project ID */ projectId: string; /** Google Cloud location/region */ location: string; /** Application Default Credentials JSON string */ adcCredentialsString: string; /** System instructions for the model */ instructions?: string; /** Voice name (e.g., 'Aoede', 'Charon') */ voice?: string; affectiveDialog?: boolean; proactiveAudio?: boolean; transcribeAgent?: boolean; transcribeUser?: boolean; httpOptions?: Record; /** Agent greeting message */ greetingMessage?: string; /** Input modalities (e.g., ['audio'], ['audio', 'text']) */ inputModalities?: string[]; /** Output modalities (e.g., ['text', 'audio']) */ outputModalities?: string[]; /** Conversation messages for short-term memory */ messages?: Record[]; /** Additional MLLM parameters */ additionalParams?: Record; /** MLLM turn detection configuration. Overrides top-level turn_detection. */ turnDetection?: MllmTurnDetectionConfig; /** Message played on failure */ failureMessage?: string; } /** * Google Gemini Live (Vertex AI) MLLM vendor. * * @example * ```typescript * const client = new AgoraClient({ area: Area.US, appId: '...', appCertificate: '...' }); * const agent = new Agent({ client }) * .withMllm(new VertexAI({ * model: 'gemini-live-2.5-flash-preview-native-audio-09-2025', * projectId: process.env.GOOGLE_PROJECT_ID, * location: 'us-central1', * adcCredentialsString: process.env.GOOGLE_ADC_CREDENTIALS, * instructions: 'You are a helpful voice assistant.', * voice: 'Aoede', * greetingMessage: 'Hello! Gemini is listening.', * })); * ``` */ export declare class VertexAI extends BaseMLLM { private readonly options; constructor(options: VertexAIOptions); toConfig(): MllmConfig; } /** * Constructor options for xAI Grok Realtime API. */ export interface XaiGrokOptions extends BaseMllmOptions { /** xAI API key */ apiKey: string; /** WebSocket URL for real-time communication (defaults to xAI Realtime API) */ url?: string; /** Voice identifier (e.g., 'eve', 'rex') */ voice?: string; /** Language code (e.g., 'en') */ language?: string; /** Audio sample rate in Hz (e.g., 24000) */ sampleRate?: number; /** Agent greeting message */ greetingMessage?: string; /** Message played on failure */ failureMessage?: string; /** Input modalities (e.g., ['audio'], ['audio', 'text']) */ inputModalities?: string[]; /** Output modalities (e.g., ['audio'], ['text', 'audio']) */ outputModalities?: string[]; /** Conversation messages for short-term memory */ messages?: Record[]; /** Additional MLLM parameters passed directly to xAI */ params?: Record; /** MLLM turn detection configuration. Overrides top-level turn_detection. */ turnDetection?: MllmTurnDetectionConfig; } /** * xAI Grok MLLM vendor (`mllm.vendor`: `"xai"`). * * Uses the xAI Realtime API WebSocket URL by default. Do not name future xAI ASR/TTS * wrappers `XaiRealtime`; use `XaiSTT` / `XaiTTS` when those pipelines are added. * * @example * ```typescript * const client = new AgoraClient({ area: Area.US, appId: '...', appCertificate: '...' }); * const agent = new Agent({ client }) * .withMllm(new XaiGrok({ * apiKey: process.env.XAI_API_KEY, * voice: 'eve', * language: 'en', * sampleRate: 24000, * greetingMessage: 'Hello, how can I help?', * })); * ``` */ export declare class XaiGrok extends BaseMLLM { private readonly options; constructor(options: XaiGrokOptions); toConfig(): MllmConfig; } /** Constructor options for Alibaba Cloud Qwen Omni Realtime. */ export interface QwenOmniOptions extends BaseMllmOptions { /** Alibaba Cloud DashScope API key */ apiKey: string; /** Qwen Omni model identifier */ model: string; /** Qwen Omni Realtime WebSocket URL */ url: string; /** Voice identifier for audio output */ voice?: string; /** System instructions that define agent behavior */ instructions?: string; /** Agent greeting message */ greetingMessage?: string; /** Message played on failure */ failureMessage?: string; /** Input modalities (e.g., ['audio'], ['audio', 'text']) */ inputModalities?: string[]; /** Output modalities (e.g., ['text', 'audio']) */ outputModalities?: string[]; /** Conversation messages for short-term memory */ messages?: Record[]; /** Additional Qwen Omni parameters */ params?: Record; /** MLLM turn detection configuration. Overrides top-level turn_detection. */ turnDetection?: MllmTurnDetectionConfig; } /** * Alibaba Cloud Qwen Omni Realtime MLLM vendor for Chinese mainland deployments. * * @example * ```typescript * const agent = new Agent({ client }).withMllm(new QwenOmni({ * apiKey: process.env.DASHSCOPE_API_KEY, * model: 'qwen3.5-omni-plus-realtime', * url: 'wss://dashscope.aliyuncs.com/api-ws/v1/realtime', * greetingMessage: '你好,有什么可以帮你?', * })); * ``` */ export declare class QwenOmni extends BaseCNMLLM { private readonly options; constructor(options: QwenOmniOptions); toConfig(): MllmConfig; }