/** * Agent class - A reusable agent definition. * * This class represents an agent configuration that can be used to create * multiple sessions. Set the unique agent instance `name` in * {@link Agent.createSession}, not on the Agent itself. */ import type { AgoraClient } from "../AgoraPoolClient.js"; import type * as Agora from "../api/index.js"; import { AgentSession } from "./AgentSession.js"; import type { AgoraArea } from "./area.js"; import type { AvatarVendor, CNMllmVendor, GlobalMllmVendor, LlmVendor, SttVendor, TtsVendor } from "./region-vendors.js"; import type { AdvancedFeatures, AvatarConfig, FillerWordsConfig, GeofenceConfig, InterruptionConfig, Labels, LlmConfig, LlmGreetingConfigs, MllmConfig, ParametersAudioScenario, RtcConfig, SalConfig, SessionOptions, SessionParamsInput, SttConfig, TtsConfig, TurnDetectionConfig } from "./types.js"; /** * Configuration options for creating an Agent. * * Use the fluent builder methods (.withLlm(), .withTts(), .withStt(), .withMllm()) * to configure vendor settings after construction. */ export interface AgentOptions { /** Agora client bound to this agent. Required for `createSession()`. */ client: AgoraClient; /** * Published AI Studio pipeline ID to use as this agent's base configuration. * Explicit Agent config such as .withLlm(), .withTts(), .withStt(), * advancedFeatures, and other builder options may send fields in * `properties` that override the saved pipeline settings. */ pipelineId?: string; /** * System instructions for the agent. * @deprecated Configure this on the LLM vendor with `systemMessages` instead. */ instructions?: string; /** Turn detection configuration */ turnDetection?: TurnDetectionConfig; /** Unified interruption control configuration */ interruption?: InterruptionConfig; /** SAL configuration */ sal?: SalConfig; /** Avatar configuration */ avatar?: AvatarConfig; /** Advanced features */ advancedFeatures?: AdvancedFeatures; /** Session parameters */ parameters?: SessionParamsInput; /** * Greeting message. * @deprecated Configure this on the LLM or MLLM vendor with `greetingMessage` instead. */ greeting?: string; /** * Failure message. * @deprecated Configure this on the LLM or MLLM vendor with `failureMessage` instead. */ failureMessage?: string; /** * Max conversation history for the standard LLM pipeline. Does not apply to MLLM. * @deprecated Configure this on the LLM vendor with `maxHistory` instead. */ maxHistory?: number; /** Regional access restriction configuration */ geofence?: GeofenceConfig; /** Custom key-value labels attached to the agent (returned in notification callbacks) */ labels?: Labels; /** RTC media encryption configuration */ rtc?: RtcConfig; /** Filler word configuration (plays filler words while waiting for LLM responses) */ fillerWords?: FillerWordsConfig; /** * Greeting playback configuration for multi-user channels. * @deprecated Configure this on the LLM vendor with `greetingConfigs` instead. */ greetingConfigs?: LlmGreetingConfigs; } /** * Agent class representing a reusable agent configuration. * * @template TTSSampleRate - The TTS sample rate literal type (tracked for avatar compatibility) * * @example * ```typescript * import { Agent, OpenAI, MicrosoftTTS, DeepgramSTT } from 'agora-agents'; * * // Use the fluent builder pattern to configure vendors * const agent = new Agent({ client, instructions: 'You are helpful.' }) * .withLlm(new OpenAI({ apiKey: '...', model: 'gpt-4', url: 'https://api.openai.com/v1/chat/completions' })) * .withTts(new ElevenLabsTTS({ key: '...', modelId: '...', voiceId: '...', baseUrl: 'wss://api.elevenlabs.io/v1', sampleRate: 24000 })) * .withStt(new DeepgramSTT({ apiKey: '...', model: 'nova-2' })); * * const session = agent.createSession({ * name: `conversation-${Date.now()}`, * channel: `demo-channel-${Date.now()}`, * agentUid: '1', * remoteUids: ['100'], * }); * ``` */ export declare class Agent { private _client; private _pipelineId?; private _llm?; private _tts?; private _stt?; private _mllm?; private _turnDetection?; private _interruption?; private _sal?; private _avatar?; private _advancedFeatures?; private _parameters?; private _instructions?; private _greeting?; private _failureMessage?; private _maxHistory?; private _geofence?; private _labels?; private _rtc?; private _fillerWords?; private _greetingConfigs?; constructor(options: AgentOptions); /** * Returns a new Agent with the specified LLM vendor. * * @param vendor - LLM vendor instance (e.g., new OpenAI({ apiKey: '...', model: 'gpt-4', url: 'https://api.openai.com/v1/chat/completions' })) */ withLlm(vendor: LlmVendor): Agent; /** * Returns a new Agent with the specified TTS vendor. * * The sample rate type is tracked for compile-time avatar compatibility checking. * * @template SR - Sample rate literal type * @param vendor - TTS vendor instance (e.g., new ElevenLabsTTS({ key: '...', modelId: '...', voiceId: '...', baseUrl: 'wss://api.elevenlabs.io/v1', sampleRate: 24000 })) * @returns Agent with tracked sample rate type */ withTts(vendor: TtsVendor): Agent; /** * Returns a new Agent with the specified STT vendor. * * @param vendor - STT vendor instance (e.g., new SpeechmaticsSTT({ key: '...', language: 'en' })) * * @example * ```typescript * import { SpeechmaticsSTT } from 'agora-agents'; * * agent.withStt(new SpeechmaticsSTT({ * key: 'your-key', * language: 'en', * })); * ``` */ withStt(vendor: SttVendor): Agent; /** * Returns a new Agent with the specified MLLM vendor. * * MLLM vendors handle real-time audio end-to-end, bypassing the standard * ASR → LLM → TTS pipeline. Calling this method automatically sets * `mllm.enable: true`, so `withLlm()`, `withTts()`, and `withStt()` * are not needed. * * Note: avatars are not supported with MLLM. The avatar publisher requires * the cascading ASR + LLM + TTS pipeline. Combining `withMllm()` and * `withAvatar()` throws at `toProperties()` / `session.start()`. * * @param vendor - MLLM vendor instance (e.g., new VertexAI({ model: '...', projectId: '...', ... })) */ withMllm(vendor: GlobalMllmVendor | CNMllmVendor): Agent; /** * Returns a new Agent with the specified Avatar vendor. * * ⚠️ IMPORTANT: avatars are only supported with the cascading * ASR + LLM + TTS pipeline. They are not supported with MLLM * (`withMllm()`); combining the two throws at `toProperties()` / * `session.start()`. * * Different avatar vendors require specific TTS sample rates: * - HeyGen / LiveAvatar: Requires 24,000 Hz (24kHz) * - Akool: Requires 16,000 Hz (16kHz) * * This method enforces sample rate compatibility at compile time. If you configure * a TTS with 16kHz and try to add a LiveAvatar avatar (which needs 24kHz), TypeScript * will show a compile error. * * @template RequiredSR - Required sample rate for the avatar * @param vendor - Avatar vendor instance (e.g., new HeyGenAvatar({ apiKey: '...', quality: 'high', ... })) * * @example * ```typescript * import { HeyGenAvatar, ElevenLabsTTS } from 'agora-agents'; * * const client = new AgoraClient({ area: Area.US, appId: '...', appCertificate: '...' }); * const agent = new Agent({ client }) * .withTts(new ElevenLabsTTS({ * key: '...', * modelId: '...', * voiceId: '...', * baseUrl: 'wss://api.elevenlabs.io/v1', * sampleRate: 24000, // Required for HeyGen * })) * .withAvatar(new HeyGenAvatar({ * apiKey: '...', * quality: 'high', * agoraUid: '12345', * })); * ``` */ withAvatar(this: Agent, vendor: AvatarVendor): Agent; /** * Returns a new Agent with the specified turn detection configuration. */ withTurnDetection(config: TurnDetectionConfig): Agent; /** * Returns a new Agent with unified interruption control configured. */ withInterruption(config: InterruptionConfig): Agent; /** * Returns a new Agent with the specified instructions. * * @deprecated Configure system messages on the LLM vendor instead. */ withInstructions(instructions: string): Agent; /** * Returns a new Agent with the specified greeting message. * * @deprecated Configure the greeting on the LLM or MLLM vendor instead. */ withGreeting(greeting: string): Agent; /** * Returns a new Agent with the specified greeting playback configuration. * * Serializes to `llm.greeting_configs`. Agent-level values override any * vendor-level `greeting_configs` configured on the LLM vendor. * * @deprecated Configure greeting playback on the LLM vendor instead. */ withGreetingConfigs(configs: LlmGreetingConfigs): Agent; /** * Returns a new Agent with the specified SAL (Selective Attention Locking) configuration. */ withSal(config: SalConfig): Agent; /** * Returns a new Agent with the specified advanced features configuration. * * Use this to enable features like RTM and others. */ withAdvancedFeatures(features: AdvancedFeatures): Agent; /** * Returns a new Agent with MCP and inline REST tool invocation enabled or disabled. */ withTools(enabled?: boolean): Agent; /** * Returns a new Agent with the specified session parameters. * * Use this to configure silence behaviour, graceful hang-up, data channel, and more. */ withParameters(parameters: SessionParamsInput): Agent; /** * Returns a new Agent with the specified RTC audio scenario. */ withAudioScenario(audioScenario: ParametersAudioScenario): Agent; /** * Returns a new Agent with the specified failure message. * * The failure message is played via TTS when the LLM call fails. * * @deprecated Configure the failure message on the LLM or MLLM vendor instead. */ withFailureMessage(message: string): Agent; /** * Returns a new Agent with the specified maximum conversation history length. * Applies to the standard LLM pipeline only. For Azure OpenAI Realtime MLLM, * configure `maxHistory` on the MLLM vendor instead. * * @deprecated Configure max history on the LLM vendor instead. */ withMaxHistory(maxHistory: number): Agent; /** * Returns a new Agent with the specified geofence configuration. * * Restricts which geographic regions the agent's backend servers may run in. */ withGeofence(geofence: GeofenceConfig): Agent; /** * Returns a new Agent with the specified custom labels. * * Labels are key-value pairs attached to the agent and returned in notification callbacks. */ withLabels(labels: Labels): Agent; /** * Returns a new Agent with the specified RTC configuration. */ withRtc(rtc: RtcConfig): Agent; /** * Returns a new Agent with the specified filler words configuration. * * Filler words are played while the agent waits for the LLM to respond. */ withFillerWords(fillerWords: FillerWordsConfig): Agent; /** * Get the AI Studio pipeline ID used as this agent's base configuration. */ get pipelineId(): string | undefined; /** * Get the LLM configuration. */ get llm(): LlmConfig | undefined; /** * Get the TTS configuration. */ get tts(): TtsConfig | undefined; /** * Get the STT configuration. */ get stt(): SttConfig | undefined; /** * Get the MLLM configuration. */ get mllm(): MllmConfig | undefined; /** * Get the turn detection configuration. */ get turnDetection(): TurnDetectionConfig | undefined; /** * Get the interruption configuration. */ get interruption(): InterruptionConfig | undefined; /** * Get the instructions. */ get instructions(): string | undefined; /** * Get the greeting message. */ get greeting(): string | undefined; /** * Get the greeting playback configuration. */ get greetingConfigs(): LlmGreetingConfigs | undefined; /** * Get the failure message (played via TTS when the LLM call fails). */ get failureMessage(): string | undefined; /** * Get the maximum conversation history length. */ get maxHistory(): number | undefined; /** * Get the avatar configuration. */ get avatar(): AvatarConfig | undefined; /** * Get the SAL (Selective Attention Locking) configuration. */ get sal(): SalConfig | undefined; /** * Get the advanced features configuration. */ get advancedFeatures(): AdvancedFeatures | undefined; /** * Get the session parameters configuration. */ get parameters(): SessionParamsInput | undefined; /** * Get the geofence configuration. */ get geofence(): GeofenceConfig | undefined; /** * Get the custom labels. */ get labels(): Labels | undefined; /** * Get the RTC configuration. */ get rtc(): RtcConfig | undefined; /** * Get the filler words configuration. */ get fillerWords(): FillerWordsConfig | undefined; /** * Get the full agent configuration as an object. * This provides read-only access to the complete configuration, * including all vendor configs set via builder methods. */ get config(): AgentOptions & { llm?: LlmConfig; tts?: TtsConfig; stt?: SttConfig; mllm?: MllmConfig; }; /** * Creates a new session from this agent configuration. * * @param options - Session connection options * @returns A new AgentSession instance ready to start * * @example * ```typescript * const client = new AgoraClient({ area: Area.US, appId: '...', appCertificate: '...' }); * const agent = new Agent({ client }) * .withLlm(new OpenAI({ apiKey: '...', model: 'gpt-4o-mini', url: 'https://api.openai.com/v1/chat/completions' })) * .withTts(new MiniMaxTTS({ model: 'speech_2_6_turbo', voiceId: 'English_captivating_female1' })); * * const session = agent.createSession({ * name: `conversation-${Date.now()}`, * channel: `demo-channel-${Date.now()}`, * agentUid: '1', * remoteUids: ['100'], * idleTimeout: 120, * }); * * const agentId = await session.start(); * ``` */ createSession(options: SessionOptions): AgentSession; /** * Converts the Agent configuration to the Fern request properties format. * * Pass either a pre-built `token` OR `appId` + `appCertificate` to have * the SDK generate one automatically. The generated token includes both RTC * and RTM privileges (required for RTM-enabled sessions). */ toProperties(opts: { channel: string; agentUid: string; remoteUids: string[]; idleTimeout?: number; enableStringUid?: boolean; /** * @deprecated Use `skipVendorValidationCategories` and * `allowMissingVendorCategories` instead. This broad escape hatch will be * removed in a future release. */ skipVendorValidation?: boolean; /** Skip generated request-shape validation for the listed provider categories. */ skipVendorValidationCategories?: ReadonlySet<"asr" | "llm" | "tts">; /** Allow the listed provider categories to be omitted from properties. */ allowMissingVendorCategories?: ReadonlySet<"asr" | "llm" | "tts">; } & ({ token: string; appId?: undefined; appCertificate?: undefined; } | { token?: undefined; appId: string; appCertificate: string; expiresIn?: number; })): Agora.StartAgentsRequest.Properties; /** * Creates a shallow copy of this Agent, preserving the TTSSampleRate type * parameter. Builder methods that do not change the sample rate can use the * return value directly. `withTts()` must cast to `Agent` afterward * since it changes the type parameter — the cast is safe because all fields * are copied before the new TTS config is assigned. * * If a new private field is added to Agent, it MUST also be added here. */ private _clone; private _resolveAsrConfig; private _resolveTurnDetectionConfig; }