/** * Type-safe TTS (Text-to-Speech) vendor classes. */ import { CredentialMode } from "../constants.mjs"; import { type MiniMaxPresetModel, type OpenAITtsPresetModel } from "../presets.mjs"; import type { TtsConfig } from "../types.mjs"; import type { CartesiaSampleRate, ElevenLabsSampleRate, GoogleTTSSampleRate, MicrosoftSampleRate } from "./base.mjs"; import { BaseTTS } from "./base.mjs"; /** * Constructor options for ElevenLabs TTS. */ export interface ElevenLabsTTSOptions { /** ElevenLabs API key */ key: string; /** Model ID (e.g., 'eleven_flash_v2_5', 'eleven_monolingual_v1') */ modelId: string; /** Voice ID */ voiceId: string; /** WebSocket base URL */ baseUrl: string; /** * Audio sample rate in Hz. * - 16000 Hz: Required for Akool avatars * - 24000 Hz: Required for HeyGen avatars * - 22050, 44100 Hz: High quality, no avatar support */ sampleRate?: SR; /** Optimize streaming latency (0-4, higher = lower latency but lower quality) */ optimizeStreamingLatency?: number; /** Voice stability (0.0-1.0) */ stability?: number; /** Voice similarity boost (0.0-1.0) */ similarityBoost?: number; /** Voice style (0.0-1.0) */ style?: number; /** Enable speaker boost */ useSpeakerBoost?: boolean; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * ElevenLabs TTS vendor. * * @example * ```typescript * const tts = new ElevenLabsTTS({ * key: process.env.ELEVENLABS_API_KEY, * modelId: 'eleven_flash_v2_5', * voiceId: 'pNInz6obpgDQGcFmaJgB', * baseUrl: 'wss://api.elevenlabs.io/v1', * sampleRate: 24000, // For HeyGen avatar * }); * ``` */ export declare class ElevenLabsTTS extends BaseTTS { private readonly options; constructor(options: ElevenLabsTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Microsoft Azure TTS. */ export interface MicrosoftTTSOptions { /** Microsoft Azure API key */ key: string; /** Azure region (e.g., 'eastus', 'westus') */ region: string; /** Voice name (e.g., 'en-US-AndrewMultilingualNeural') */ voiceName: string; /** * Audio sample rate in Hz. * Supported values: 16000, 24000, 48000 */ sampleRate?: SR; /** Skip patterns for bracketed content */ skipPatterns?: number[]; /** Speaking rate multiplier. Values between 0.5 and 2.0. */ speed?: number; /** Audio volume. Values between 0.0 and 100.0. */ volume?: number; } /** * Microsoft Azure TTS vendor. * * @example * ```typescript * const tts = new MicrosoftTTS({ * key: process.env.AZURE_SPEECH_KEY, * region: 'eastus', * voiceName: 'en-US-JennyNeural', * sampleRate: 24000, * }); * ``` */ export declare class MicrosoftTTS extends BaseTTS { private readonly options; constructor(options: MicrosoftTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for OpenAI TTS. */ type OpenAITTSCommonOptions = { /** OpenAI API key. Optional only for the Agora-managed `tts-1` path. */ apiKey?: string; /** Voice name (e.g., 'alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer') */ voice: string; /** Model name (e.g., 'tts-1', 'tts-1-hd') */ model?: string; /** Endpoint URL for the OpenAI TTS service */ baseUrl?: string; /** Custom instructions for voice style, accent, pace, and tone */ instructions?: string; /** Speech speed multiplier */ speed?: number; /** Skip patterns for bracketed content */ skipPatterns?: number[]; }; export type OpenAITTSOptions = (OpenAITTSCommonOptions & { apiKey: string; model: string; baseUrl: string; }) | (Omit & { apiKey?: undefined; model?: OpenAITtsPresetModel; baseUrl?: undefined; }); /** * OpenAI TTS vendor. * * Note: OpenAI TTS is fixed at 24kHz and does not support changing the sample rate. * * @example * ```typescript * const tts = new OpenAITTS({ * apiKey: process.env.OPENAI_API_KEY, * model: 'gpt-4o-mini-tts', * baseUrl: 'https://api.openai.com/v1', * voice: 'alloy', * }); * ``` */ export declare class OpenAITTS extends BaseTTS<24000> { private readonly options; constructor(options: OpenAITTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Cartesia TTS. */ export interface CartesiaTTSOptions { /** Cartesia API key */ apiKey: string; /** Voice ID */ voiceId: string; /** Model ID */ modelId: string; /** WebSocket URL for the Cartesia streaming API */ baseUrl?: string; /** Target language for speech synthesis */ language?: string; /** * Audio sample rate in Hz. * Supported values: 8000, 16000, 22050, 24000, 44100, 48000 */ sampleRate?: SR; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Cartesia TTS vendor. * * @example * ```typescript * const tts = new CartesiaTTS({ * apiKey: process.env.CARTESIA_API_KEY, * voiceId: 'voice-id-here', * sampleRate: 24000, * }); * ``` */ export declare class CartesiaTTS extends BaseTTS { private readonly options; constructor(options: CartesiaTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Google TTS. */ export interface GoogleTTSOptions { /** Google Cloud service account credentials JSON string */ key: string; /** Voice name */ voiceName: string; /** Language code (e.g., 'en-US') */ languageCode?: string; /** * Audio sample rate in Hz. * Supported values: 8000, 16000, 22050, 24000, 44100, 48000 */ sampleRate?: SR; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Google TTS vendor. * * @example * ```typescript * const tts = new GoogleTTS({ * key: process.env.GOOGLE_API_KEY, * voiceName: 'en-US-Wavenet-D', * sampleRate: 24000, * }); * ``` */ export declare class GoogleTTS extends BaseTTS { private readonly options; constructor(options: GoogleTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Amazon Polly TTS. */ export interface AmazonTTSOptions { /** AWS access key */ accessKey: string; /** AWS secret key */ secretKey: string; /** AWS region (e.g., 'us-east-1') */ region: string; /** Amazon Polly voice ID */ voiceId: string; /** Amazon Polly engine type */ engine: "standard" | "neural" | "long-form" | "generative"; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Amazon Polly TTS vendor. * * @example * ```typescript * const tts = new AmazonTTS({ * accessKey: process.env.AWS_ACCESS_KEY_ID, * secretKey: process.env.AWS_SECRET_ACCESS_KEY, * region: 'us-east-1', * voiceId: 'Joanna', * engine: 'neural', * }); * ``` */ export declare class AmazonTTS extends BaseTTS { private readonly options; constructor(options: AmazonTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Deepgram TTS (Beta). */ export interface DeepgramTTSOptions { /** Deepgram API key */ apiKey: string; /** Deepgram TTS model (e.g., 'aura-2-thalia-en') */ model: string; /** WebSocket endpoint (defaults server-side to Deepgram's speak endpoint) */ baseUrl?: string; /** Audio sample rate in Hz */ sampleRate?: number; /** Additional Deepgram TTS parameters, flattened into tts.params */ additionalParams?: Record; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Deepgram TTS vendor (Beta). */ export declare class DeepgramTTS extends BaseTTS { private readonly options; constructor(options: DeepgramTTSOptions); toConfig(): TtsConfig; } /** Constructor options for Gradium TTS. */ export interface GradiumTTSOptions { /** Gradium API key */ apiKey: string; /** WebSocket endpoint for streaming TTS output */ url?: string; /** Gradium TTS model name (e.g., 'default') */ modelName?: string; /** Gradium voice identifier */ voiceId?: string; /** Audio sample rate in Hz */ sampleRate?: number; /** Additional vendor-specific parameters */ additionalParams?: Record; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** Gradium TTS vendor. */ export declare class GradiumTTS extends BaseTTS { private readonly options; constructor(options: GradiumTTSOptions); toConfig(): TtsConfig; } /** Constructor options for Mistral TTS. */ export interface MistralTTSOptions { /** Mistral API key */ apiKey: string; /** Mistral TTS model name (e.g., 'voxtral-mini-tts-2603') */ model?: string; /** Mistral voice identifier */ voice?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** Mistral TTS vendor. */ export declare class MistralTTS extends BaseTTS { private readonly options; constructor(options: MistralTTSOptions); toConfig(): TtsConfig; } /** Constructor options for Typecast TTS. */ export interface TypecastTTSOptions { /** Typecast API key */ apiKey: string; /** Typecast voice identifier */ voiceId: string; /** Typecast TTS model name (e.g., 'ssfm-v30') */ model: string; /** Additional vendor-specific parameters */ additionalParams?: Record; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** Typecast TTS vendor. */ export declare class TypecastTTS extends BaseTTS { private readonly options; constructor(options: TypecastTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Hume AI TTS. */ export interface HumeAITTSOptions { /** Hume AI API key */ key: string; /** Configuration ID */ configId?: string; /** Hume AI voice ID */ voiceId: string; /** Base URL for the Hume AI API */ baseUrl?: string; /** Voice provider type */ provider: "HUME_AI" | "CUSTOM_VOICE"; /** Playback speed of the generated speech */ speed?: number; /** Duration of silence in seconds to add at the end of each utterance */ trailingSilence?: number; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Hume AI TTS vendor. * * @example * ```typescript * const tts = new HumeAITTS({ * key: process.env.HUME_API_KEY, * voiceId: 'voice-id', * provider: 'CUSTOM_VOICE', * }); * ``` */ export declare class HumeAITTS extends BaseTTS { private readonly options; constructor(options: HumeAITTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Rime TTS. * * - `credentialMode: CredentialMode.Managed` requires `baseUrl` and `modelId` * - omitted / `CredentialMode.Byok` requires `key`, `speaker`, and `modelId` */ type RimeTTSCommonOptions = { /** Skip patterns for bracketed content */ skipPatterns?: number[]; }; export type RimeTTSOptions = (RimeTTSCommonOptions & { /** Use Agora-managed Rime credentials */ credentialMode: typeof CredentialMode.Managed; /** Model ID */ modelId: string; /** WebSocket URL for the Rime streaming API */ baseUrl: string; /** Rime API key (optional for managed credentials) */ key?: string; /** Speaker ID (optional for managed credentials) */ speaker?: string; }) | (RimeTTSCommonOptions & { /** Bring-your-own Rime credentials (default when omitted) */ credentialMode?: typeof CredentialMode.Byok; /** Rime API key */ key: string; /** Speaker ID */ speaker: string; /** Model ID */ modelId: string; /** WebSocket URL for the Rime streaming API */ baseUrl?: string; }); /** * Rime TTS vendor. * * @example * Agora-managed credentials: * ```typescript * const managedTts = new RimeTTS({ * credentialMode: CredentialMode.Managed, * baseUrl: 'wss://users.rime.ai/ws', * modelId: 'mist', * }); * ``` * * @example * Bring your own credentials (the default mode): * ```typescript * const byokTts = new RimeTTS({ * key: process.env.RIME_API_KEY, * speaker: 'speaker-id', * modelId: 'mist', * }); * ``` */ export declare class RimeTTS extends BaseTTS { private readonly options; constructor(options: RimeTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Fish Audio TTS. */ export interface FishAudioTTSOptions { /** Fish Audio API key */ key: string; /** Reference ID */ referenceId: string; /** Backend used by Fish Audio */ backend: string; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Fish Audio TTS vendor. * * @example * ```typescript * const tts = new FishAudioTTS({ * key: process.env.FISH_AUDIO_API_KEY, * referenceId: 'reference-id', * backend: 'speech-1.5', * }); * ``` */ export declare class FishAudioTTS extends BaseTTS { private readonly options; constructor(options: FishAudioTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for MiniMax TTS. */ type MiniMaxTTSCommonOptions = { /** MiniMax API key. Optional only for AgentKit-supported Agora-managed models. */ key?: string; /** MiniMax group identifier */ groupId?: string; /** TTS model (e.g., 'speech-02-turbo') */ model: string; /** Voice style identifier (e.g., 'English_captivating_female1') */ voiceId?: string; /** WebSocket endpoint (e.g., 'wss://api-uw.minimax.io/ws/v1/t2a_v2') */ url?: string; /** Skip patterns for bracketed content */ skipPatterns?: number[]; }; export type MiniMaxTTSOptions = (MiniMaxTTSCommonOptions & { key: string; }) | (Omit & { key?: undefined; model: MiniMaxPresetModel; groupId?: string; voiceId?: string; url?: string; }); /** * MiniMax TTS vendor. * * @example * ```typescript * const tts = new MiniMaxTTS({ * key: process.env.MINIMAX_API_KEY, * groupId: 'your-group-id', * model: 'speech-02-turbo', * voiceId: 'English_captivating_female1', * url: 'wss://api-uw.minimax.io/ws/v1/t2a_v2', * }); * ``` */ export declare class MiniMaxTTS extends BaseTTS { private readonly options; constructor(options: MiniMaxTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Sarvam TTS (Beta). */ export interface SarvamTTSOptions { /** Sarvam API subscription key */ key: string; /** Speaker/voice ID (e.g., 'anushka', 'abhilash', 'karun', 'hitesh', 'manisha', 'vidya', 'arya') */ speaker: string; /** Target language code (e.g., 'en-IN', 'hi-IN', 'ta-IN') */ targetLanguageCode: import("../types.mjs").SarvamTtsParams["target_language_code"]; /** Pitch adjustment for the voice */ pitch?: number; /** Speed of speech */ pace?: number; /** Volume level of the speech */ loudness?: number; /** Audio sample rate in Hz */ sampleRate?: number; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Sarvam TTS vendor (Beta). * * @example * ```typescript * const tts = new SarvamTTS({ * key: process.env.SARVAM_API_KEY, * speaker: 'anushka', * targetLanguageCode: 'en-IN', * }); * ``` */ export declare class SarvamTTS extends BaseTTS { private readonly options; constructor(options: SarvamTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Murf TTS. */ export interface MurfTTSOptions { /** Murf API key */ key: string; /** Voice ID (e.g., 'Ariana', 'Natalie', 'Ken') */ voiceId?: string; /** WebSocket endpoint for streaming TTS output */ baseUrl?: string; /** Locale for the selected voice */ locale?: string; /** Speech rate adjustment */ rate?: number; /** Pitch adjustment */ pitch?: number; /** TTS model to use */ model?: string; /** Audio sample rate in Hz */ sampleRate?: number; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Murf TTS vendor. * * @example * ```typescript * const tts = new MurfTTS({ * key: process.env.MURF_API_KEY, * voiceId: 'Ariana', * }); * ``` */ export declare class MurfTTS extends BaseTTS { private readonly options; constructor(options: MurfTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for Generic OpenAI-compatible TTS. */ export interface GenericTTSOptions { /** HTTP(S) endpoint of the generic TTS service */ url: string; /** Custom headers to include in requests to the generic TTS service */ headers?: Record; /** API key for the generic TTS service */ apiKey?: string; /** TTS model name */ model?: string; /** Voice name */ voice?: string; /** Speech rate */ speed?: number; /** Output audio sample rate in Hz */ sampleRate?: number; /** Output audio format. ConvoAI currently supports `pcm` only. */ responseFormat?: "pcm"; /** Additional voice style control instruction */ instruction?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * Generic OpenAI-compatible TTS vendor. */ export declare class GenericTTS extends BaseTTS { private readonly options; constructor(options: GenericTTSOptions); toConfig(): TtsConfig; } /** * Constructor options for xAI TTS. */ export interface XAiTTSOptions { /** xAI API key */ apiKey: string; /** BCP-47 language code for speech synthesis */ language: string; /** xAI voice identifier */ voiceId?: string; /** Audio sample rate in Hz */ sampleRate?: number; /** Additional vendor-specific parameters */ additionalParams?: Record; /** Skip patterns for bracketed content */ skipPatterns?: number[]; } /** * xAI TTS vendor. */ export declare class XAiTTS extends BaseTTS { private readonly options; constructor(options: XAiTTSOptions); toConfig(): TtsConfig; } /** Constructor options for Smallest AI TTS. */ export interface SmallestAITTSOptions { /** Smallest AI API key. */ apiKey: string; /** HTTP endpoint for the Smallest AI streaming TTS API. */ url?: string; model?: string; voiceId?: string; sampleRate?: number; speed?: number; language?: string; numberPronunciationLanguage?: string; mathNotation?: boolean; pronunciationDicts?: string[]; sessionId?: string; requestId?: string; /** Additional Smallest AI parameters. Explicit options take precedence. */ additionalParams?: Partial; /** Skip patterns for bracketed content. */ skipPatterns?: number[]; } /** Smallest AI streaming TTS vendor. */ export declare class SmallestAITTS extends BaseTTS { private readonly options; constructor(options: SmallestAITTSOptions); toConfig(): TtsConfig; } export {};