/** * Type-safe STT (Speech-to-Text) vendor classes. */ import { type DeepgramPresetModel } from "../presets.js"; import type { SttConfig, TurnDetectionLanguage } from "../types.js"; import { BaseSTT, type SampleRate } from "./base.js"; /** Gemini transcription models. */ export declare const GeminiSTTModels: { readonly Transcribe35Live: "gemini-3.5-transcribe-live"; }; /** A Gemini transcription model. Known models autocomplete while newer model names remain accepted. */ export type GeminiSTTModel = (typeof GeminiSTTModels)[keyof typeof GeminiSTTModels] | (string & {}); /** Transcription output mode for Gemini ASR. */ export declare const GeminiTranscriptionMode: { readonly Smart: "SMART"; readonly Verbatim: "VERBATIM"; }; export type GeminiTranscriptionMode = (typeof GeminiTranscriptionMode)[keyof typeof GeminiTranscriptionMode]; /** Constructor options for Google Gemini STT. */ export interface GeminiSTTOptions { /** Google Gemini API key. */ apiKey: string; /** Gemini transcription model. Defaults to `gemini-3.5-transcribe-live`. */ model?: GeminiSTTModel; /** Language code for speech recognition (for example, `en-US`). */ language?: string; /** Candidate transcription languages, serialized as `params.language_hints`. */ languageHints?: readonly TurnDetectionLanguage[]; /** @deprecated Use `languageHints` instead. */ languageCodes?: readonly TurnDetectionLanguage[]; /** Words and phrases used to bias recognition. */ customVocabulary?: readonly string[]; /** Audio sample rate in Hz. Defaults to 16000. */ sampleRate?: SampleRate; /** Whether to include word-level timestamps. Cannot be true when custom vocabulary or SMART mode is used. */ wordTimestamp?: boolean; /** Transcript cleanup mode. Defaults to VERBATIM when omitted. */ mode?: GeminiTranscriptionMode; /** Whether to include speaker labels. Cannot be true when mode is SMART. */ diarization?: boolean; /** Additional vendor-specific parameters. Explicit options take precedence. */ additionalParams?: Record; } /** Google Gemini STT vendor. */ export declare class GeminiSTT extends BaseSTT { private readonly options; constructor(options: GeminiSTTOptions); toConfig(): SttConfig; } /** * Constructor options for Speechmatics STT. */ interface SpeechmaticsSTTCommonOptions { /** Language code (e.g., 'en', 'es', 'fr') */ language: string; /** Model name */ model?: string; /** Speechmatics streaming WebSocket URL (for example, wss://eu2.rt.speechmatics.com/v2) */ uri?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; } export type SpeechmaticsSTTOptions = SpeechmaticsSTTCommonOptions & ({ /** Speechmatics API key */ key: string; /** @deprecated Use `key` instead. This alias is normalized to the REST API's `key` field. */ apiKey?: string; } | { key?: undefined; /** @deprecated Use `key` instead. This alias is normalized to the REST API's `key` field. */ apiKey: string; }); /** * Speechmatics STT vendor. * * @example * ```typescript * const stt = new SpeechmaticsSTT({ * key: process.env.SPEECHMATICS_API_KEY, * language: 'en', * }); * ``` */ export declare class SpeechmaticsSTT extends BaseSTT { private readonly options; constructor(options: SpeechmaticsSTTOptions); toConfig(): SttConfig; } /** * Constructor options for Deepgram STT. */ type DeepgramSTTCommonOptions = { /** Deepgram API key. Optional only for the Agora-managed `nova-2` and `nova-3` path. */ apiKey?: string; /** Model to use (e.g., 'nova-2', 'enhanced', 'base') */ model?: string; /** Language code (e.g., 'en-US', 'es', 'fr') */ language?: string; /** Enable smart formatting */ smartFormat?: boolean; /** Enable punctuation */ punctuation?: boolean; /** Boost specialized terms and brands for Deepgram */ keyterm?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; }; export type DeepgramSTTOptions = (DeepgramSTTCommonOptions & { apiKey: string; }) | (Omit & { apiKey?: undefined; model: DeepgramPresetModel; }); /** * Deepgram STT vendor. * * @example * ```typescript * const stt = new DeepgramSTT({ * apiKey: process.env.DEEPGRAM_API_KEY, * model: 'nova-2', * smartFormat: true, * }); * ``` */ export declare class DeepgramSTT extends BaseSTT { private readonly options; constructor(options?: DeepgramSTTOptions); toConfig(): SttConfig; } /** * Constructor options for Microsoft Azure Speech STT. */ export interface MicrosoftSTTOptions { /** Microsoft Azure subscription key */ key: string; /** Azure region (e.g., 'eastus', 'westus') */ region: string; /** Language code (e.g., 'en-US', 'es-ES') */ language: string; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * Microsoft Azure Speech STT vendor. * * @example * ```typescript * const stt = new MicrosoftSTT({ * key: process.env.AZURE_SPEECH_KEY, * region: 'eastus', * language: 'en-US', * }); * ``` */ export declare class MicrosoftSTT extends BaseSTT { private readonly options; constructor(options: MicrosoftSTTOptions); toConfig(): SttConfig; } /** * Constructor options for OpenAI Whisper STT. */ export interface OpenAISTTOptions { /** OpenAI API key */ apiKey: string; /** Model to use (default: 'whisper-1') */ model?: string; /** Language code */ language?: string; /** Prompt that guides OpenAI transcription */ prompt?: string; /** Full OpenAI input_audio_transcription override */ inputAudioTranscription?: Record; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * OpenAI Whisper STT vendor. * * @example * ```typescript * const stt = new OpenAISTT({ * apiKey: process.env.OPENAI_API_KEY, * }); * ``` */ export declare class OpenAISTT extends BaseSTT { private readonly options; constructor(options: OpenAISTTOptions); toConfig(): SttConfig; } /** * Constructor options for Google Cloud Speech-to-Text STT. */ export interface GoogleSTTOptions { /** Google Cloud project ID where Speech-to-Text is enabled */ projectId: string; /** Google Cloud region for the recognizer (for example, global) */ location: string; /** Google service account credentials JSON string */ adcCredentialsString: string; /** Language code (e.g., 'en-US', 'es-ES') */ language: string; /** Recognition model to use */ model?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * Google Cloud Speech-to-Text STT vendor. * * @example * ```typescript * const stt = new GoogleSTT({ * projectId: process.env.GOOGLE_ASR_PROJECT_ID, * location: 'global', * adcCredentialsString: process.env.GOOGLE_APPLICATION_CREDENTIALS_STRING, * language: 'en-US', * }); * ``` */ export declare class GoogleSTT extends BaseSTT { private readonly options; constructor(options: GoogleSTTOptions); toConfig(): SttConfig; } /** * Constructor options for Amazon Transcribe STT. */ export interface AmazonSTTOptions { /** AWS Access Key ID */ accessKey: string; /** AWS Secret Access Key */ secretKey: string; /** AWS region (e.g., 'us-east-1') */ region: string; /** Language code */ language: string; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * Amazon Transcribe STT vendor. * * @example * ```typescript * const stt = new AmazonSTT({ * accessKey: process.env.AWS_ACCESS_KEY_ID, * secretKey: process.env.AWS_SECRET_ACCESS_KEY, * region: 'us-east-1', * language: 'en-US', * }); * ``` */ export declare class AmazonSTT extends BaseSTT { private readonly options; constructor(options: AmazonSTTOptions); toConfig(): SttConfig; } /** * Constructor options for AssemblyAI STT. */ export interface AssemblyAISTTOptions { /** AssemblyAI API key */ apiKey: string; /** Language code */ language: string; /** AssemblyAI streaming WebSocket URL */ ws_url?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * AssemblyAI STT vendor. * * @example * ```typescript * const stt = new AssemblyAISTT({ * apiKey: process.env.ASSEMBLYAI_API_KEY, * language: 'en-US', * }); * ``` */ export declare class AssemblyAISTT extends BaseSTT { private readonly options; constructor(options: AssemblyAISTTOptions); toConfig(): SttConfig; } /** * Constructor options for Agora ARES STT. */ export interface AresSTTOptions { /** Hotwords that improve recognition accuracy */ keywords?: string[]; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * Agora ARES (Adaptive Recognition Engine for Speech) STT vendor. * * @example * ```typescript * const stt = new AresSTT(); * ``` */ export declare class AresSTT extends BaseSTT { private readonly options; constructor(options?: AresSTTOptions); toConfig(): SttConfig; } /** * Constructor options for Sarvam STT. */ export interface SarvamSTTOptions { /** Sarvam API key */ apiKey: string; /** Language code (e.g., 'en', 'hi', 'ta') */ language: string; /** Model name */ model?: string; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * Sarvam STT vendor (Beta). * * @example * ```typescript * const stt = new SarvamSTT({ * apiKey: process.env.SARVAM_API_KEY, * language: 'en', * }); * ``` */ export declare class SarvamSTT extends BaseSTT { private readonly options; constructor(options: SarvamSTTOptions); toConfig(): SttConfig; } /** * Constructor options for xAI STT. */ export interface XAiSTTOptions { /** xAI API key */ apiKey: string; /** Language code for speech recognition */ language?: string; /** WebSocket endpoint URL for the xAI streaming STT API */ baseUrl?: string; /** Audio sample rate in Hz */ sampleRate?: number; /** Additional vendor-specific parameters */ additionalParams?: Record; } /** * xAI STT vendor. * * @example * ```typescript * const stt = new XAiSTT({ * apiKey: process.env.XAI_API_KEY, * language: 'en', * }); * ``` */ export declare class XAiSTT extends BaseSTT { private readonly options; constructor(options: XAiSTTOptions); toConfig(): SttConfig; } /** Constructor options for Smallest AI STT. */ export interface SmallestAISTTOptions { /** Smallest AI API key. */ apiKey: string; /** Language code for speech recognition. */ language?: string; /** WebSocket endpoint for the Smallest AI streaming STT API. */ url?: string; /** Input audio sample rate in Hz. */ sampleRate?: number; /** Input audio encoding. */ encoding?: string; /** Boolean options are serialized as `"true"` or `"false"` for the Smallest AI wire protocol. */ wordTimestamps?: boolean; sentenceTimestamps?: boolean; diarize?: boolean; vadEvents?: boolean; endpointing?: boolean; /** End-of-utterance timeout in milliseconds. */ eouTimeoutMs?: number; format?: boolean; finalizeOnWords?: boolean; /** Maximum number of words per result. */ maxWords?: string; punctuate?: boolean; capitalize?: boolean; itnNormalize?: boolean; fullTranscript?: boolean; /** Comma-separated keyword boosts in `keyword:weight` format. */ keywords?: string; redactPii?: boolean; redactPci?: boolean; /** Additional Smallest AI parameters. Explicit options take precedence. */ additionalParams?: Partial; } /** Smallest AI streaming STT vendor. */ export declare class SmallestAISTT extends BaseSTT { private readonly options; constructor(options: SmallestAISTTOptions); toConfig(): SttConfig; } export {};