import type { Observable } from 'rxjs'; import type { AsrConfig, AsrHandle } from './asr-handle'; import type { TtsConfig, TtsHandle } from './tts-handle'; import type { DataMessage } from './events'; import type { MixerQueueControl, PlayOptions, PresayOptions, PreloadOptions } from './mixer'; import type { LegacyPhraseRecord } from './legacy-phrase'; import type { TextInput } from './text-input'; import type { MediaError } from './errors'; import type { ChannelLlm } from './llm'; import type { ChannelSip } from './sip'; /** * Audio playback API — TTS synthesis, file/URL playback, caching, and multi-queue mixing. * * Audio is routed through a **multi-queue mixer** (indices **0–4**). Each queue has its own * volume and lifecycle; use different **`queue`** values in {@link PlayOptions} so earcons, * hold music, and agent TTS can be **`stop`**ped independently. * * ### Pre-answer behaviour (SIP) * * For SIP calls, `say()` and `play()` automatically wait for the RTP pipeline to be ready * (early media or 200 OK) before starting playback. You do **not** need to call * `sip.waitForEarly()` / `sip.waitForAnswer()` manually before speaking — the audio * operation waits and starts as soon as the call enters `"early"` or `"active"` state. * * If the call terminates **before** any media is available, deferred `say`/`play` calls * resolve as no-ops (they do not throw). */ export interface ChannelAudio { /** * Synthesize text with TTS and play the result on the selected mixer queue. * * On WS channels the runtime also emits text/audio progress events to the client * (`TEXT_*`, `AUDIO_*`). On headless channels this is a no-op that only logs a warning. * * @param input - Plain string, or `Observable` of chunks for streaming synthesis. * @param options - Queue, alias, queue volume, {@link PlayOptions.ttsVendor}, * {@link PlayOptions.name} (LE `key_storage` row), {@link PlayOptions.ttsConfig}, * {@link PlayOptions.ttsStrategy}, etc. * @returns Resolves when playback of this **`say`** invocation has finished (queue may still * contain other items). * * On SIP channels, if the call has not yet reached `"early"`/`"active"` state when * `say()` is invoked, the call automatically waits for media availability. The Promise * resolves as a no-op if the call terminates before media starts flowing. */ say(input: string | Observable, options?: PlayOptions): Promise; /** * Play audio from a URL/path string, or a {@link LegacyPhraseRecord} from the Voctiv platform phrase DB. * * The runtime decodes the full source to PCM and pushes it to the mixer. `ttsVendor`, * `ttsConfig`, `ttsStrategy`, and `cache` do not affect raw playback. * * @param source - HTTP(S) URL, local path, or structured {@link LegacyPhraseRecord} (no TTS). * @param options - Queue, alias, volume, loop; **`tts*`** fields apply only if the host * ever wraps playback (normally ignored for raw audio). * * On SIP channels, automatically waits for early/active state before starting * (same behaviour as `say()`). */ play(source: string | LegacyPhraseRecord, options?: PlayOptions): Promise; /** * Download/decode **`source`** through the audio player. * * In ScriptEngine this warms the decoder/audio-player path for `play()`. * In legacy mode the decoded PCM is always persisted into `record_phrase` / * `record_phrase_file` (default phrase name `_cache`). * * @param source - URL or file path accepted by the host's fetch layer. * @param options - Optional; only `cache` is used to override persist name/flag/language. */ preload(source: string, options?: PreloadOptions): Promise; /** * Run TTS ahead of time and store PCM in the host TTS cache. * * Later `say()` calls with the same resolved TTS config and text can reuse the cached * file. In legacy mode the result is always persisted into `record_phrase` * (default phrase name `_cache`). * * @param text - Full text to synthesize. * @param options - TTS vendor/name/config and optional cache name/flag override. */ presay(text: string, options?: PresayOptions): Promise; /** * @param index - Mixer queue **0–4**. * @returns Control handle for that queue's volume and item lifecycle observables. */ queue(index: number): MixerQueueControl; /** * Remove one item by **`alias`**. * * When **`queue`** is omitted, the SIP/WS runtime searches all five queues. When set, * only that queue is inspected. For sentence-split TTS, queue item aliases are * suffixed as `alias-0`, `alias-1`, etc.; remove the concrete suffix if you need to * cancel one synthesized sentence. * @param alias - **`PlayOptions.alias`** of the item to drop. * @param queue - When set, only this queue index is searched; when omitted, all five queues are searched. */ remove(alias: string, queue?: number): void; /** * Stop playback and clear all pending items on a single queue. * * Also aborts in-flight sentence TTS generation for that queue. * @param queue - Queue index **0–4**. */ stop(queue: number): void; /** Stop and clear **all** queues; WS clients also receive an audio interrupt signal. */ stopAll(): void; } /** Channel-level observables: VAD-related speech, barge-in, session end, and arbitrary data messages. */ export interface ChannelEvents { /** User started speaking (VAD speech start, socket speech event, or synthetic text input). */ readonly speechStart$: Observable; /** User stopped speaking (VAD speech end, socket speech event, ASR final, or synthetic text input). */ readonly speechEnd$: Observable; /** User speech caused an interrupt of bot audio (barge-in path; may be inert without VAD). */ readonly interrupt$: Observable; /** Session is ending: hangup, client disconnect, or explicit **`channel.destroy()`**. */ readonly terminated$: Observable; /** Structured messages from the WS client or bridge (event + payload). */ readonly message$: Observable; /** * Canonical stream of runtime media errors (ASR connector creation/runtime failures, * TTS synthesis/playback failures, missing API keys, etc.). * * TTS methods still reject their own promises when the awaited operation fails, and * `AsrHandle.error$` still exposes recognizer-scoped errors, but the same failures * are also emitted here for one-place monitoring. Subscribing is optional — * unhandled errors are always logged server-side. * * ```ts * channel.events.error$.subscribe(err => { * console.log(`[${err.source}] ${err.message}`); * }); * ``` */ readonly error$: Observable; } /** * Primary script API for real-time voice: ASR, TTS/audio, SIP, LLM, and **`params`** from the host. * * **`type`** is **`"sip"`** for telephony or **`"ws"`** for WebSocket/script-manager sessions. * Headless sessions currently also expose a `"ws"` typed synthetic channel; use * {@link import('./script-context').ScriptDialogContext.headless} to distinguish them. * In headless mode audio/SIP/text input are mostly no-ops while **`llm`** / **`platform`** * still work. */ export interface MediaChannel { readonly type: 'sip' | 'ws'; readonly callerId: string; readonly calledNumber: string; /** * Merged session parameters from env, Omni, Voctiv platform defaults, and route. * * Relevant to ASR/TTS when **Voctiv platform** compatibility is on (host-dependent keys), for example: * - **`asrVendor`**, **`ttsVendor`**, **`asrConfig`**, **`ttsConfig`** * - **`defaultAsrName`**, **`defaultTtsName`**: default **`key_storage.name`** when the script * omits **`name`** on {@link AsrConfig} / {@link TtsConfig} / {@link PlayOptions} * - **`authentication_data`**: may include **`legacyAsrKeysByName`**, **`legacyTtsKeysByName`** * (built from LE DB for this dialog's agent + company) * * Treat as read-only unless your integration explicitly documents mutable keys. */ readonly params: Record; /** * Create (or warm) a speech recognizer for this session so recognition can reuse one * TCP/WebSocket (one SSL handshake) for the dialog + vendor + credentials. * * SIP channels feed remote RTP audio. WS channels feed socket `audio` frames and can * create a per-session VAD on first ASR creation. Headless channels return an inert * handle with empty observables. A second call with the same resolved config returns * the existing warm handle. * * @param config - Optional {@link AsrConfig}: **`vendor`**, **`name`** (storage row), **`language`**, * **`data`** overlays, VAD / smart-turn tuning. * @returns A handle you should {@link AsrHandle.destroy} when you no longer need recognition. * If connector creation fails, SIP/WS return a degraded handle with VAD observables but no STT results. */ createAsr(config?: AsrConfig): Promise; /** * Create (or warm) a TTS connector for this session so later `audio.say` / `audio.presay` * can reuse the SSL / WebSocket connection. * * For streaming-capable vendors (e.g. ElevenLabs with `ttsStrategy: 'streaming'`), the host * opens the streaming socket during this call. For batch HTTP vendors, the connector instance * is registered in the dialog-scoped factory (keep-alive pool). * * Prefer calling {@link TtsHandle.say} / {@link TtsHandle.presay} on the returned handle * (same pattern as {@link AsrHandle}). You may also pass the handle as {@link PlayOptions.tts} * to `channel.audio.say` / `presay`, or omit it when a later `say` resolves to the same * vendor+config — the host will reuse a matching session. * * @param config - Optional {@link TtsConfig}: **`vendor`**, **`name`** (storage row), **`data`** overlays. * @returns A handle you should {@link TtsHandle.destroy} when you no longer need the connection. * If connector creation fails, SIP/WS return a degraded handle (synthesis falls back to ephemeral). */ createTts(config?: TtsConfig): Promise; /** Push synthetic ASR results (testing / WS debug). No-op on headless channels. */ readonly textInput: TextInput; readonly audio: ChannelAudio; readonly events: ChannelEvents; readonly llm: ChannelLlm; readonly sip: ChannelSip; /** Emit a structured message to the remote peer. Implemented for WS; SIP ignores it and headless logs it. */ sendMessage(data: DataMessage): void; /** Release connectors, mixer, recordings, persistent LLM streams, and subscriptions. */ destroy(): void; } //# sourceMappingURL=media-channel.d.ts.map