/** * transformers.js (ONNX, dual-environment) STT (speech-to-text) adapter battery. * * @remarks * Specialist battery backed by transformers.js's `automatic-speech-recognition` (Whisper-family) * pipeline. **Environment-neutral** — runs in Node (via `onnxruntime-node`) and the browser (via * `onnxruntime-web` / WebGPU), auto-selected by the package; there is no WebGPU requirement, mirroring * the transformers.js Embeddings battery this adapter is modeled on. * * Accepts any {@link @nhtio/adk/batteries/specialists/_shared!SpecialistAudioInput} form: already- * decoded mono PCM (skips decoding), or an encoded container (bytes / bytes+mime / a duck-typed * `Media`) decoded via an injectable {@link DecodeAudioFn}. Either path is resampled to the 16 kHz mono * PCM Whisper expects before the pipeline call. * * `@huggingface/transformers` is an optional peer dependency, imported lazily. */ import type { SpecialistAudioInput } from "../../_shared/index"; import type { TranscribeOptions, TranscribeResult } from "./types"; /** * STT adapter for transformers.js's automatic-speech-recognition (Whisper-family) pipeline. * * @remarks * Reusable: construct once, call {@link TransformersJsSttAdapter.transcribe} as many times as needed. * The pipeline is resolved lazily on first use (or via {@link preload}) and cached with single-flight * semantics so concurrent calls share one load. */ export declare class TransformersJsSttAdapter { #private; /** * Whether this battery is available. transformers.js is environment-neutral (Node + browser), so * this is `true` whenever the runtime can import the peer — there is no WebGPU requirement. */ static isAvailable(): boolean; /** * @param options - Constructor options. Validated eagerly. * @throws {@link @nhtio/adk/batteries!E_INVALID_TRANSFORMERS_JS_STT_OPTIONS} when invalid. */ constructor(options: unknown); /** Instance availability probe (honours an injected `isAvailable`). */ isAvailable(): boolean; /** Eagerly loads (and caches) the pipeline so the first `transcribe` call is fast. Idempotent. */ preload(): Promise; /** Drops the cached pipeline and in-flight load so the next call reloads. */ reset(): void; /** * Release the loaded model's ONNX sessions + GPU/wasm buffers, then drop the cached pipeline. * * @remarks * `reset()` only nulls the JS reference; the native ONNX Runtime sessions and WebGPU/wasm device * memory stay alive until GC. `AutomaticSpeechRecognitionPipeline` extends `Pipeline`, which exposes * `dispose()` — this awaits it so the memory is reclaimed between loads, swallows a disposal error * (teardown must not throw), and finishes with `reset()`. Idempotent. */ dispose(): Promise; /** * Transcribes an audio clip. * * @param input - The audio input, in any {@link @nhtio/adk/batteries/specialists/_shared!SpecialistAudioInput} * form (pre-decoded PCM or an encoded container). * @param opts - Per-call transcription options (language / translate / timestamps). * @returns The recognized text, plus per-segment timing when `opts.timestamps` was set. * @throws {@link @nhtio/adk/batteries!E_TRANSFORMERS_JS_STT_ENGINE_ERROR} when the pipeline fails to * load or the transcription call fails. */ transcribe(input: SpecialistAudioInput, opts?: TranscribeOptions): Promise; }