{"version":3,"file":"adapter.cjs","names":["#options","#pipeline","#resolvePipeline","#pipelinePromise"],"sources":["../../../../src/batteries/tts/transformers_js/adapter.ts"],"sourcesContent":["/**\n * transformers.js (ONNX, dual-environment) TTS (text-to-speech) adapter battery.\n *\n * @module @nhtio/adk/batteries/tts/transformers_js/adapter\n *\n * @remarks\n * Battery backed by transformers.js's `text-to-speech` (TextToAudio) pipeline. **Environment-neutral**\n * — runs in Node (via `onnxruntime-node`) and the browser (via `onnxruntime-web` / WebGPU), auto-selected\n * by the package; there is no WebGPU requirement, mirroring the transformers.js STT and Embeddings\n * batteries this adapter is modeled on.\n *\n * Accepts plain text input, resolves speaker embeddings / speed / inference steps from the merged\n * constructor + per-call options, and returns a WAV-encoded {@link GeneratedMediaOutput}.\n *\n * `@huggingface/transformers` is an optional peer dependency, imported lazily.\n */\n\nimport { isError } from '@nhtio/adk/guards'\nimport { validateOptions } from './validation'\nimport { emitLifecycle } from '../../llm/chat_common/lifecycle'\nimport { withModelSource } from '../../llm/transformers_js/model_source'\nimport {\n  E_INVALID_TRANSFORMERS_JS_TTS_OPTIONS,\n  E_TRANSFORMERS_JS_TTS_ENGINE_ERROR,\n} from './exceptions'\nimport type { RawAudioLike, GeneratedMediaOutput } from '../_shared'\nimport type {\n  TransformersJsTtsAdapterOptions,\n  TransformersJsTtsPipeline,\n  CreateTransformersJsTtsPipeline,\n  TransformersJsSynthesizeOptions,\n} from './types'\n\nconst makeDefaultCreatePipeline = (\n  modelSource: TransformersJsTtsAdapterOptions['modelSource']\n): CreateTransformersJsTtsPipeline => {\n  return async ({ model, device, dtype, onInitProgress }) => {\n    const transformers = await import('@huggingface/transformers')\n    const { pipeline, env } = transformers\n    const load = async () =>\n      (await pipeline('text-to-speech', model, {\n        ...(device ? { device } : {}),\n        ...(dtype ? { dtype } : {}),\n        ...(onInitProgress ? { progress_callback: onInitProgress } : {}),\n      } as never)) as unknown as TransformersJsTtsPipeline\n    // When a custom model source is configured, serve files through it behind the global-`env` mutex.\n    return modelSource ? withModelSource(env as never, modelSource, load) : load()\n  }\n}\n\n/**\n * TTS adapter for transformers.js's text-to-speech (TextToAudio) pipeline.\n *\n * @remarks\n * Reusable: construct once, call {@link TransformersJsTtsAdapter.synthesize} as many times as needed.\n * The pipeline is resolved lazily on first use (or via {@link preload}) and cached with single-flight\n * semantics so concurrent calls share one load.\n */\nexport class TransformersJsTtsAdapter {\n  readonly #options: TransformersJsTtsAdapterOptions\n  #pipeline: TransformersJsTtsPipeline | undefined\n  #pipelinePromise: Promise<TransformersJsTtsPipeline> | undefined\n\n  /**\n   * Whether this battery is available. transformers.js is environment-neutral (Node + browser), so\n   * this is `true` whenever the runtime can import the peer — there is no WebGPU requirement.\n   */\n  public static isAvailable(): boolean {\n    return true\n  }\n\n  /**\n   * @param options - Constructor options. Validated eagerly.\n   * @throws {@link @nhtio/adk/batteries!E_INVALID_TRANSFORMERS_JS_TTS_OPTIONS} when invalid.\n   */\n  constructor(options: unknown) {\n    this.#options = validateOptions(options)\n    this.#pipeline = this.#options.pipeline\n  }\n\n  /** Instance availability probe (honours an injected `isAvailable`). */\n  isAvailable(): boolean {\n    return (this.#options.isAvailable ?? TransformersJsTtsAdapter.isAvailable)()\n  }\n\n  /** Eagerly loads (and caches) the pipeline so the first `synthesize` call is fast. Idempotent. */\n  async preload(): Promise<void> {\n    await this.#resolvePipeline()\n  }\n\n  /** Drops the cached pipeline and in-flight load so the next call reloads. */\n  reset(): void {\n    this.#pipeline = undefined\n    this.#pipelinePromise = undefined\n  }\n\n  /**\n   * Release the loaded model's ONNX sessions + GPU/wasm buffers, then drop the cached pipeline.\n   *\n   * @remarks\n   * `reset()` only nulls the JS reference; the native ONNX Runtime sessions and WebGPU/wasm device\n   * memory stay alive until GC. `TextToAudioPipeline` extends `Pipeline`, which exposes `dispose()` —\n   * this awaits it so the memory is reclaimed between loads, swallows a disposal error (teardown must\n   * not throw), and finishes with `reset()`. Idempotent.\n   */\n  async dispose(): Promise<void> {\n    const pipeline = this.#pipeline ?? (await this.#pipelinePromise?.catch(() => undefined))\n    const pipeWithDispose = pipeline as { dispose?: () => Promise<unknown> } | undefined\n    if (typeof pipeWithDispose?.dispose === 'function') {\n      await Promise.resolve(pipeWithDispose.dispose()).catch(() => undefined)\n    }\n    this.reset()\n  }\n\n  async #resolvePipeline(): Promise<TransformersJsTtsPipeline> {\n    if (this.#pipeline) return this.#pipeline\n    if (!this.isAvailable()) {\n      throw new E_INVALID_TRANSFORMERS_JS_TTS_OPTIONS([\n        'the transformers.js TTS battery is not available in this runtime',\n      ])\n    }\n    const opts = this.#options\n    this.#pipelinePromise ??= (async () => {\n      emitLifecycle(opts, 'transformers_js_tts', opts.model, 'loading', {\n        detail: 'loading text-to-speech pipeline',\n      })\n      // Forward each provider download event into a normalized `loading` lifecycle report.\n      const hasLifecycle =\n        opts.onLifecycle ?? opts.onLoading ?? opts.onReady ?? opts.onGenerating ?? opts.onError\n      const forwardedInitProgress = hasLifecycle\n        ? (info: unknown) => {\n            const p = (info as { progress?: number } | undefined)?.progress\n            emitLifecycle(opts, 'transformers_js_tts', opts.model, 'loading', {\n              ...(typeof p === 'number' ? { progress: p / 100 } : {}),\n              raw: info,\n            })\n            opts.onInitProgress?.(info as never)\n          }\n        : opts.onInitProgress\n      const createPipeline = opts.createPipeline ?? makeDefaultCreatePipeline(opts.modelSource)\n      try {\n        // `from_pretrained` covers both fetch (reported via progress_callback → `loading`) and the\n        // ONNX-graph / WebGPU-WASM warmup. Mark the latter as `compiling` — a COARSE upper-bound marker\n        // (fetch + compile overlap inside the call), consistent with the LLM/embeddings/STT batteries.\n        emitLifecycle(opts, 'transformers_js_tts', opts.model, 'compiling', {\n          detail: 'compiling text-to-speech graph',\n        })\n        const pipe = await createPipeline({\n          model: opts.model,\n          device: opts.device,\n          dtype: opts.dtype,\n          onInitProgress: forwardedInitProgress,\n        })\n        this.#pipeline = pipe\n        emitLifecycle(opts, 'transformers_js_tts', opts.model, 'ready', {\n          detail: 'text-to-speech pipeline ready',\n        })\n        return pipe\n      } catch (err) {\n        this.#pipelinePromise = undefined\n        emitLifecycle(opts, 'transformers_js_tts', opts.model, 'error', { error: err })\n        throw new E_TRANSFORMERS_JS_TTS_ENGINE_ERROR([\n          `could not load the transformers.js pipeline: ${isError(err) ? err.message : String(err)} — install the peer dependency (pnpm add @huggingface/transformers)`,\n        ])\n      }\n    })()\n    return this.#pipelinePromise\n  }\n\n  /**\n   * Synthesizes text into a WAV audio clip.\n   *\n   * @param text - The text to speak. Passed verbatim to the pipeline.\n   * @param opts - Per-call options; each field overrides the constructor default of the same name.\n   * @returns A {@link GeneratedMediaOutput} descriptor with `kind: 'audio'`, `mimeType: 'audio/wav'`,\n   *   and the WAV bytes.\n   * @throws {@link @nhtio/adk/batteries!E_TRANSFORMERS_JS_TTS_ENGINE_ERROR} when the pipeline fails to\n   *   load, the synthesis call throws, or the returned audio lacks `toBlob()`.\n   */\n  async synthesize(\n    text: string,\n    opts?: TransformersJsSynthesizeOptions\n  ): Promise<GeneratedMediaOutput> {\n    const pipe = await this.#resolvePipeline()\n\n    const eff = {\n      voice: opts?.voice ?? this.#options.voice,\n      rate: opts?.rate ?? this.#options.rate,\n      speakerEmbeddings: opts?.speakerEmbeddings ?? this.#options.speakerEmbeddings,\n      numInferenceSteps: opts?.numInferenceSteps ?? this.#options.numInferenceSteps,\n    }\n\n    // Explicit Float32Array/URL wins; a string voice is forwarded as a speaker-embedding\n    // identifier (the documented transformers.js usage for string references).\n    const speakerEmbeddings =\n      eff.speakerEmbeddings ?? (typeof eff.voice === 'string' ? eff.voice : undefined)\n\n    emitLifecycle(this.#options, 'transformers_js_tts', this.#options.model, 'generating')\n\n    let result: unknown\n    try {\n      result = await (\n        pipe as unknown as (text: string, options: Record<string, unknown>) => Promise<unknown>\n      )(text, {\n        ...(typeof speakerEmbeddings !== 'undefined'\n          ? { speaker_embeddings: speakerEmbeddings }\n          : {}),\n        ...(typeof eff.rate === 'number' ? { speed: eff.rate } : {}),\n        ...(typeof eff.numInferenceSteps === 'number'\n          ? { num_inference_steps: eff.numInferenceSteps }\n          : {}),\n      })\n    } catch (err) {\n      emitLifecycle(this.#options, 'transformers_js_tts', this.#options.model, 'error', {\n        error: err,\n      })\n      throw new E_TRANSFORMERS_JS_TTS_ENGINE_ERROR([isError(err) ? err.message : String(err)])\n    }\n\n    // Encode the result to WAV bytes. A null/undefined result, a missing/throwing `toBlob`, or a\n    // rejecting `arrayBuffer()` all funnel through the SAME error path: emit `error` + throw\n    // E_TRANSFORMERS_JS_TTS_ENGINE_ERROR (never a raw exception escaping the contract).\n    let bytes: Uint8Array\n    let mimeType: string\n    try {\n      const rawAudio = result as Partial<RawAudioLike> | null | undefined\n      if (!rawAudio || typeof rawAudio.toBlob !== 'function') {\n        throw new Error('text-to-speech returned a result without a toBlob method')\n      }\n      const blob = rawAudio.toBlob()\n      bytes = new Uint8Array(await blob.arrayBuffer())\n      mimeType = blob.type || 'audio/wav'\n    } catch (err) {\n      emitLifecycle(this.#options, 'transformers_js_tts', this.#options.model, 'error', {\n        error: err,\n      })\n      throw new E_TRANSFORMERS_JS_TTS_ENGINE_ERROR([isError(err) ? err.message : String(err)])\n    }\n\n    emitLifecycle(this.#options, 'transformers_js_tts', this.#options.model, 'complete')\n    return {\n      kind: 'audio',\n      mimeType,\n      bytes,\n      filename: 'speech.wav',\n    }\n  }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;AAiCA,IAAM,6BACJ,gBACoC;CACpC,OAAO,OAAO,EAAE,OAAO,QAAQ,OAAO,qBAAqB;EAEzD,MAAM,EAAE,UAAU,QAAQ,MADC,OAAO;EAElC,MAAM,OAAO,YACV,MAAM,SAAS,kBAAkB,OAAO;GACvC,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;GAC3B,GAAI,QAAQ,EAAE,MAAM,IAAI,CAAC;GACzB,GAAI,iBAAiB,EAAE,mBAAmB,eAAe,IAAI,CAAC;EAChE,CAAU;EAEZ,OAAO,cAAc,mDAAA,gBAAgB,KAAc,aAAa,IAAI,IAAI,KAAK;CAC/E;AACF;;;;;;;;;AAUA,IAAa,2BAAb,MAAa,yBAAyB;CACpC;CACA;CACA;;;;;CAMA,OAAc,cAAuB;EACnC,OAAO;CACT;;;;;CAMA,YAAY,SAAkB;EAC5B,KAAKA,WAAW,iDAAA,gBAAgB,OAAO;EACvC,KAAKC,YAAY,KAAKD,SAAS;CACjC;;CAGA,cAAuB;EACrB,QAAQ,KAAKA,SAAS,eAAe,yBAAyB,aAAa;CAC7E;;CAGA,MAAM,UAAyB;EAC7B,MAAM,KAAKE,iBAAiB;CAC9B;;CAGA,QAAc;EACZ,KAAKD,YAAY,KAAA;EACjB,KAAKE,mBAAmB,KAAA;CAC1B;;;;;;;;;;CAWA,MAAM,UAAyB;EAE7B,MAAM,kBADW,KAAKF,aAAc,MAAM,KAAKE,kBAAkB,YAAY,KAAA,CAAS;EAEtF,IAAI,OAAO,iBAAiB,YAAY,YACtC,MAAM,QAAQ,QAAQ,gBAAgB,QAAQ,CAAC,EAAE,YAAY,KAAA,CAAS;EAExE,KAAK,MAAM;CACb;CAEA,MAAMD,mBAAuD;EAC3D,IAAI,KAAKD,WAAW,OAAO,KAAKA;EAChC,IAAI,CAAC,KAAK,YAAY,GACpB,MAAM,IAAI,iDAAA,sCAAsC,CAC9C,kEACF,CAAC;EAEH,MAAM,OAAO,KAAKD;EAClB,KAAKG,sBAAsB,YAAY;GACrC,kBAAA,cAAc,MAAM,uBAAuB,KAAK,OAAO,WAAW,EAChE,QAAQ,kCACV,CAAC;GAID,MAAM,wBADJ,KAAK,eAAe,KAAK,aAAa,KAAK,WAAW,KAAK,gBAAgB,KAAK,WAE7E,SAAkB;IACjB,MAAM,IAAK,MAA4C;IACvD,kBAAA,cAAc,MAAM,uBAAuB,KAAK,OAAO,WAAW;KAChE,GAAI,OAAO,MAAM,WAAW,EAAE,UAAU,IAAI,IAAI,IAAI,CAAC;KACrD,KAAK;IACP,CAAC;IACD,KAAK,iBAAiB,IAAa;GACrC,IACA,KAAK;GACT,MAAM,iBAAiB,KAAK,kBAAkB,0BAA0B,KAAK,WAAW;GACxF,IAAI;IAIF,kBAAA,cAAc,MAAM,uBAAuB,KAAK,OAAO,aAAa,EAClE,QAAQ,iCACV,CAAC;IACD,MAAM,OAAO,MAAM,eAAe;KAChC,OAAO,KAAK;KACZ,QAAQ,KAAK;KACb,OAAO,KAAK;KACZ,gBAAgB;IAClB,CAAC;IACD,KAAKF,YAAY;IACjB,kBAAA,cAAc,MAAM,uBAAuB,KAAK,OAAO,SAAS,EAC9D,QAAQ,gCACV,CAAC;IACD,OAAO;GACT,SAAS,KAAK;IACZ,KAAKE,mBAAmB,KAAA;IACxB,kBAAA,cAAc,MAAM,uBAAuB,KAAK,OAAO,SAAS,EAAE,OAAO,IAAI,CAAC;IAC9E,MAAM,IAAI,iDAAA,mCAAmC,CAC3C,gDAAgD,eAAA,QAAQ,GAAG,IAAI,IAAI,UAAU,OAAO,GAAG,EAAE,oEAC3F,CAAC;GACH;EACF,GAAG;EACH,OAAO,KAAKA;CACd;;;;;;;;;;;CAYA,MAAM,WACJ,MACA,MAC+B;EAC/B,MAAM,OAAO,MAAM,KAAKD,iBAAiB;EAEzC,MAAM,MAAM;GACV,OAAO,MAAM,SAAS,KAAKF,SAAS;GACpC,MAAM,MAAM,QAAQ,KAAKA,SAAS;GAClC,mBAAmB,MAAM,qBAAqB,KAAKA,SAAS;GAC5D,mBAAmB,MAAM,qBAAqB,KAAKA,SAAS;EAC9D;EAIA,MAAM,oBACJ,IAAI,sBAAsB,OAAO,IAAI,UAAU,WAAW,IAAI,QAAQ,KAAA;EAExE,kBAAA,cAAc,KAAKA,UAAU,uBAAuB,KAAKA,SAAS,OAAO,YAAY;EAErF,IAAI;EACJ,IAAI;GACF,SAAS,MACP,KACA,MAAM;IACN,GAAI,OAAO,sBAAsB,cAC7B,EAAE,oBAAoB,kBAAkB,IACxC,CAAC;IACL,GAAI,OAAO,IAAI,SAAS,WAAW,EAAE,OAAO,IAAI,KAAK,IAAI,CAAC;IAC1D,GAAI,OAAO,IAAI,sBAAsB,WACjC,EAAE,qBAAqB,IAAI,kBAAkB,IAC7C,CAAC;GACP,CAAC;EACH,SAAS,KAAK;GACZ,kBAAA,cAAc,KAAKA,UAAU,uBAAuB,KAAKA,SAAS,OAAO,SAAS,EAChF,OAAO,IACT,CAAC;GACD,MAAM,IAAI,iDAAA,mCAAmC,CAAC,eAAA,QAAQ,GAAG,IAAI,IAAI,UAAU,OAAO,GAAG,CAAC,CAAC;EACzF;EAKA,IAAI;EACJ,IAAI;EACJ,IAAI;GACF,MAAM,WAAW;GACjB,IAAI,CAAC,YAAY,OAAO,SAAS,WAAW,YAC1C,MAAM,IAAI,MAAM,0DAA0D;GAE5E,MAAM,OAAO,SAAS,OAAO;GAC7B,QAAQ,IAAI,WAAW,MAAM,KAAK,YAAY,CAAC;GAC/C,WAAW,KAAK,QAAQ;EAC1B,SAAS,KAAK;GACZ,kBAAA,cAAc,KAAKA,UAAU,uBAAuB,KAAKA,SAAS,OAAO,SAAS,EAChF,OAAO,IACT,CAAC;GACD,MAAM,IAAI,iDAAA,mCAAmC,CAAC,eAAA,QAAQ,GAAG,IAAI,IAAI,UAAU,OAAO,GAAG,CAAC,CAAC;EACzF;EAEA,kBAAA,cAAc,KAAKA,UAAU,uBAAuB,KAAKA,SAAS,OAAO,UAAU;EACnF,OAAO;GACL,MAAM;GACN;GACA;GACA,UAAU;EACZ;CACF;AACF"}