/** * Dual-environment (Node + browser) executor adapter for transformers.js (`@huggingface/transformers`). * * @module @nhtio/adk/batteries/llm/transformers_js/adapter * * @remarks * On-device text generation via ONNX Runtime — `onnxruntime-node` (native) in Node, `onnxruntime-web` * (WASM + WebGPU) in the browser, auto-selected by the package. So this battery is * **environment-neutral**: it does NOT gate on WebGPU. * * **transformers.js is text-in / text-out.** It injects tool definitions into the chat template but * does NOT return structured tool calls or reasoning — the model emits both as **family-specific raw * text**. This adapter parses them out via the shared, configurable parser layer (`toolCallParser` / * `reasoningParser`, both defaulting to `'auto'`): after generation, the reasoning parser pulls * thinking into ADK Thoughts and the tool-call parser pulls calls into ADK ToolCalls, leaving clean * prose as the assistant Message. * * Three pluggable layers mirror the other LLM batteries: swappable translation helpers, three-layer * options merging (constructor → `executor()` overrides → `ctx.stash.transformersJs`), and an * injectable/lazy pipeline (`pipeline` or `createPipeline`, defaulting to a dynamic import). */ import type { DispatchExecutorFn } from "../../../dispatch_runner"; import type { TransformersJsAdapterOptions, TransformersJsPipeline } from "./types"; /** * Dual-environment executor adapter for transformers.js text generation. * * @remarks * Construct with at least `{ model }`; wire `new TransformersJsAdapter(opts).executor()` into a * `DispatchRunner` as the `executorCallback`. The pipeline is resolved lazily on first dispatch (or * eagerly via {@link TransformersJsAdapter.preload}); pass `pipeline` to inject a pre-built one. */ export declare class TransformersJsAdapter { #private; /** The `ctx.stash` key under which per-dispatch option overrides are read. */ static readonly STASH_KEY: "transformersJs"; /** * Whether this battery is available. transformers.js is environment-neutral (Node + browser), so this * is `true` whenever the runtime can import the peer — there is no WebGPU requirement. Static form * returns `true`; the instance form honours an injected `isAvailable` override. */ static isAvailable(): boolean; /** * @param options - Raw adapter options, validated against `transformersJsOptionsSchema`. * @throws {@link @nhtio/adk/batteries!E_INVALID_TRANSFORMERS_JS_OPTIONS} when `options` are invalid. */ constructor(options: unknown); /** Instance availability probe (honours the `isAvailable` option override). */ isAvailable(): boolean; /** * Eagerly resolve (load) the pipeline before the first dispatch. * * @param overrides - Optional option overrides applied for this load. */ preload(overrides?: Partial): Promise; /** Drop the cached pipeline/engine and any in-flight load so the next dispatch re-resolves it. */ reset(): void; /** * Release the loaded model's underlying ONNX sessions + GPU/wasm buffers, then drop all cached * references (so the next dispatch re-resolves a fresh pipeline). * * @remarks * `reset()` only nulls the JS references — it does NOT free the native ONNX Runtime sessions or the * WebGPU/wasm device memory they hold. Those leak until GC, and in a browser session that loads many * models back-to-back (e.g. a full matrix run) the accumulated sessions exhaust the heap, surfacing as * `Can't create a session … Failed to load external data file … memory copy`. transformers.js exposes * `PreTrainedModel.dispose()` ("disposes of all the ONNX sessions created during inference") and * `Pipeline.dispose()` — this awaits them so the memory is actually reclaimed between loads. Settles any * in-flight load first, swallows per-handle disposal errors (a half-loaded model must not throw out of * teardown), and finishes with `reset()`. Idempotent and safe to call when nothing is loaded. */ dispose(): Promise; /** * Free the WebGPU buffer cache by releasing the model's ONNX sessions, then reload the same model. * * @remarks * The consumer-facing lever for the ONNX Runtime Web WebGPU **buffer-freelist high-water-mark** (see * {@link @nhtio/adk/batteries!probeGpuBudget} and the battery's GPU-budget notes). ORT-web parks freed * activation buffers in per-size buckets sized to the largest tensor shape the model has run; the pool * is flushed ONLY when every session of the model is released (ORT clears the cache at * `sessionCount === 0`; microsoft/onnxruntime#22490). There is no public flag to flush it mid-life, so * the supported way to reclaim that retained working-set without permanently unloading the model is to * dispose the sessions and load again. * * This is exactly `dispose()` followed by `preload()` — surfaced as a named method because "recycle to * free the GPU buffer cache" is a distinct, intentional operation (e.g. an application offering a * "free GPU memory" action after a {@link @nhtio/adk/batteries!E_LLM_GPU_OUT_OF_MEMORY}), not a * teardown. It is NOT invoked automatically by the battery — the ADK surfaces the lever and leaves the * decision to the consumer. Re-incurs the cold-load cost (download is cached; the WebGPU graph/shader * compile is not). Idempotent. * * @param overrides - Optional option overrides applied to the reload (same as {@link preload}). */ recycle(overrides?: Partial): Promise; /** * Produce the bound {@link DispatchExecutorFn} the `DispatchRunner` invokes. * * @param overrides - Option overrides layered above the constructor baseline (below `ctx.stash`). */ executor(overrides?: Partial): DispatchExecutorFn; } /** * Pull the newly-generated assistant text out of a transformers.js text-generation result. * * @remarks * Chat input → `[{ generated_text: Message[] }]` (the last message is the new assistant turn); * string input → `[{ generated_text: string }]`. We always send chat input, so we take the last * message's content, falling back defensively to a string `generated_text`. */ declare const extractGeneratedText: (output: unknown) => string; export { extractGeneratedText as __extractGeneratedText };