/** * Browser/WebGPU executor adapter for Google's LiteRT-LM (`@litert-lm/core`). * * @module @nhtio/adk/batteries/llm/litert_lm/adapter * * @remarks * On-device LLM inference via WebGPU + a bundled wasm runtime, `.litertlm` models. Unlike the WebLLM * battery (a thin extension of the OpenAI Chat Completions wire shape), LiteRT-LM has its **own** API — * `Engine.create() → engine.createConversation({ preface }) → conversation.sendMessageStreaming(): * ReadableStream` — with native `Message`/`Tool`/`tool_calls`/`tool_response` shapes (tool-call * `arguments` arrive as a parsed object, not a JSON string). So this is a standalone adapter that reuses * the ADK's format-agnostic render helpers but maps history/tools/results to LiteRT's shapes. * * Three pluggable layers mirror the other LLM batteries: swappable translation helpers, three-layer * options merging (constructor → `executor()` overrides → `ctx.stash.liteRtLm`), and an * injectable/lazy engine (`engine` or `createEngine`, defaulting to a dynamic `@litert-lm/core` import). * * **The published `@litert-lm/core` docs lag the library** — every wire field here is mapped against the * installed package's type declarations, the source of truth. The dependency is young (pinned exact); * re-verify on upgrade. */ import type { DispatchExecutorFn } from "../../../dispatch_runner"; import type { LiteRtLmAdapterOptions, LiteRtLmEngine } from "./types"; /** * Does this raw engine message report an INPUT context-cap overflow — the prompt's token ids exceed the * engine's fixed `maxNumTokens`? The LiteRT-web runtime throws e.g. `Input token ids are too long. * Exceeding the maximum number of tokens allowed: 12596 >= 12288`. Matched so the raw throw can be * translated into the typed {@link E_LITERT_LM_CONTEXT_OVERFLOW} instead of the generic stream error — * this is the ENGINE BACKSTOP that fires when the optional pre-dispatch guard is unarmed or undercounts. * Exported so a host can classify a thrown/caught error the same way. */ export declare const isEngineContextOverflowMessage: (message: string) => boolean; /** * Cross-environment executor adapter for LiteRT-LM. * * @remarks * Construct with at least `{ model }`; wire `new LiteRtLmAdapter(opts).executor()` into a * `DispatchRunner` as the `executorCallback`. The engine is resolved lazily on first dispatch (or * eagerly via {@link LiteRtLmAdapter.preload}); pass `engine` to inject a pre-built one (e.g. in tests). */ export declare class LiteRtLmAdapter { #private; /** The `ctx.stash` key under which per-dispatch option overrides are read. */ static readonly STASH_KEY: "liteRtLm"; /** * Returns `true` when the current runtime exposes WebGPU (`navigator.gpu`). */ static isAvailable(): boolean; /** * @param options - Raw adapter options, validated against `liteRtLmOptionsSchema`. * @throws {@link @nhtio/adk/batteries!E_INVALID_LITERT_LM_OPTIONS} when `options` are invalid. */ constructor(options: unknown); /** Instance WebGPU-availability probe (honours the `isWebGPUAvailable` option override). */ isAvailable(): boolean; /** * Eagerly resolve (load) the engine before the first dispatch. * * @param overrides - Optional option overrides applied for this load. * @returns The resolved {@link LiteRtLmEngine}. */ preload(overrides?: Partial): Promise; /** Drop the cached engine and any in-flight load so the next dispatch re-resolves it. */ reset(): void; /** * Release the loaded engine's native resources (`Engine.delete()`), then drop the cached reference. * * @remarks * `reset()` only nulls the JS reference; the LiteRT engine holds a WebGPU device + compiled graph that * stay alive until GC. Loading many `.litertlm` engines back-to-back in one browser session accumulates * those until the GPU heap is exhausted. `@litert-lm/core`'s `Engine` exposes `delete()` — this settles * any in-flight load, awaits `delete()`, swallows a teardown error (teardown must not throw), and * finishes with `reset()`. Idempotent and safe when nothing is loaded. */ dispose(): Promise; /** * Free the WebGPU buffer cache by deleting the engine, then reload the same model. * * @remarks * The consumer-facing lever for the ONNX Runtime Web WebGPU buffer-freelist high-water-mark (see * {@link @nhtio/adk/batteries!probeGpuBudget}). The pool is flushed only when the engine's sessions are * released; there is no public flag to flush it mid-life, so the supported way to reclaim the retained * working-set without permanently unloading is to delete the engine and load again. This is exactly * `dispose()` then `preload()`, surfaced as a named, intentional operation (e.g. an application * offering a "free GPU memory" action after a {@link @nhtio/adk/batteries!E_LLM_GPU_OUT_OF_MEMORY}). * NOT invoked automatically — the ADK surfaces the lever and leaves the decision to the consumer. * Re-incurs the cold-load cost. Idempotent. * * @param overrides - Optional option overrides applied to the reload (same as {@link preload}). */ recycle(overrides?: Partial): Promise; /** * Produce the bound {@link DispatchExecutorFn} the `DispatchRunner` invokes. * * @param overrides - Option overrides layered above the constructor baseline (below `ctx.stash`). */ executor(overrides?: Partial): DispatchExecutorFn; }