/** * Translation helpers for the LiteRT-LM adapter. * * @module @nhtio/adk/batteries/llm/litert_lm/helpers * * @remarks * Two layers: * * 1. **Re-exported format-agnostic helpers** from `chat_common` — they operate on ADK primitives and * produce plain strings / trust envelopes / a JSON-Schema from a joi description, with no wire-format * coupling. LiteRT reuses them verbatim (the same way the WebLLM battery does). * 2. **LiteRT-native mappers** defined here — the wire-shape functions the chat-completions batteries * implement against the OpenAI wire format, rewritten against LiteRT's `Message` / `Tool` / * `tool_response` / `Preface` shapes. */ import { Media } from "../../../index"; import { SpooledArtifact } from "../../../index"; import { defaultRenderArtifactHandleBody } from "../chat_common/helpers"; import { renderUntrustedContent as commonRenderUntrustedContent, renderTrustedContent as commonRenderTrustedContent, renderChatCompletionsSystemPrompt, renderStandingInstructions, renderMemories, renderRetrievables, renderRetrievableSafetyDirective, renderFirstPartyRetrievables, renderThirdPartyPublicRetrievables, renderThirdPartyPrivateRetrievables, renderRetrievableHandleBody, renderThought, filterThoughts } from "../openai_chat_completions/helpers"; import type { Tokenizable } from "../../../index"; import type { ArtifactTool, Tool } from "../../../index"; import type { Message, Memory, Retrievable, Thought, ToolCall, ToolRegistry } from "../../../index"; import type { LiteRtMessage, LiteRtMessageContentItem, LiteRtTool, LiteRtPreface, LiteRtLmBucketOrder, UnsupportedMediaPolicy, DescriptionLike, JsonSchema } from "./types"; export { descriptionToChatCompletionsJsonSchema, defaultDescriptionToChatCompletionsJsonSchema, renderUntrustedContent, defaultRenderUntrustedContent, renderTrustedContent, defaultRenderTrustedContent, renderStandingInstructions, defaultRenderStandingInstructions, renderMemories, defaultRenderMemories, renderRetrievables, defaultRenderRetrievables, renderRetrievableSafetyDirective, defaultRenderRetrievableSafetyDirective, renderFirstPartyRetrievables, defaultRenderFirstPartyRetrievables, renderThirdPartyPublicRetrievables, defaultRenderThirdPartyPublicRetrievables, renderThirdPartyPrivateRetrievables, defaultRenderThirdPartyPrivateRetrievables, renderThought, defaultRenderThought, filterThoughts, defaultFilterThoughts, renderChatCompletionsSystemPrompt, defaultRenderChatCompletionsSystemPrompt, extractReasoningFields, } from "../openai_chat_completions/helpers"; export { renderArtifactHandleBody, defaultRenderArtifactHandleBody, renderRetrievableHandleBody, defaultRenderRetrievableHandleBody, looksLikeSpooledArtifact, } from "../chat_common/helpers"; export * from "../chat_common/tool_parsers"; export * from "../chat_common/reasoning_parsers"; export * from "../chat_common/lifecycle"; export * from "../chat_common/generation"; export * from "../chat_common/gpu_budget"; /** * Convert ADK {@link @nhtio/adk!Tool} / {@link @nhtio/adk!ArtifactTool} instances into LiteRT * {@link LiteRtTool} definitions. * * @remarks * Reuses {@link descriptionToChatCompletionsJsonSchema} (format-agnostic: joi `describe()` → JSON * Schema) for the `parameters` field — LiteRT's `Tool.parameters` follows JSON Schema, same as the * chat-completions `function.parameters`. */ export declare const toolsToLiteRtTools: (tools: ReadonlyArray, deps?: { descriptionToChatCompletionsJsonSchema: (d: DescriptionLike) => JsonSchema; }) => LiteRtTool[]; /** Default {@link toolsToLiteRtTools}. */ export declare const defaultToolsToLiteRtTools: (tools: ReadonlyArray, deps?: { descriptionToChatCompletionsJsonSchema: (d: DescriptionLike) => JsonSchema; }) => LiteRtTool[]; /** * Render tool definitions as a SYSTEM-PROMPT text block (the prompt-injection tool-delivery path), * rather than the native `preface.tools` field. * * @remarks * **Why this exists.** LiteRT-LM applies the model's OWN bundled chat template; for the Gemma-4 * `.litertlm` preview builds, the template's tools branch is broken — passing `preface.tools` throws * `Failed to apply template: undefined value` inside the wasm runtime (a known Gemma-4 chat-template * bug, also seen in llama.cpp / mlx-lm / LM Studio when `tools[]` hits the native template). The * portable fix every other browser runtime uses (WebLLM/MLC, Open WebUI's "default" mode) is to * describe the tools as TEXT in the system prompt and parse the model's emitted call out of the output * — which the shared {@link createAutoToolCallParser} already does (its `gemma` family handles the * decoder-stripped `call:NAME{…}` runtime form, plus hermes/pythonic as fallbacks). * * The block lists each tool's name, description, and JSON-Schema parameters, then instructs Gemma's * OWN trained call format `call:NAME{key:value, …}` — NOT the pythonic `[func(arg=value)]` form. This * matters: the LiteRT-web runtime is Gemma-only, and a real Gemma E2B/E4B run emits the * decoder-stripped `call:NAME{…}` shape natively (verified via the real-model matrix; the `gemma` * family in {@link createAutoToolCallParser} is built for exactly this). Instructing the pythonic form * instead FIGHTS the model's training — a small instruct model, caught between its trained format and a * conflicting instruction, degenerates to an unparseable hybrid (e.g. `say_i_dont_know\nreason: …`) * that no parser catches, so the "call" leaks into the answer as prose. Teaching the model the format * it already knows makes its natural output parse on the first try. A concrete example is included * because a 2B follows a shown example far more reliably than an abstract grammar. */ export declare const renderToolsAsPromptText: (tools: ReadonlyArray, deps?: { descriptionToChatCompletionsJsonSchema: (d: DescriptionLike) => JsonSchema; }) => string; /** Default {@link renderToolsAsPromptText}. */ export declare const defaultRenderToolsAsPromptText: (tools: ReadonlyArray, deps?: { descriptionToChatCompletionsJsonSchema: (d: DescriptionLike) => JsonSchema; }) => string; /** * Render a media kind/mime/filename into a LiteRT content item, or fall back per * `unsupportedMediaPolicy`. * * @remarks * The preview `.litertlm` models are text-in/text-out, and the exact multimodal content-item wire * shape is not yet stable in the published types. This maps image/audio/document/video to a * best-effort `{ type, path }`-style item when the matching modality flag is enabled, and otherwise * degrades through the shared `unsupportedMediaPolicy` (stash text / synthetic description / throw). * Verify the content-item shape against the installed `.d.ts` + a real multimodal model before * relying on the native path. */ export declare const renderMediaToLiteRtContent: (input: { media: Media; nonce: string; unsupportedMediaPolicy: UnsupportedMediaPolicy; modalityEnabled: boolean; renderUntrustedContent: typeof commonRenderUntrustedContent; renderTrustedContent: typeof commonRenderTrustedContent; warn?: (msg: string) => void; }) => Promise; /** * Render a {@link @nhtio/adk!ToolCall}'s `results` into a LiteRT `tool_response` content item. * * @remarks * A {@link @nhtio/adk!SpooledArtifact} result renders as a HANDLE (metadata + the forged `artifact_*` * tools to read it) when its `ToolCall.inline === false` — the secure default — and inline via * `asString()` only when a producer opted into `inline: true`. Applies the trust envelope (reusing the * shared `renderTrustedContent`/`renderUntrustedContent`). Media results degrade to text via * {@link renderMediaToLiteRtContent}'s fallback path (LiteRT tool responses are text-shaped). */ export declare const renderLiteRtToolResult: (input: { toolCall: ToolCall; results: Tokenizable | SpooledArtifact | SpooledArtifact[] | Media | Media[]; tool: Tool | ArtifactTool | undefined; unsupportedMediaPolicy: UnsupportedMediaPolicy; renderUntrustedContent: typeof commonRenderUntrustedContent; renderTrustedContent: typeof commonRenderTrustedContent; /** * Override for the artifact-handle body renderer (see {@link renderArtifactHandleBody}). Defaults to * the shared {@link defaultRenderArtifactHandleBody}. The adapter threads the consumer's * `helpers.renderArtifactHandleBody` here so an app can change which forged `artifact_*` reader the * model is steered toward first. */ renderArtifactHandleBody?: typeof defaultRenderArtifactHandleBody; warn?: (msg: string) => void; }) => Promise; /** Default {@link renderLiteRtToolResult}. */ export declare const defaultRenderLiteRtToolResult: (input: { toolCall: ToolCall; results: Tokenizable | SpooledArtifact | SpooledArtifact[] | Media | Media[]; tool: Tool | ArtifactTool | undefined; unsupportedMediaPolicy: UnsupportedMediaPolicy; renderUntrustedContent: typeof commonRenderUntrustedContent; renderTrustedContent: typeof commonRenderTrustedContent; /** * Override for the artifact-handle body renderer (see {@link renderArtifactHandleBody}). Defaults to * the shared {@link defaultRenderArtifactHandleBody}. The adapter threads the consumer's * `helpers.renderArtifactHandleBody` here so an app can change which forged `artifact_*` reader the * model is steered toward first. */ renderArtifactHandleBody?: typeof defaultRenderArtifactHandleBody; warn?: (msg: string) => void; }) => Promise; /** * Build the LiteRT conversation input from the ADK dispatch context buckets. * * @remarks * Maps the ADK history model onto LiteRT's `createConversation({ preface })` + per-turn * `sendMessage(messages)` shape: * * - **Leading buckets** (system prompt + standingInstructions / memories / retrievables before * `timeline` in `bucketOrder`) → a single `preface.messages` system message. * - **Tools** → `preface.tools`. * - **Timeline** (messages, surviving thoughts, tool calls — chronological) → the `messages` array * sent for this turn, each a LiteRT `Message`. Thoughts render via the shared `renderThought` into * assistant messages; tool calls render an assistant `tool_calls` message + a `tool` role message * carrying the pre-rendered `tool_response` content item. * * Returns `{ preface, messages }` for `engine.createConversation({ preface })` then * `conversation.sendMessageStreaming(messages)`. */ export declare const buildLiteRtConversationInput: (input: { systemPrompt: Tokenizable; standingInstructions: Iterable; memories: Iterable; retrievables: Iterable; messages: Iterable; thoughts: Iterable; toolCalls: Iterable; tools: ToolRegistry; renderedToolCallResults: Map; /** * The live dispatch context, threaded so DYNAMIC (evaluatable) {@link Tokenizable} content resolves * against it at assembly via `.render(renderCtx)`. Optional — a static Tokenizable ignores it, and * callers outside a dispatch may omit it (the evaluator's no-context fallback applies). Typed loosely * (the primitive types the arg as `DispatchContext`; this battery does not import that contract). */ renderCtx?: unknown; bucketOrder: LiteRtLmBucketOrder; selfIdentity: string; thoughtSurfacing: "all-self" | "latest-self" | "all"; replayCompatibility: ReadonlyArray; /** * How tool definitions reach the model. `'prompt'` (default) renders them as system-prompt text (the * portable path — the Gemma-4 `.litertlm` template throws on native `tools`); `'native'` uses * `preface.tools` (the chat-template path — only for models whose template handles tools). */ toolDelivery?: "prompt" | "native"; /** * Whether the model's "thinking" mode is enabled. Passed EXPLICITLY into the chat template via * `preface.extra_context.enable_thinking` so the template never decides for itself. Defaults to * `false`. */ enableThinking?: boolean; toolsToLiteRtTools: typeof toolsToLiteRtTools; renderToolsAsPromptText?: typeof renderToolsAsPromptText; renderThought: typeof renderThought; filterThoughts: typeof filterThoughts; renderUntrustedContent: typeof commonRenderUntrustedContent; renderTrustedContent: typeof commonRenderTrustedContent; renderChatCompletionsSystemPrompt: typeof renderChatCompletionsSystemPrompt; renderStandingInstructions: typeof renderStandingInstructions; renderMemories: typeof renderMemories; renderRetrievables: typeof renderRetrievables; renderRetrievableSafetyDirective: typeof renderRetrievableSafetyDirective; renderFirstPartyRetrievables: typeof renderFirstPartyRetrievables; renderThirdPartyPublicRetrievables: typeof renderThirdPartyPublicRetrievables; renderThirdPartyPrivateRetrievables: typeof renderThirdPartyPrivateRetrievables; renderRetrievableHandleBody?: typeof renderRetrievableHandleBody; visionModalityEnabled?: boolean; audioModalityEnabled?: boolean; unsupportedMediaPolicy?: UnsupportedMediaPolicy; renderMediaToLiteRtContent?: typeof renderMediaToLiteRtContent; warn?: (msg: string) => void; }) => Promise<{ preface: LiteRtPreface; messages: LiteRtMessage[]; }>; /** Default {@link buildLiteRtConversationInput}. */ export declare const defaultBuildLiteRtConversationInput: (input: { systemPrompt: Tokenizable; standingInstructions: Iterable; memories: Iterable; retrievables: Iterable; messages: Iterable; thoughts: Iterable; toolCalls: Iterable; tools: ToolRegistry; renderedToolCallResults: Map; /** * The live dispatch context, threaded so DYNAMIC (evaluatable) {@link Tokenizable} content resolves * against it at assembly via `.render(renderCtx)`. Optional — a static Tokenizable ignores it, and * callers outside a dispatch may omit it (the evaluator's no-context fallback applies). Typed loosely * (the primitive types the arg as `DispatchContext`; this battery does not import that contract). */ renderCtx?: unknown; bucketOrder: LiteRtLmBucketOrder; selfIdentity: string; thoughtSurfacing: "all-self" | "latest-self" | "all"; replayCompatibility: ReadonlyArray; /** * How tool definitions reach the model. `'prompt'` (default) renders them as system-prompt text (the * portable path — the Gemma-4 `.litertlm` template throws on native `tools`); `'native'` uses * `preface.tools` (the chat-template path — only for models whose template handles tools). */ toolDelivery?: "prompt" | "native"; /** * Whether the model's "thinking" mode is enabled. Passed EXPLICITLY into the chat template via * `preface.extra_context.enable_thinking` so the template never decides for itself. Defaults to * `false`. */ enableThinking?: boolean; toolsToLiteRtTools: typeof toolsToLiteRtTools; renderToolsAsPromptText?: typeof renderToolsAsPromptText; renderThought: typeof renderThought; filterThoughts: typeof filterThoughts; renderUntrustedContent: typeof commonRenderUntrustedContent; renderTrustedContent: typeof commonRenderTrustedContent; renderChatCompletionsSystemPrompt: typeof renderChatCompletionsSystemPrompt; renderStandingInstructions: typeof renderStandingInstructions; renderMemories: typeof renderMemories; renderRetrievables: typeof renderRetrievables; renderRetrievableSafetyDirective: typeof renderRetrievableSafetyDirective; renderFirstPartyRetrievables: typeof renderFirstPartyRetrievables; renderThirdPartyPublicRetrievables: typeof renderThirdPartyPublicRetrievables; renderThirdPartyPrivateRetrievables: typeof renderThirdPartyPrivateRetrievables; renderRetrievableHandleBody?: typeof renderRetrievableHandleBody; visionModalityEnabled?: boolean; audioModalityEnabled?: boolean; unsupportedMediaPolicy?: UnsupportedMediaPolicy; renderMediaToLiteRtContent?: typeof renderMediaToLiteRtContent; warn?: (msg: string) => void; }) => Promise<{ preface: LiteRtPreface; messages: LiteRtMessage[]; }>; /** * A streaming accumulator over LiteRT `ReadableStream` chunks. * * @remarks * Each chunk is a partial {@link LiteRtMessage}. The `@litert-lm/core` v0.13.1 JS runtime is * **text-in / text-out**: it emits `content` only and does NOT populate `tool_calls` or `channels` on * output (those wire fields exist for feeding history back IN). Tool calls and reasoning come out as * **raw text inside `content`**, in the model family's own format — the adapter parses them out after * the stream drains via the shared tool-call / reasoning parser layer. * * So the accumulator collects assistant `content` text only, handling BOTH delivery conventions — * incremental deltas (append) and full-accumulated snapshots (replace) — by detecting whether the * incoming value extends the running buffer (`startsWith`) or is a fresh fragment. It also tolerates * `content` arriving as a `MessageContentItem[]` by concatenating the text items. */ export interface LiteRtStreamAccumulator { /** Feed one chunk; returns the newly-appended content text (for incremental reporting). */ feed(chunk: LiteRtMessage): { contentDelta: string; }; /** Final assembled content text. */ content(): string; } /** Create a {@link LiteRtStreamAccumulator}. */ export declare const createLiteRtStreamAccumulator: () => LiteRtStreamAccumulator; /** Default {@link createLiteRtStreamAccumulator}. */ export declare const defaultCreateLiteRtStreamAccumulator: () => LiteRtStreamAccumulator;