{"version":3,"file":"gpu_budget-Bi2nUUil.mjs","names":[],"sources":["../src/batteries/llm/chat_common/reasoning_parsers.ts","../src/batteries/llm/chat_common/generation.ts","../src/batteries/llm/chat_common/gpu_budget.ts"],"sourcesContent":["/**\n * Runtime-agnostic reasoning/thinking text parsers for text-only LLM batteries.\n *\n * @remarks\n * **Why this exists.** Text-only on-device runtimes (transformers.js, LiteRT-LM v0.13.1) emit a\n * reasoning model's chain-of-thought as **raw text inside the assistant message**, delimited in a\n * format specific to the model family — not as a structured `reasoning` field the way OpenAI-style\n * providers do (those are handled by `extractReasoningFields`). To surface that thinking as ADK\n * {@link @nhtio/adk!Thought}s rather than leaking `<think>…</think>` markup into the visible answer,\n * the battery must parse it out of the text.\n *\n * Same shape as the tool-call parser layer: one parser per family, anchored on a literal marker, run\n * post-hoc. The bundled defaults cover the dominant conventions; `'auto'` tries them in order and a\n * custom {@link ReasoningParserFn} is the escape hatch.\n *\n * NOT `@module`-tagged: private to the bundled LLM batteries, re-exported through their public\n * surfaces.\n */\n\n// ─── Contract ─────────────────────────────────────────────────────────────────────────────────────\n\n/**\n * The result of running a {@link ReasoningParserFn} over assistant text.\n *\n * @remarks\n * `reasoning` holds each extracted thinking trace in document order; `cleanedText` is the prose with\n * every consumed reasoning span removed and trimmed. On no-match a parser MUST return\n * `{ reasoning: [], cleanedText: rawText }` verbatim.\n */\nexport interface ReasoningParseResult {\n  /** Each extracted thinking trace, in document order. Empty when no reasoning was found. */\n  reasoning: string[]\n  /** The prose with every consumed reasoning span removed; equals the input on no-match. */\n  cleanedText: string\n}\n\n/** A synchronous reasoning text parser. */\nexport type ReasoningParserFn = (rawText: string) => ReasoningParseResult\n\n/** The bundled reasoning parser names, plus `'auto'` (try-all) and `'none'` (disable). */\nexport type ReasoningParserName =\n  | 'auto'\n  | 'think_tag'\n  | 'harmony_analysis'\n  | 'gemma_channel'\n  | 'none'\n\n/**\n * Options shared by the bundled reasoning parsers.\n *\n * @remarks\n * `orphanRecovery` (default `true`) controls whether an **unpaired** reasoning marker is recovered by\n * inferring the missing half from the pseudo-streaming order, rather than being left to leak into the\n * visible answer. A real-world gemma-4-E4B WebGPU quant \"randomly emits `</think>`\" with no matching\n * open; because generation is start→end, a lone close implies the block opened at the previous close\n * (or start-of-output), and a lone open with no close implies reasoning ran to end-of-stream. Turn this\n * off for strict pair-only behaviour (markers without a matching partner are left verbatim).\n */\nexport interface ReasoningParserOptions {\n  /** Recover unpaired markers by inferring the missing half (default `true`). */\n  orphanRecovery?: boolean\n}\n\n// ─── Shared ───────────────────────────────────────────────────────────────────────────────────────\n\nconst NO_MATCH = (rawText: string): ReasoningParseResult => ({\n  reasoning: [],\n  cleanedText: rawText,\n})\n\nconst removeSpans = (text: string, spans: Array<[number, number]>): string => {\n  let out = text\n  for (const [start, end] of [...spans].sort((a, b) => b[0] - a[0])) {\n    out = out.slice(0, start) + out.slice(end)\n  }\n  return out.replace(/\\n{3,}/g, '\\n\\n').trim()\n}\n\n/**\n * A literal open/close marker pair for a reasoning family. `openRe` is a `g`-flagged regex (so the open\n * marker can carry trailing attributes, e.g. gemma's `<|channel>thought\\n`); `close` is a literal\n * string. After each match the captured trace is the text strictly between the open match's end and the\n * close.\n */\ninterface ReasoningMarkers {\n  /** Global-flagged regex matching the OPEN marker (its full match is consumed). */\n  openRe: RegExp\n  /** Literal CLOSE marker string. */\n  close: string\n}\n\n/**\n * Collect reasoning spans, recovering UNPAIRED markers by inferring the missing half from the\n * pseudo-streaming order (see {@link ReasoningParserOptions}). Algorithm, in precedence:\n *\n * 1. **Paired** spans first — each OPEN whose following text contains a CLOSE forms a complete span\n *    `[openStart, closeEnd]`; the trace is the text between. (Unchanged from the strict path.)\n * 2. **Lone closes** — in the text NOT covered by a pair, scan left-to-right for CLOSE markers that\n *    have no preceding OPEN. Each spans `[cursor, closeEnd]` where `cursor` starts at start-of-text and\n *    advances to each consumed close, so `A </c> B </c> C` → traces `A`, `B` and answer `C` (the second\n *    orphan-close opens at the first close, not back at 0).\n * 3. **Lone open** — a final OPEN with no following CLOSE spans `[openStart, end-of-text]` (truncated\n *    stream).\n *\n * When `orphanRecovery` is false this collapses to strict paired-only behaviour (identical to the old\n * `collect`). Returns NO_MATCH only when nothing — no pair, no orphan — was found.\n */\nconst collectWithOrphans = (\n  rawText: string,\n  markers: ReasoningMarkers,\n  orphanRecovery: boolean\n): ReasoningParseResult => {\n  const { openRe, close } = markers\n  const reasoning: string[] = []\n  const spans: Array<[number, number]> = []\n\n  // Reset lastIndex defensively (these regexes are module-level and `g`-flagged).\n  openRe.lastIndex = 0\n\n  // Walk the text once. At each step, find the next OPEN and the next CLOSE from the cursor.\n  // `cursor` is the start of the not-yet-consumed remainder.\n  let cursor = 0\n  // Track whether we are currently \"inside\" an implied/explicit open. For strict pairing we only\n  // consume an open when a close follows it.\n  while (cursor <= rawText.length) {\n    openRe.lastIndex = cursor\n    const openMatch = openRe.exec(rawText)\n    const openStart = openMatch ? openMatch.index : -1\n    const openEnd = openMatch ? openMatch.index + openMatch[0].length : -1\n    const closeStart = rawText.indexOf(close, cursor)\n\n    if (closeStart !== -1 && (openStart === -1 || closeStart < openStart)) {\n      // A CLOSE appears before the next OPEN (or there is no further OPEN). This is an ORPHAN close:\n      // the implied open is the cursor (start-of-remainder == previous close position).\n      if (!orphanRecovery) {\n        // Strict mode: an unpaired close is not consumed — advance past it untouched.\n        cursor = closeStart + close.length\n        continue\n      }\n      const trace = rawText.slice(cursor, closeStart).trim()\n      if (trace.length > 0) reasoning.push(trace)\n      spans.push([cursor, closeStart + close.length])\n      cursor = closeStart + close.length\n      continue\n    }\n\n    if (openStart !== -1) {\n      // We have an OPEN. Look for its matching CLOSE after the open.\n      const pairedClose = rawText.indexOf(close, openEnd)\n      if (pairedClose !== -1) {\n        // Complete pair.\n        const trace = rawText.slice(openEnd, pairedClose).trim()\n        if (trace.length > 0) reasoning.push(trace)\n        spans.push([openStart, pairedClose + close.length])\n        cursor = pairedClose + close.length\n        continue\n      }\n      // Lone OPEN with no following CLOSE → truncated stream: reasoning runs to end-of-text.\n      if (!orphanRecovery) break\n      const trace = rawText.slice(openEnd).trim()\n      if (trace.length > 0) reasoning.push(trace)\n      spans.push([openStart, rawText.length])\n      break\n    }\n\n    // No further OPEN and no further CLOSE — done.\n    break\n  }\n\n  return spans.length > 0\n    ? { reasoning, cleanedText: removeSpans(rawText, spans) }\n    : NO_MATCH(rawText)\n}\n\n// ─── think_tag: <think>…</think> (Qwen3, DeepSeek-R1 — the dominant convention) ───────────────────────\n// Two delimiter shapes share this family: `<think>`/`</think>` and the `<thinking>`/`</thinking>`\n// variant. They are processed independently (a `<think>` never pairs with a `</thinking>`).\n\nconst THINK_OPEN_RE = /<think>/g\nconst THINKING_OPEN_RE = /<thinking>/g\n\n/**\n * Parse `<think>…</think>` (and the `<thinking>…</thinking>` variant) reasoning blocks — the dominant\n * convention, used by Qwen3, DeepSeek-R1, and most distilled reasoning models. Unpaired markers (a lone\n * `</think>` or a truncated `<think>`) are recovered by default — see {@link ReasoningParserOptions}.\n */\nexport const makeThinkTagReasoningParser =\n  (opts: ReasoningParserOptions = {}): ReasoningParserFn =>\n  (rawText) => {\n    const orphan = opts.orphanRecovery ?? true\n    // Run the two delimiter shapes in sequence over the running cleaned text so spans from one don't\n    // collide with the other. `<think>` first (the common form), then `<thinking>`.\n    const first = collectWithOrphans(rawText, { openRe: THINK_OPEN_RE, close: '</think>' }, orphan)\n    const second = collectWithOrphans(\n      first.cleanedText,\n      { openRe: THINKING_OPEN_RE, close: '</thinking>' },\n      orphan\n    )\n    const reasoning = [...first.reasoning, ...second.reasoning]\n    return reasoning.length > 0 || second.cleanedText !== rawText\n      ? { reasoning, cleanedText: second.cleanedText }\n      : NO_MATCH(rawText)\n  }\n\n/** Default {@link makeThinkTagReasoningParser} (orphan recovery on). */\nexport const thinkTagReasoningParser: ReasoningParserFn = makeThinkTagReasoningParser()\n\n/** Default {@link thinkTagReasoningParser}. */\nexport const defaultThinkTagReasoningParser = thinkTagReasoningParser\n\n// ─── harmony_analysis: gpt-oss Harmony analysis channel ───────────────────────────────────────────────\n\nconst HARMONY_OPEN_RE = /<\\|channel\\|>analysis\\s*<\\|message\\|>/g\n\n/**\n * Parse gpt-oss Harmony chain-of-thought on the `analysis` channel:\n * `<|channel|>analysis<|message|>…<|end|>`. (The user-visible answer is the separate `final` channel;\n * tool calls are `commentary` — handled by the tool-call parser.) Unpaired markers are recovered by\n * default — see {@link ReasoningParserOptions}.\n */\nexport const makeHarmonyAnalysisReasoningParser =\n  (opts: ReasoningParserOptions = {}): ReasoningParserFn =>\n  (rawText) =>\n    collectWithOrphans(\n      rawText,\n      { openRe: HARMONY_OPEN_RE, close: '<|end|>' },\n      opts.orphanRecovery ?? true\n    )\n\n/** Default {@link makeHarmonyAnalysisReasoningParser} (orphan recovery on). */\nexport const harmonyAnalysisReasoningParser: ReasoningParserFn =\n  makeHarmonyAnalysisReasoningParser()\n\n/** Default {@link harmonyAnalysisReasoningParser}. */\nexport const defaultHarmonyAnalysisReasoningParser = harmonyAnalysisReasoningParser\n\n// ─── gemma_channel: Gemma E2B/E4B <|channel>thought\\n…<channel|> ──────────────────────────────────────\n// Verified byte-exact against onnx-community/gemma-4-E2B-it-ONNX tokenizer_config.json. NOTE the\n// asymmetric markers: open `<|channel>thought` (no closing pipe before `>`), close `<channel|>`.\n\nconst GEMMA_OPEN_RE = /<\\|channel>thought\\b[^\\n]*\\n?/g\n\n/**\n * Parse Gemma E2B/E4B reasoning emitted on the thought channel:\n * `<|channel>thought\\n…<channel|>`. Targets the E2B/E4B delimited form (the transformers.js-runnable\n * one). Reasoning is only emitted when `<|think|>` is injected into the system prompt. Unpaired markers\n * are recovered by default — see {@link ReasoningParserOptions}.\n */\nexport const makeGemmaChannelReasoningParser =\n  (opts: ReasoningParserOptions = {}): ReasoningParserFn =>\n  (rawText) =>\n    collectWithOrphans(\n      rawText,\n      { openRe: GEMMA_OPEN_RE, close: '<channel|>' },\n      opts.orphanRecovery ?? true\n    )\n\n/** Default {@link makeGemmaChannelReasoningParser} (orphan recovery on). */\nexport const gemmaChannelReasoningParser: ReasoningParserFn = makeGemmaChannelReasoningParser()\n\n/** Default {@link gemmaChannelReasoningParser}. */\nexport const defaultGemmaChannelReasoningParser = gemmaChannelReasoningParser\n\n// ─── none ─────────────────────────────────────────────────────────────────────────────────────────\n\n/** A parser that never extracts anything — disables reasoning parsing entirely. */\nexport const noneReasoningParser: ReasoningParserFn = (rawText) => NO_MATCH(rawText)\n\n/** Default {@link noneReasoningParser}. */\nexport const defaultNoneReasoningParser = noneReasoningParser\n\n// ─── auto ─────────────────────────────────────────────────────────────────────────────────────────\n\n/** The bundled reasoning parsers keyed by name (excluding `'auto'`/`'none'`), orphan recovery ON. */\nexport const BUNDLED_REASONING_PARSERS: Readonly<\n  Record<Exclude<ReasoningParserName, 'auto' | 'none'>, ReasoningParserFn>\n> = {\n  think_tag: thinkTagReasoningParser,\n  harmony_analysis: harmonyAnalysisReasoningParser,\n  gemma_channel: gemmaChannelReasoningParser,\n}\n\n/** Build the bundled family parsers honouring {@link ReasoningParserOptions} (e.g. orphan recovery). */\nexport const buildBundledReasoningParsers = (\n  opts: ReasoningParserOptions = {}\n): Record<Exclude<ReasoningParserName, 'auto' | 'none'>, ReasoningParserFn> => ({\n  think_tag: makeThinkTagReasoningParser(opts),\n  harmony_analysis: makeHarmonyAnalysisReasoningParser(opts),\n  gemma_channel: makeGemmaChannelReasoningParser(opts),\n})\n\n/** The default `'auto'` precedence. All three are literal-marker-anchored, so order is collision-free. */\nexport const DEFAULT_REASONING_PARSER_ORDER: ReadonlyArray<\n  Exclude<ReasoningParserName, 'auto' | 'none'>\n> = ['think_tag', 'harmony_analysis', 'gemma_channel']\n\n/**\n * Compose an `'auto'` reasoning parser: run each parser in `order` until one returns a non-empty\n * `reasoning` array; that result wins. Returns no-match if none claim the text.\n */\nexport const createAutoReasoningParser = (\n  parsers: Partial<\n    Record<Exclude<ReasoningParserName, 'auto' | 'none'>, ReasoningParserFn>\n  > = BUNDLED_REASONING_PARSERS,\n  order: ReadonlyArray<\n    Exclude<ReasoningParserName, 'auto' | 'none'>\n  > = DEFAULT_REASONING_PARSER_ORDER\n): ReasoningParserFn => {\n  return (rawText) => {\n    for (const name of order) {\n      const parser = parsers[name]\n      if (!parser) continue\n      const result = parser(rawText)\n      // A parser \"claims\" the text if it extracted reasoning OR consumed/stripped any markup\n      // (e.g. an empty thought channel — strip the markers even when there's no trace).\n      if (result.reasoning.length > 0 || result.cleanedText !== rawText) return result\n    }\n    return NO_MATCH(rawText)\n  }\n}\n\n/** Default {@link createAutoReasoningParser}. */\nexport const defaultCreateAutoReasoningParser = createAutoReasoningParser\n\n/**\n * Resolve a `reasoningParser` option (a name, `'auto'`, `'none'`, or a custom fn) to a concrete\n * {@link ReasoningParserFn}.\n *\n * @param option - The option value. Defaults to `'auto'` when undefined.\n * @param parsers - Override the bundled parsers. Ignored when `opts.orphanRecovery` is set (the bundled\n *   family parsers are rebuilt with that setting); pass a custom `option` fn for full control.\n * @param opts - {@link ReasoningParserOptions}; `orphanRecovery` defaults to `true`. When `false`, the\n *   named/auto bundled parsers are rebuilt in strict pair-only mode.\n */\nexport const resolveReasoningParser = (\n  option: ReasoningParserName | ReasoningParserFn | undefined,\n  parsers: Partial<\n    Record<Exclude<ReasoningParserName, 'auto' | 'none'>, ReasoningParserFn>\n  > = BUNDLED_REASONING_PARSERS,\n  opts: ReasoningParserOptions = {}\n): ReasoningParserFn => {\n  if (typeof option === 'function') return option\n  if (option === 'none') return noneReasoningParser\n  // When orphan recovery is explicitly disabled, rebuild the bundled parsers in strict mode (and ignore\n  // a `parsers` override, which would otherwise carry the default orphan-on instances).\n  const resolved =\n    opts.orphanRecovery === false\n      ? buildBundledReasoningParsers(opts)\n      : { ...BUNDLED_REASONING_PARSERS, ...parsers }\n  if (option === undefined || option === 'auto') return createAutoReasoningParser(resolved)\n  return resolved[option] ?? noneReasoningParser\n}\n\n/** Default {@link resolveReasoningParser}. */\nexport const defaultResolveReasoningParser = resolveReasoningParser\n","/**\n * The portable, battery-agnostic GENERATION contract shared by the on-device LLM batteries.\n *\n * @remarks\n * INTERNAL to the bundled LLM batteries — intentionally NOT `@module`-tagged, so it stays private and\n * is inlined into each consumer by the bundler (the same convention as `chat_common/helpers.ts`).\n *\n * **Why this exists.** transformers.js and LiteRT-LM express the same generation concepts with different\n * option names and shapes — `maxNewTokens` vs `maxOutputTokens`, a flat `doSample`+`temperature`/`topK`/\n * `topP` vs a nested `samplerParams:{type,k,p,…}`, one `multimodal:{image,audio}` object vs two\n * `visionModalityEnabled`/`audioModalityEnabled` booleans. Swapping batteries therefore meant rewriting\n * the config. This module defines ONE canonical surface ({@link ChatGenerationOptions}) that both\n * batteries accept; each adapter maps it onto its native API via {@link resolveGenerationOptions}.\n *\n * **Contract.** The canonical fields are additive — every native field a battery already exposed\n * remains a working escape hatch. When BOTH a canonical field and its native equivalent are set, the\n * **canonical value wins** (it is the portable intent; the native field is the low-level override that\n * the resolver only consults when the canonical one is absent). All values carry the same\n * deterministic-friendly defaults across batteries (greedy, temp 0.7, top-k 40, top-p 0.95, max 1024).\n */\n\n/**\n * The portable sampler strategy. Maps to each battery's native mechanism:\n * - `'greedy'` — deterministic argmax. transformers.js `do_sample:false`; LiteRT `SamplerType.GREEDY`\n *   (which is top-1, so `k` is forced to 1).\n * - `'top-k'` — sample from the top-`k` logits. transformers.js `do_sample:true`+`top_k`; LiteRT\n *   `SamplerType.TOP_K`+`k`.\n * - `'top-p'` — nucleus sampling. transformers.js `do_sample:true`+`top_p`; LiteRT `SamplerType.TOP_P`+`p`.\n */\nexport type ChatSampler = 'greedy' | 'top-k' | 'top-p'\n\n/** The canonical generation options both on-device batteries accept. Every field optional + defaulted. */\nexport interface ChatGenerationOptions {\n  /**\n   * Maximum tokens to GENERATE this turn. The portable spelling of transformers.js `maxNewTokens` /\n   * LiteRT `maxOutputTokens`. Default `1024`.\n   */\n  maxTokens?: number\n  /** Sampler strategy (default `'greedy'` — deterministic). See {@link ChatSampler}. */\n  sampler?: ChatSampler\n  /** Sampling temperature, used when `sampler` is `'top-k'`/`'top-p'`. Always pinned. Default `0.7`. */\n  temperature?: number\n  /** Top-K cutoff, used when `sampler` is `'top-k'`. Default `40`. */\n  topK?: number\n  /** Top-P (nucleus) cutoff, used when `sampler` is `'top-p'`. Default `0.95`. */\n  topP?: number\n  /** RNG seed for reproducible sampling (best-effort; only honoured by batteries that expose it). */\n  seed?: number\n  /**\n   * Whether to enable the model's \"thinking\"/reasoning mode, passed EXPLICITLY to the chat template.\n   * Default `false` — many reasoning templates default thinking ON and burn the budget. Already shared\n   * by both batteries under this exact name; restated here so the whole generation contract is in one place.\n   */\n  enableThinking?: boolean\n  /**\n   * Enable multimodal input by kind. The portable spelling of transformers.js `multimodal:{image,audio}`\n   * / LiteRT `visionModalityEnabled`+`audioModalityEnabled`.\n   */\n  multimodal?: { image?: boolean; audio?: boolean }\n}\n\n/** The resolved, fully-defaulted generation config the adapters consume. */\nexport interface ResolvedGenerationOptions {\n  /** Max tokens to generate this turn. */\n  maxTokens: number\n  /** The resolved sampler strategy. */\n  sampler: ChatSampler\n  /** Sampling temperature (used by `'top-k'`/`'top-p'`). */\n  temperature: number\n  /** Top-K cutoff (used by `'top-k'`). */\n  topK: number\n  /** Top-P (nucleus) cutoff (used by `'top-p'`). */\n  topP: number\n  /** RNG seed, when provided. */\n  seed?: number\n  /** Whether thinking/reasoning mode is enabled. */\n  enableThinking: boolean\n  /** Resolved multimodal-input flags by kind. */\n  multimodal: { image: boolean; audio: boolean }\n}\n\n/** Deterministic-friendly defaults, identical across batteries. */\nexport const GENERATION_DEFAULTS: ResolvedGenerationOptions = {\n  maxTokens: 1024,\n  sampler: 'greedy',\n  temperature: 0.7,\n  topK: 40,\n  topP: 0.95,\n  enableThinking: false,\n  multimodal: { image: false, audio: false },\n}\n\n/** Pick the first defined value (canonical-wins precedence is encoded by argument order at call sites). */\nconst firstDefined = <T>(...values: Array<T | undefined>): T | undefined => {\n  for (const v of values) if (v !== undefined) return v\n  return undefined\n}\n\n/**\n * Resolve the canonical {@link ChatGenerationOptions} (already merged across option layers) into a\n * fully-defaulted {@link ResolvedGenerationOptions}. Native per-battery fallbacks are passed via\n * `nativeFallbacks` and consulted ONLY when the canonical field is absent (canonical wins). The adapter\n * then maps the resolved shape onto its runtime API.\n *\n * @param canonical - The canonical fields from the merged adapter options.\n * @param nativeFallbacks - Battery-native equivalents to fall back to when a canonical field is unset\n *   (e.g. transformers.js `maxNewTokens`, LiteRT `maxOutputTokens`). Each is consulted second.\n */\nexport const resolveGenerationOptions = (\n  canonical: ChatGenerationOptions,\n  nativeFallbacks: {\n    maxTokens?: number\n    sampler?: ChatSampler\n    temperature?: number\n    topK?: number\n    topP?: number\n    seed?: number\n    enableThinking?: boolean\n    multimodal?: { image?: boolean; audio?: boolean }\n  } = {}\n): ResolvedGenerationOptions => {\n  const mm = firstDefined(canonical.multimodal, nativeFallbacks.multimodal) ?? {}\n  return {\n    maxTokens:\n      firstDefined(canonical.maxTokens, nativeFallbacks.maxTokens) ?? GENERATION_DEFAULTS.maxTokens,\n    sampler:\n      firstDefined(canonical.sampler, nativeFallbacks.sampler) ?? GENERATION_DEFAULTS.sampler,\n    temperature:\n      firstDefined(canonical.temperature, nativeFallbacks.temperature) ??\n      GENERATION_DEFAULTS.temperature,\n    topK: firstDefined(canonical.topK, nativeFallbacks.topK) ?? GENERATION_DEFAULTS.topK,\n    topP: firstDefined(canonical.topP, nativeFallbacks.topP) ?? GENERATION_DEFAULTS.topP,\n    seed: firstDefined(canonical.seed, nativeFallbacks.seed),\n    enableThinking:\n      firstDefined(canonical.enableThinking, nativeFallbacks.enableThinking) ??\n      GENERATION_DEFAULTS.enableThinking,\n    multimodal: {\n      image: mm.image ?? GENERATION_DEFAULTS.multimodal.image,\n      audio: mm.audio ?? GENERATION_DEFAULTS.multimodal.audio,\n    },\n  }\n}\n\n/** Default {@link resolveGenerationOptions}. */\nexport const defaultResolveGenerationOptions = resolveGenerationOptions\n","/**\n * WebGPU memory observability for the on-device LLM batteries — surface the budget, don't impose a cap.\n *\n * @remarks\n * This module is INTERNAL to the bundled LLM batteries (no `@module` tag → no public subpath; inlined\n * into each consumer by the bundler). It is re-exported from each on-device battery's public barrel.\n *\n * **Why this exists.** On the WebGPU execution provider, ONNX Runtime Web (which transformers.js and\n * LiteRT-LM both drive in the browser) keeps a per-size **buffer freelist** inside its JSEP GPU data\n * manager: a freed activation buffer is NOT destroyed, it is parked in a bucket for reuse. The bucket\n * sizing follows the largest tensor shape the model has run, so as prompts grow the retained\n * working-set grows with them — a **high-water-mark**, not an unbounded count leak. The pool is flushed\n * only when every session of the model is released (`InferenceSession` disposal → ORT clears the cache\n * when `sessionCount === 0`; see microsoft/onnxruntime#22490). There is **no public ORT flag** to bound\n * or flush the freelist mid-life, and `freeDimensionOverrides` cannot pin a KV-cached autoregressive\n * decoder's dynamic seq/past dims (it throws `ShapeInferenceError`). So past a point, a long-context\n * turn exhausts the device budget and ORT throws `Failed to allocate memory for buffer mapping`.\n *\n * **The ADK stance is to SURFACE this, not to silently clamp the consumer's request.** A battery that\n * auto-caps the context window to \"protect\" the caller is imposing a restriction; the ADK makes the\n * trade-off VISIBLE (this module's probe + the opt-in live instrument) and ACTIONABLE (a typed,\n * catchable error — see {@link isGpuOutOfMemoryError} and the battery's `E_*_GPU_OUT_OF_MEMORY`), and\n * leaves the levers REACHABLE (`session_options` passthrough, an explicit `recycle()` that triggers the\n * freelist flush). The DEFAULT stays \"honor what the caller asked for\"; the application layer decides\n * how to react (warn + retry smaller, recycle, switch model, …).\n */\n\n/**\n * A snapshot of the WebGPU device's memory budget, as reported by the adapter/device limits.\n *\n * @remarks\n * All sizes are bytes. `maxBufferSize` and `maxStorageBufferBindingSize` are the hard per-allocation\n * ceilings a single ONNX tensor buffer cannot exceed — the practical wall an over-large context window\n * hits first (e.g. ~4 GiB on Apple Metal). These are LIMITS, not live usage; pair with\n * {@link GpuBufferInstrument} for live high-water-mark tracking.\n */\nexport interface GpuBudget {\n  /** Largest single `GPUBuffer` the device will allocate, in bytes (`GPUSupportedLimits.maxBufferSize`). */\n  maxBufferBytes: number\n  /** Largest storage-buffer binding, in bytes (`GPUSupportedLimits.maxStorageBufferBindingSize`). */\n  maxStorageBufferBindingBytes: number\n  /** Best-effort adapter description (vendor/architecture/device), when the runtime exposes it. */\n  adapterInfo?: { vendor?: string; architecture?: string; device?: string; description?: string }\n  /** `true` when these numbers came from a real `navigator.gpu` device; `false`/absent ⇒ unavailable. */\n  available: boolean\n}\n\n/**\n * Detect the WebGPU \"out of memory / failed to allocate\" family of errors from a message string.\n *\n * @remarks\n * ORT-web surfaces GPU exhaustion through several non-obvious signatures depending on where the\n * allocation failed (buffer mapping, device loss, unaligned-access fallback, Dawn validation). This\n * matcher is the single source of truth both on-device batteries use to translate a raw provider throw\n * into a typed, catchable `E_*_GPU_OUT_OF_MEMORY`. Substring/case-insensitive; intentionally broad on\n * the well-known phrases but anchored enough not to swallow unrelated errors. Verified against the\n * signatures observed in the flagship agent's WebGPU OOM repro (Apple Metal, 4 GiB budget).\n *\n * @param message - The error message (or any stringifiable value's `String(...)` form).\n * @returns `true` when the message matches a known GPU-exhaustion signature.\n */\nexport const isGpuOutOfMemoryError = (message: string): boolean =>\n  // WebGPU device-buffer exhaustion (the VRAM ceiling) ...\n  /failed to allocate memory for buffer mapping|out of memory|device is lost|requestdevicefailed|operation does not support unaligned accesses|failed to create.*buffer|mapasync|exceeds the max buffer size limit|buffer is bound to|memory copy/i.test(\n    message\n  ) ||\n  // ... AND WASM linear-memory exhaustion (the ONNX-runtime heap, hit first at large context windows on\n  // the WASM/JSEP path). transformers.js rethrows ORT's `RuntimeError: memory access out of bounds` /\n  // emscripten `Cannot enlarge memory` / `abort(OOM)` verbatim. Same capacity signal, same remedy\n  // (reduce the window / recycle / smaller model), so it maps to the same typed error.\n  /memory access out of bounds|cannot enlarge memory|abort\\(oom\\)|out of bounds memory access/i.test(\n    message\n  )\n\n/** Minimal structural view of the bits of `navigator.gpu` we read — keeps the module env-neutral. */\ninterface NavigatorGpuLike {\n  gpu?: {\n    requestAdapter: (opts?: unknown) => Promise<GpuAdapterLike | null | undefined>\n  }\n}\ninterface GpuAdapterLike {\n  readonly limits?: {\n    readonly maxBufferSize?: number\n    readonly maxStorageBufferBindingSize?: number\n  }\n  readonly info?: {\n    vendor?: string\n    architecture?: string\n    device?: string\n    description?: string\n  }\n  requestAdapterInfo?: () => Promise<GpuAdapterLike['info']>\n}\n\n/**\n * Probe the host WebGPU device for its memory budget (per-allocation ceilings + adapter info).\n *\n * @remarks\n * Reads `navigator.gpu` adapter limits — this is OBSERVABILITY ONLY; it allocates nothing and changes\n * no behavior. Returns `{ available: false }` (with zeroed sizes) when WebGPU is absent (Node, a\n * browser without the API, or a refused adapter) so callers can branch without try/catch. Best-effort\n * and never throws. A battery emits the result through its lifecycle hook so the application can show\n * \"you have ~N GiB of GPU budget\" and let the user choose a context window accordingly.\n *\n * @param nav - Injectable `navigator`-like object for tests; defaults to the global `navigator`.\n * @returns The {@link GpuBudget} snapshot.\n */\nexport const probeGpuBudget = async (nav?: NavigatorGpuLike): Promise<GpuBudget> => {\n  const unavailable: GpuBudget = {\n    maxBufferBytes: 0,\n    maxStorageBufferBindingBytes: 0,\n    available: false,\n  }\n  const n =\n    nav ??\n    (typeof navigator !== 'undefined' ? (navigator as unknown as NavigatorGpuLike) : undefined)\n  if (!n?.gpu?.requestAdapter) return unavailable\n  try {\n    const adapter = await n.gpu.requestAdapter()\n    if (!adapter) return unavailable\n    const limits = adapter.limits ?? {}\n    // `info` is sync on recent specs; older builds expose `requestAdapterInfo()`. Best-effort either way.\n    let info = adapter.info\n    if (!info && typeof adapter.requestAdapterInfo === 'function') {\n      info = await adapter.requestAdapterInfo().catch(() => undefined)\n    }\n    return {\n      maxBufferBytes: Number(limits.maxBufferSize ?? 0),\n      maxStorageBufferBindingBytes: Number(limits.maxStorageBufferBindingSize ?? 0),\n      ...(info\n        ? {\n            adapterInfo: {\n              ...(info.vendor ? { vendor: info.vendor } : {}),\n              ...(info.architecture ? { architecture: info.architecture } : {}),\n              ...(info.device ? { device: info.device } : {}),\n              ...(info.description ? { description: info.description } : {}),\n            },\n          }\n        : {}),\n      available: true,\n    }\n  } catch {\n    return unavailable\n  }\n}\n\n/** A live GPU-memory sample taken by a {@link GpuBufferInstrument}. */\nexport interface GpuBufferSample {\n  /** Total `createBuffer` calls observed since instrumentation began. */\n  created: number\n  /** Total `buffer.destroy()` calls observed. */\n  destroyed: number\n  /** Currently-live buffer count (`created - destroyed`). */\n  live: number\n  /** Currently-live buffer bytes (the number that climbs as the freelist retains larger working sets). */\n  liveBytes: number\n  /** Peak live bytes seen — the high-water-mark that, against {@link GpuBudget}, predicts the OOM. */\n  peakBytes: number\n}\n\n/** A handle returned by {@link instrumentGpuBuffers}: read live samples, then `uninstall()` to restore. */\nexport interface GpuBufferInstrument {\n  /** Take a live sample of GPU buffer usage. */\n  sample: () => GpuBufferSample\n  /** Restore the original `createBuffer`/`destroy` (idempotent). Call when done measuring. */\n  uninstall: () => void\n}\n\n/**\n * Install an OPT-IN live GPU-buffer instrument by wrapping `GPUDevice.prototype.createBuffer`.\n *\n * @remarks\n * This is the packaged, productionized form of the diagnostic probe used to PROVE the ORT-web freelist\n * high-water-mark growth. It wraps `createBuffer` (and the returned buffer's `destroy`) on the device\n * prototype to tally live/peak GPU bytes, so an application can watch the working-set climb toward the\n * {@link GpuBudget} ceiling and surface \"you're at X of Y GiB\" — turning the invisible cliff into a\n * gauge the user can act on BEFORE the OOM.\n *\n * It is **purely observational and strictly opt-in**: nothing in the batteries installs it. Wrapping a\n * global prototype is intrusive, so this is never on by default — an application enables it consciously\n * (typically only in a debug/diagnostics build). Always `uninstall()` when done. Returns a no-op\n * instrument when WebGPU is unavailable (Node / no API).\n *\n * @param globalScope - Injectable global for tests; defaults to `globalThis`.\n * @returns A {@link GpuBufferInstrument}; `sample()` returns zeroes when WebGPU is unavailable.\n */\nexport const instrumentGpuBuffers = (globalScope?: unknown): GpuBufferInstrument => {\n  const g = (globalScope ?? (typeof globalThis !== 'undefined' ? globalThis : {})) as {\n    GPUDevice?: { prototype?: Record<string, unknown> }\n  }\n  const proto = g.GPUDevice?.prototype\n  const state: GpuBufferSample = {\n    created: 0,\n    destroyed: 0,\n    live: 0,\n    liveBytes: 0,\n    peakBytes: 0,\n  }\n  const noop: GpuBufferInstrument = {\n    sample: () => ({ ...state }),\n    uninstall: () => {},\n  }\n  if (!proto || typeof proto.createBuffer !== 'function') return noop\n\n  const origCreate = proto.createBuffer as (this: unknown, desc: { size?: number }) => unknown\n  const wrappedCreate = function (this: unknown, desc: { size?: number }): unknown {\n    const buf = origCreate.call(this, desc) as { destroy?: () => void } | null\n    const size = Number(desc?.size ?? 0)\n    state.created += 1\n    state.live += 1\n    state.liveBytes += size\n    if (state.liveBytes > state.peakBytes) state.peakBytes = state.liveBytes\n    if (buf && typeof buf.destroy === 'function') {\n      const origDestroy = buf.destroy.bind(buf)\n      buf.destroy = () => {\n        state.destroyed += 1\n        state.live -= 1\n        state.liveBytes -= size\n        return origDestroy()\n      }\n    }\n    return buf\n  }\n  proto.createBuffer = wrappedCreate as unknown as Record<string, unknown>['createBuffer']\n\n  let installed = true\n  return {\n    sample: () => ({ ...state }),\n    uninstall: () => {\n      if (!installed) return\n      // Only restore if no one wrapped on top of us in the meantime.\n      if (proto.createBuffer === (wrappedCreate as unknown)) {\n        proto.createBuffer = origCreate as unknown as Record<string, unknown>['createBuffer']\n      }\n      installed = false\n    },\n  }\n}\n"],"mappings":";AAiEA,IAAM,YAAY,aAA2C;CAC3D,WAAW,CAAC;CACZ,aAAa;AACf;AAEA,IAAM,eAAe,MAAc,UAA2C;CAC5E,IAAI,MAAM;CACV,KAAK,MAAM,CAAC,OAAO,QAAQ,CAAC,GAAG,KAAK,EAAE,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,GAC9D,MAAM,IAAI,MAAM,GAAG,KAAK,IAAI,IAAI,MAAM,GAAG;CAE3C,OAAO,IAAI,QAAQ,WAAW,MAAM,EAAE,KAAK;AAC7C;;;;;;;;;;;;;;;;;AA+BA,IAAM,sBACJ,SACA,SACA,mBACyB;CACzB,MAAM,EAAE,QAAQ,UAAU;CAC1B,MAAM,YAAsB,CAAC;CAC7B,MAAM,QAAiC,CAAC;CAGxC,OAAO,YAAY;CAInB,IAAI,SAAS;CAGb,OAAO,UAAU,QAAQ,QAAQ;EAC/B,OAAO,YAAY;EACnB,MAAM,YAAY,OAAO,KAAK,OAAO;EACrC,MAAM,YAAY,YAAY,UAAU,QAAQ;EAChD,MAAM,UAAU,YAAY,UAAU,QAAQ,UAAU,GAAG,SAAS;EACpE,MAAM,aAAa,QAAQ,QAAQ,OAAO,MAAM;EAEhD,IAAI,eAAe,OAAO,cAAc,MAAM,aAAa,YAAY;GAGrE,IAAI,CAAC,gBAAgB;IAEnB,SAAS,aAAa,MAAM;IAC5B;GACF;GACA,MAAM,QAAQ,QAAQ,MAAM,QAAQ,UAAU,EAAE,KAAK;GACrD,IAAI,MAAM,SAAS,GAAG,UAAU,KAAK,KAAK;GAC1C,MAAM,KAAK,CAAC,QAAQ,aAAa,MAAM,MAAM,CAAC;GAC9C,SAAS,aAAa,MAAM;GAC5B;EACF;EAEA,IAAI,cAAc,IAAI;GAEpB,MAAM,cAAc,QAAQ,QAAQ,OAAO,OAAO;GAClD,IAAI,gBAAgB,IAAI;IAEtB,MAAM,QAAQ,QAAQ,MAAM,SAAS,WAAW,EAAE,KAAK;IACvD,IAAI,MAAM,SAAS,GAAG,UAAU,KAAK,KAAK;IAC1C,MAAM,KAAK,CAAC,WAAW,cAAc,MAAM,MAAM,CAAC;IAClD,SAAS,cAAc,MAAM;IAC7B;GACF;GAEA,IAAI,CAAC,gBAAgB;GACrB,MAAM,QAAQ,QAAQ,MAAM,OAAO,EAAE,KAAK;GAC1C,IAAI,MAAM,SAAS,GAAG,UAAU,KAAK,KAAK;GAC1C,MAAM,KAAK,CAAC,WAAW,QAAQ,MAAM,CAAC;GACtC;EACF;EAGA;CACF;CAEA,OAAO,MAAM,SAAS,IAClB;EAAE;EAAW,aAAa,YAAY,SAAS,KAAK;CAAE,IACtD,SAAS,OAAO;AACtB;AAMA,IAAM,gBAAgB;AACtB,IAAM,mBAAmB;;;;;;AAOzB,IAAa,+BACV,OAA+B,CAAC,OAChC,YAAY;CACX,MAAM,SAAS,KAAK,kBAAkB;CAGtC,MAAM,QAAQ,mBAAmB,SAAS;EAAE,QAAQ;EAAe,OAAO;CAAW,GAAG,MAAM;CAC9F,MAAM,SAAS,mBACb,MAAM,aACN;EAAE,QAAQ;EAAkB,OAAO;CAAc,GACjD,MACF;CACA,MAAM,YAAY,CAAC,GAAG,MAAM,WAAW,GAAG,OAAO,SAAS;CAC1D,OAAO,UAAU,SAAS,KAAK,OAAO,gBAAgB,UAClD;EAAE;EAAW,aAAa,OAAO;CAAY,IAC7C,SAAS,OAAO;AACtB;;AAGF,IAAa,0BAA6C,4BAA4B;;AAGtF,IAAa,iCAAiC;AAI9C,IAAM,kBAAkB;;;;;;;AAQxB,IAAa,sCACV,OAA+B,CAAC,OAChC,YACC,mBACE,SACA;CAAE,QAAQ;CAAiB,OAAO;AAAU,GAC5C,KAAK,kBAAkB,IACzB;;AAGJ,IAAa,iCACX,mCAAmC;;AAGrC,IAAa,wCAAwC;AAMrD,IAAM,gBAAgB;;;;;;;AAQtB,IAAa,mCACV,OAA+B,CAAC,OAChC,YACC,mBACE,SACA;CAAE,QAAQ;CAAe,OAAO;AAAa,GAC7C,KAAK,kBAAkB,IACzB;;AAGJ,IAAa,8BAAiD,gCAAgC;;AAG9F,IAAa,qCAAqC;;AAKlD,IAAa,uBAA0C,YAAY,SAAS,OAAO;;AAGnF,IAAa,6BAA6B;;AAK1C,IAAa,4BAET;CACF,WAAW;CACX,kBAAkB;CAClB,eAAe;AACjB;;AAGA,IAAa,gCACX,OAA+B,CAAC,OAC8C;CAC9E,WAAW,4BAA4B,IAAI;CAC3C,kBAAkB,mCAAmC,IAAI;CACzD,eAAe,gCAAgC,IAAI;AACrD;;AAGA,IAAa,iCAET;CAAC;CAAa;CAAoB;AAAe;;;;;AAMrD,IAAa,6BACX,UAEI,2BACJ,QAEI,mCACkB;CACtB,QAAQ,YAAY;EAClB,KAAK,MAAM,QAAQ,OAAO;GACxB,MAAM,SAAS,QAAQ;GACvB,IAAI,CAAC,QAAQ;GACb,MAAM,SAAS,OAAO,OAAO;GAG7B,IAAI,OAAO,UAAU,SAAS,KAAK,OAAO,gBAAgB,SAAS,OAAO;EAC5E;EACA,OAAO,SAAS,OAAO;CACzB;AACF;;AAGA,IAAa,mCAAmC;;;;;;;;;;;AAYhD,IAAa,0BACX,QACA,UAEI,2BACJ,OAA+B,CAAC,MACV;CACtB,IAAI,OAAO,WAAW,YAAY,OAAO;CACzC,IAAI,WAAW,QAAQ,OAAO;CAG9B,MAAM,WACJ,KAAK,mBAAmB,QACpB,6BAA6B,IAAI,IACjC;EAAE,GAAG;EAA2B,GAAG;CAAQ;CACjD,IAAI,WAAW,KAAA,KAAa,WAAW,QAAQ,OAAO,0BAA0B,QAAQ;CACxF,OAAO,SAAS,WAAW;AAC7B;;AAGA,IAAa,gCAAgC;;;;AChR7C,IAAa,sBAAiD;CAC5D,WAAW;CACX,SAAS;CACT,aAAa;CACb,MAAM;CACN,MAAM;CACN,gBAAgB;CAChB,YAAY;EAAE,OAAO;EAAO,OAAO;CAAM;AAC3C;;AAGA,IAAM,gBAAmB,GAAG,WAAgD;CAC1E,KAAK,MAAM,KAAK,QAAQ,IAAI,MAAM,KAAA,GAAW,OAAO;AAEtD;;;;;;;;;;;AAYA,IAAa,4BACX,WACA,kBASI,CAAC,MACyB;CAC9B,MAAM,KAAK,aAAa,UAAU,YAAY,gBAAgB,UAAU,KAAK,CAAC;CAC9E,OAAO;EACL,WACE,aAAa,UAAU,WAAW,gBAAgB,SAAS,KAAK,oBAAoB;EACtF,SACE,aAAa,UAAU,SAAS,gBAAgB,OAAO,KAAK,oBAAoB;EAClF,aACE,aAAa,UAAU,aAAa,gBAAgB,WAAW,KAC/D,oBAAoB;EACtB,MAAM,aAAa,UAAU,MAAM,gBAAgB,IAAI,KAAK,oBAAoB;EAChF,MAAM,aAAa,UAAU,MAAM,gBAAgB,IAAI,KAAK,oBAAoB;EAChF,MAAM,aAAa,UAAU,MAAM,gBAAgB,IAAI;EACvD,gBACE,aAAa,UAAU,gBAAgB,gBAAgB,cAAc,KACrE,oBAAoB;EACtB,YAAY;GACV,OAAO,GAAG,SAAS,oBAAoB,WAAW;GAClD,OAAO,GAAG,SAAS,oBAAoB,WAAW;EACpD;CACF;AACF;;AAGA,IAAa,kCAAkC;;;;;;;;;;;;;;;;;ACnF/C,IAAa,yBAAyB,YAEpC,kPAAkP,KAChP,OACF,KAKA,8FAA8F,KAC5F,OACF;;;;;;;;;;;;;;AAmCF,IAAa,iBAAiB,OAAO,QAA+C;CAClF,MAAM,cAAyB;EAC7B,gBAAgB;EAChB,8BAA8B;EAC9B,WAAW;CACb;CACA,MAAM,IACJ,QACC,OAAO,cAAc,cAAe,YAA4C,KAAA;CACnF,IAAI,CAAC,GAAG,KAAK,gBAAgB,OAAO;CACpC,IAAI;EACF,MAAM,UAAU,MAAM,EAAE,IAAI,eAAe;EAC3C,IAAI,CAAC,SAAS,OAAO;EACrB,MAAM,SAAS,QAAQ,UAAU,CAAC;EAElC,IAAI,OAAO,QAAQ;EACnB,IAAI,CAAC,QAAQ,OAAO,QAAQ,uBAAuB,YACjD,OAAO,MAAM,QAAQ,mBAAmB,EAAE,YAAY,KAAA,CAAS;EAEjE,OAAO;GACL,gBAAgB,OAAO,OAAO,iBAAiB,CAAC;GAChD,8BAA8B,OAAO,OAAO,+BAA+B,CAAC;GAC5E,GAAI,OACA,EACE,aAAa;IACX,GAAI,KAAK,SAAS,EAAE,QAAQ,KAAK,OAAO,IAAI,CAAC;IAC7C,GAAI,KAAK,eAAe,EAAE,cAAc,KAAK,aAAa,IAAI,CAAC;IAC/D,GAAI,KAAK,SAAS,EAAE,QAAQ,KAAK,OAAO,IAAI,CAAC;IAC7C,GAAI,KAAK,cAAc,EAAE,aAAa,KAAK,YAAY,IAAI,CAAC;GAC9D,EACF,IACA,CAAC;GACL,WAAW;EACb;CACF,QAAQ;EACN,OAAO;CACT;AACF;;;;;;;;;;;;;;;;;;;AA0CA,IAAa,wBAAwB,gBAA+C;CAIlF,MAAM,SAHK,gBAAgB,OAAO,eAAe,cAAc,aAAa,CAAC,IAG7D,WAAW;CAC3B,MAAM,QAAyB;EAC7B,SAAS;EACT,WAAW;EACX,MAAM;EACN,WAAW;EACX,WAAW;CACb;CACA,MAAM,OAA4B;EAChC,eAAe,EAAE,GAAG,MAAM;EAC1B,iBAAiB,CAAC;CACpB;CACA,IAAI,CAAC,SAAS,OAAO,MAAM,iBAAiB,YAAY,OAAO;CAE/D,MAAM,aAAa,MAAM;CACzB,MAAM,gBAAgB,SAAyB,MAAkC;EAC/E,MAAM,MAAM,WAAW,KAAK,MAAM,IAAI;EACtC,MAAM,OAAO,OAAO,MAAM,QAAQ,CAAC;EACnC,MAAM,WAAW;EACjB,MAAM,QAAQ;EACd,MAAM,aAAa;EACnB,IAAI,MAAM,YAAY,MAAM,WAAW,MAAM,YAAY,MAAM;EAC/D,IAAI,OAAO,OAAO,IAAI,YAAY,YAAY;GAC5C,MAAM,cAAc,IAAI,QAAQ,KAAK,GAAG;GACxC,IAAI,gBAAgB;IAClB,MAAM,aAAa;IACnB,MAAM,QAAQ;IACd,MAAM,aAAa;IACnB,OAAO,YAAY;GACrB;EACF;EACA,OAAO;CACT;CACA,MAAM,eAAe;CAErB,IAAI,YAAY;CAChB,OAAO;EACL,eAAe,EAAE,GAAG,MAAM;EAC1B,iBAAiB;GACf,IAAI,CAAC,WAAW;GAEhB,IAAI,MAAM,iBAAkB,eAC1B,MAAM,eAAe;GAEvB,YAAY;EACd;CACF;AACF"}