/** * The subtractive pass — the whole Token Thrift thesis, in code. * * @module @nhtio/adk/batteries/context/thrift/subtractive_pass * * @remarks * A context window is NOT a chat history. It is just what you send for ONE dispatch. So: hold a * large WORKING set (messages, memories, retrievables, thoughts, an image, tools), then SUBTRACT it * down to the high-signal slice that fits the active model's window — focus, don't accumulate. * * This module is the model-facing plane only. Where the human-facing chat history lives (a SQLite * store, an in-memory array, anything else) is entirely the caller's concern — this battery never * touches it. Everything here is window-agnostic: the same function runs at `contextWindow: 4096` * and `contextWindow: 1_000_000` — the span is a parameter, not a rewrite. * * This is a direct extraction of the flagship reference agent's production subtractive pass * (evaluated head-to-head across five models — see the battery barrel's TSDoc for the results), * retargeted at the local structural contracts in * {@link @nhtio/adk/batteries/context/thrift/contracts} so it couples to nothing in `@nhtio/adk` * core — every value it reads is duck-typed, and its one true dependency (token estimation) is an * injected function, never a bundled tokenizer. */ import type { EstimateTokensFn, ContentLike, WorkingMessage, WorkingMemory, WorkingRetrievable, WorkingThought, WorkingTool, WorkingToolRegistry, WorkingToolCall, WorkingImage, RenderToolsFn, ShedRankFn, IsEphemeralMessageFn, IsSummaryMessageFn } from "./contracts"; export type { WorkingToolCall, WorkingImage, WorkingMessage, WorkingMemory, WorkingRetrievable, WorkingThought, WorkingTool, WorkingToolRegistry, RenderToolsFn, ShedRankFn, IsEphemeralMessageFn, IsSummaryMessageFn, EstimateTokensFn, }; /** * The token encoding this battery measures under, by default. * * @remarks * `'cl100k_base'` — a widely-available, model-agnostic tiktoken encoding — is the default because * this battery is core-agnostic: unlike the flagship reference agent it was extracted from (which * hard-codes the `gemma` encoding because its engine IS Gemma via LiteRT-LM), this battery has no * opinion on which model you run. `cl100k_base` is a reasonable, broadly-supported baseline for a * caller who hasn't thought about encodings yet; a caller running Gemma, Claude, or any other model * with a distinct tokenizer should pass their own `encoding` (and a matching {@link EstimateTokensFn}) * so the budget math agrees with what their model/battery actually counts. The encoding identifier is * an opaque string as far as this module is concerned — it is never validated or interpreted here, * only threaded through to the injected {@link EstimateTokensFn}. */ export declare const DEFAULT_ENCODING = "cl100k_base"; /** * Fallback output reserve, as a fraction of the window, used ONLY when the exact max-output budget is * unknown. * * @remarks * Prefer passing the caller's actual generation `maxTokens` as `outputReserve` — reserving a flat * fraction of the window when the model can emit at most, say, 2048 tokens throws away real input * budget (RAG chunks, history) for output that can never be produced. This fraction is the calibrated * value from the flagship reference agent (0.35 of the window), kept as the default for a caller who * genuinely doesn't know their generation cap yet. */ export declare const DEFAULT_RESERVE_FRACTION = 0.35; /** * How many of the NEWEST this-turn tool-result bodies to protect from budget-shedding as a * last-resort backstop, by default. * * @remarks * A normal read→answer turn produces 1–2 this-turn results, so `N = 3` leaves the common case * untouched; a deep read-loop (several searches/reads in one turn) that piles up more can shed its * OLDEST bodies past this cap (the model has moved past them by the time it has made this many * calls). The single newest this-turn result is always kept regardless of this cap — see * {@link subtractToFit}'s step 3b. */ export declare const DEFAULT_THIS_TURN_RESULT_KEEP = 3; /** A line in the "what got cut" trace — one bucket's before/after token weight. */ export interface BucketTrace { /** The bucket's name (e.g. `'thoughts'`, `'tools'`, `'image'`, `'toolCalls'`, `'retrievables'`, * `'memories'`, `'messages'`, `'thoughts-shed'`, `'tools-shed'`) — one entry per step of the pass. */ bucket: string; /** This bucket's measured token weight before this step ran. */ beforeTokens: number; /** This bucket's measured token weight after this step ran. */ afterTokens: number; /** How many items this bucket held before this step ran. */ beforeCount: number; /** How many items this bucket held after this step ran. */ afterCount: number; /** A short human-readable rationale for what this step did (or didn't do), for diagnostics/tracing. */ note?: string; /** Item identifiers affected by this bucket, when the working items expose ids. */ ids?: string[]; } /** The full record of one subtractive pass — before/after weights per bucket, and whether the * dispatch ultimately fits the budget. */ export interface ThriftTrace { /** The active model's context window this pass was run against. */ contextWindow: number; /** Tokens held back for the model's own output (see {@link resolveBudget}). */ reserve: number; /** The INPUT token budget — `contextWindow - reserve`. */ budget: number; /** Total measured token weight BEFORE any shedding (the "everything reasonable" starting point). */ totalBefore: number; /** Total measured token weight AFTER every shedding step ran. */ totalAfter: number; /** Whether `totalAfter` fits within `budget`. */ fits: boolean; /** The dispatch should be refused: `!fits` — even after every possible shed, the irreducible floor * (system prompt + standing instructions + newest turn + output reserve) still exceeds the window. */ refused: boolean; /** One entry per step of the pass, in the order the steps ran, for diagnostics/tracing. */ buckets: BucketTrace[]; } /** The mutable working set a dispatch starts from — the "everything reasonable" set this pass * subtracts down to what fits. Every field is a local structural type from * {@link @nhtio/adk/batteries/context/thrift/contracts} — nothing here requires a core ADK value. */ export interface WorkingSet { /** The dispatch's system prompt. Measured ctx-resolved (see {@link subtractToFit}'s `renderCtx` * option) to match what a caller's own overflow guard counts. */ systemPrompt: ContentLike | string; /** * Durable directives the caller renders into every dispatch (the caller's battery counts these * too). They are a FIXED cost like the system prompt — never shed — so the pass only ADDS them to * the running total, never trims them. Omit (or pass an empty array) when the caller uses none. A * consumer that DOES render standing instructions must pass them here or thrift would undercount * relative to the caller's own guard. */ standingInstructions?: Array; /** The conversation history this dispatch would replay — sheds stale ephemeral control messages * first (step 6), then the oldest turns (step 7); the newest turn is always kept. */ messages: WorkingMessage[]; /** Durable memories available to this dispatch — sheds lowest-`importance` first (step 5). */ memories: WorkingMemory[]; /** Retrieved (RAG) passages available to this dispatch — sheds the tail of the ranking (lowest * `score`) first (step 4), keeping the best-ranked chunks. */ retrievables: WorkingRetrievable[]; /** Model-internal guidance content (plans, per-iteration nudges, and — unless * `stripPriorTurnThoughts` is disabled — prior-turn chain-of-thought, dropped in step 1). Surviving * thoughts are sheddable oldest-first as a last resort (step 8), except any ids named in * `protectThoughtIds`. */ thoughts: WorkingThought[]; /** The tool registry this dispatch draws visible tools from — mutated via `setHidden` as tools are * shed (steps 2 and 9). */ tools: WorkingToolRegistry; /** * Prior-turn (and this-turn) tool calls whose RENDERED RESULTS the caller puts in the prompt. Omit * (or pass an empty array) when the caller doesn't measure them — the pass then treats tool-result * weight as `0`. */ toolCalls?: WorkingToolCall[]; /** Optional image (or other flat-cost media) attachment. The single biggest token hog. */ image?: WorkingImage; } /** * Options shared by every entry point in this module that needs to measure a value's token cost — * the injected estimator plus the encoding it measures under. */ export interface EstimatorOptions { /** * REQUIRED. Measures a rendered string's token cost under `encoding`, optionally resolved against * a live dispatch context. There is no default — this battery ships with no bundled tokenizer, so * a caller who omits this gets a clear thrown error naming the option, not a silent guess. */ estimateTokens: EstimateTokensFn; /** The encoding to measure under. Default: {@link DEFAULT_ENCODING}. */ encoding?: string; } /** * Options accepted by {@link subtractToFit}. `estimateTokens` (via {@link EstimatorOptions}) is the * only option with no default; every other field is a calibrated default, documented on its own * declaration below, that a caller can override. * * @remarks * Earlier positional-argument forms of this function (as it existed in the flagship reference agent * this battery was extracted from) took `outputReserve`, `keepThoughtIds`, `renderTools`, * `protectThoughtIds`, `renderCtx`, and `protectedToolNames` as seven trailing positional parameters. * They are collected here into one options object to keep the call site legible and to give each a * documented default — the mapping from the old positional order to these keys is: position 4 → * `outputReserve`, 5 → `keepThoughtIds`, 6 → `renderTools`, 7 → `protectThoughtIds`, 8 → `renderCtx`, * 9 → `protectedToolNames`. */ export interface SubtractToFitOptions extends EstimatorOptions { /** * The EXACT number of tokens to hold back for the model's own output — pass the generation * `maxTokens` the model is configured with. When omitted, falls back to * {@link DEFAULT_RESERVE_FRACTION} of the window (a guess, for callers that don't know the cap). * Clamped so the reserve never exceeds the window. Forwarded to {@link resolveBudget}. */ outputReserve?: number; /** * The fallback reserve fraction used when `outputReserve` is omitted. Default: * {@link DEFAULT_RESERVE_FRACTION}. Forwarded to {@link resolveBudget}. */ reserveFraction?: number; /** * Whether to apply the prior-turn thought strip (step 1) at all. Default `true`. * * @remarks * The default is driven by Gemma's model card §3, "No Thinking Content in History": thoughts from * previous model turns must not be re-added before the next user turn. This is also pure thrift — * prior-turn reasoning is the highest-volume, lowest-reuse content there is, Gemma or not — so the * default stays `true` even for callers on a different model family; a caller whose model * genuinely benefits from replaying its own prior chain-of-thought (uncommon) can set this `false` * to skip the strip entirely and let every thought flow into the later per-thought budget shed * (step 8) instead. */ stripPriorTurnThoughts?: boolean; /** * Thought ids to PRESERVE through the prior-turn strip (step 1) — e.g. a planner's synthetic * THIS-TURN plan thought, which is fresh guidance generated for the current request (not prior-turn * chain-of-thought, and not subject to the §3 policy above). Everything else is dropped when * stripping is enabled. Omit to drop every thought when stripping is enabled. */ keepThoughtIds?: ReadonlySet; /** * The caller's tool-declaration renderer. When supplied, the tools bucket is measured against the * REAL rendered tool-definitions block (e.g. full JSON-Schema per tool) instead of a * `name: description` proxy — the proxy can undercount a schema-heavy tool block by an order of * magnitude relative to what actually gets dispatched. Omit to fall back to the proxy (keeps this * battery usable standalone, without a real tool-rendering pipeline, e.g. in tests). */ renderTools?: RenderToolsFn; /** * Thought ids that must NEVER be shed for budget (step 8) even when the dispatch is over — the * this-turn scaffolding the model needs to answer at all (e.g. a plan thought + a citation * reinforcement thought). Everything else in the surviving keep-set (per-iteration nudge thoughts, * older synthetic guidance) is sheddable oldest-first when the dispatch still doesn't fit. This is * a SUBSET of `keepThoughtIds`: `keepThoughtIds` decides what survives the prior-turn strip (step * 1); `protectThoughtIds` decides what additionally survives the budget shed (step 8). Omit to make * every surviving thought sheddable. */ protectThoughtIds?: ReadonlySet; /** * The live dispatch context, so a DYNAMIC (evaluatable) value's token count reflects the string it * will resolve to for THIS dispatch — forwarded as the `ctx` argument to `estimateTokens`. Without * it, a dynamic value is measured at its no-`ctx` fallback size, and the budget can disagree with * what the caller's own battery ships (an under-count, since evaluated content is typically LARGER * than its static form — e.g. an interpolated system prompt). Optional; static content measures * identically with or without it. */ renderCtx?: unknown; /** * Tool names the caller's CURRENT PLAN has committed to that have NOT yet been called this turn. * These are UNSHEDDABLE by the last-resort tool shed (step 9) until every other tool has already * shed — removing a plan-committed tool from the visible set leaves the model instructed to call a * tool it can no longer see. The protection is bounded: once the tool HAS been called (its result * is already in context) it should be dropped from this set by the caller, so it is never a * permanent floor. Omit to make every visible tool sheddable on equal footing. */ protectedToolNames?: ReadonlySet; /** * Ranks a tool by last-resort shed priority for step 9 (lower sheds first). Default: a single * generic tier — every tool ranks equally, so the shed proceeds in the order `relevantToolNames` * listed them (a stable sort preserves input order when every rank ties). This battery has no * domain knowledge of which of a caller's tools are cheap-to-lose "gather" tools versus * load-bearing "delivery" tools, so it does not guess a tiering. A caller WHO DOES have that * knowledge (as the flagship reference agent does — it ranks ~90 known tool names into seven * tiers, sheds its `provide_answer` tool before its core artifact readers, etc.) should inject a * `ShedRankFn` that encodes it; see {@link ShedRankFn} for the contract. */ shedRank?: ShedRankFn; /** * Decides whether a message is an ephemeral control message (step 6) — re-derived fresh every * dispatch iteration, never persisted, so only the LATEST surviving copy carries live information. * Default: `(m) => m.id.startsWith('__eph-')`, the flagship reference agent's own convention. A * caller with a different (or no) ephemeral-message convention should override this; the default * simply never matches when a caller's ids don't use that prefix, degrading step 6 to a no-op. */ isEphemeralMessage?: IsEphemeralMessageFn; /** * Decides whether a message is a summarizing strategy's load-bearing running summary (step 7) — * content that stands in for every older turn that strategy folded away, and so must never be shed * like an ordinary old turn even though it renders as the chronologically oldest message. Default: * `(m) => m.id === '__compact-summary'`, the flagship reference agent's own convention for its * paired summarizing ("compact") strategy. Callers who never run a summarizing strategy alongside * this battery can ignore this option entirely — the default predicate simply never matches. */ isSummaryMessage?: IsSummaryMessageFn; /** * How many of the NEWEST this-turn tool-result bodies the newest-N backstop (step 3b) protects * from shedding once ordinary prior-turn shedding is exhausted and the dispatch still doesn't fit. * Default: {@link DEFAULT_THIS_TURN_RESULT_KEEP}. */ thisTurnResultKeep?: number; } /** * Resolve the per-dispatch INPUT token budget from the ACTIVE model's window: the window minus the * room held back for the model's own output. The same call works for a 4K model and a 1M model. * * @param contextWindow - The active model's context window (input + output share it). * @param outputReserve - The EXACT number of tokens to hold back for output — pass the generation * `maxTokens` the model is configured with. When omitted, falls back to `reserveFraction` of the * window (a guess, for callers that don't know the cap). Clamped so the reserve never exceeds the * window. * @param reserveFraction - The fallback fraction used when `outputReserve` is omitted. Default: * {@link DEFAULT_RESERVE_FRACTION}. */ export declare const resolveBudget: (contextWindow: number, outputReserve?: number, reserveFraction?: number) => number; /** * Gemma model card §3 — "No Thinking Content in History": thoughts from previous model turns MUST * NOT be re-added before the next user turn. Enforced here as a standalone, callable step (not a * silent battery-wide flag), because it is also pure thrift: prior-turn reasoning is the * highest-volume, lowest-reuse content a working set carries. * * @remarks * `keepIds` is an allow-list of thought ids to PRESERVE — used for e.g. a planner's synthetic * THIS-TURN plan thought, which is fresh guidance generated for the current request (NOT prior-turn * chain-of-thought, and NOT subject to the §3 policy): it must survive into the next dispatch's * prompt so the model follows the plan. Everything else is dropped. * * `subtractToFit` calls this internally as step 1 (gated by its `stripPriorTurnThoughts` option, * default `true`); it is also exported standalone for a caller who wants the strip without running * the rest of the pass. * * @param ws - A working set exposing (at least) a mutable `thoughts` array; mutated in place. * @param options - The estimator used to measure the tokens reclaimed by dropped thoughts. * @param keepIds - Thought ids to preserve. * @returns How many thoughts were dropped, and how many tokens that reclaimed. */ export declare const stripPriorTurnThoughts: (ws: Pick, options: EstimatorOptions, keepIds?: ReadonlySet) => { dropped: number; tokens: number; }; /** * The subtractive pass. Start wide; measure every bucket; then shed lowest-signal first until the * dispatch fits the active window's budget — the image (the single biggest hog) goes first when it * doesn't fit, then tools the turn doesn't need, the tail of the retrieval ranking, low-value * memories, the oldest conversation turns. Each cut is by EVIDENCE (a measured bucket), never a * guess. If even the floor (system prompt + newest turn) won't fit, the dispatch REFUSES (`fits: * false`) — a bounded refusal beats a truncated, incoherent dispatch. * * @remarks * `options.estimateTokens` is REQUIRED — this battery ships with no bundled tokenizer, so a missing * estimator throws {@link @nhtio/adk/batteries/context/exceptions!E_CONTEXT_RESOLVER_MISSING} * immediately (naming the option) rather than silently guessing at token counts. * * @param ws - The working set to subtract in place. Mutated: `thoughts`, `retrievables`, * `memories`, `messages`, `toolCalls`, `image.kept`, and the visible/hidden state of `tools` may * all change. * @param contextWindow - The active model's context window. * @param relevantToolNames - The names (from `ws.tools.all()`) that should start VISIBLE for this * turn — every other registered tool starts hidden (0 schema tokens, still callable via a * catalog). The last-resort shed (step 9) may hide some of these too. * @param options - See {@link SubtractToFitOptions}. * @throws {@link @nhtio/adk/batteries/context/exceptions!E_CONTEXT_RESOLVER_MISSING} When * `options.estimateTokens` is not a function. */ export declare const subtractToFit: (ws: WorkingSet, contextWindow: number, relevantToolNames: string[], options: SubtractToFitOptions) => ThriftTrace;