import { byteLength, DEFAULT_IMAGE_TOKENS, messagesBytes, toolCallBytes, BYTES_PER_TOKEN } from "./estimate.js"; import { DEFAULT_IMAGE_BYTES_CEILING, DEFAULT_IMAGE_BYTES_LOW_WATER, OVERFLOW_MARGIN } from "./limits.js"; export type AgentMessage = Record; /** How much of one tool call survives into the prompt. */ export enum Tier { /** The call and its result as they were. */ Verbatim = 0, /** * The call exactly as it was made, with its result replaced by a notice. * * Only the result folds. Arguments are what the model wrote, and a folded * one is a string that still reads as content: it gets retyped — a severed * heredoc, a path ending mid-segment, a notice written into a source file by * the write that copied it. Keeping them whole also costs little, because a * result outweighs the call that asked for it by orders of magnitude. */ Folded = 1, /** * Gone: the call and the result answering it are both removed, and a run of * them leaves one line naming what went. * * This is the only tier that reclaims what a folded call still costs. It * keeps its arguments and its id, and an id is carried twice over a call's * life — once on the call, once on the result that answers it. In a session * of thousands of calls that residue is most of the floor, and because the * ceiling is set above the floor, a floor that cannot fall is a prompt that * cannot stop growing. * * Removing one half alone is what the provider rejects, so both go together; * a whole turn leaves at once and the alternation the provider expects is * preserved. */ Drop = 2, } // How many of the newest calls are never dropped, however tight the budget. // Recent tool traffic is what the model is still working from, so it is trimmed // as before and left in place; only what it has finished with leaves entirely. const DROP_KEEP_RECENT = 150; // At most this many tool names are listed in the line standing for a dropped // run, so one long run cannot cost more than the calls it replaced. const DROP_NAMES_LISTED = 6; // Held back from each dropped call against the line that will stand for it. // The line is written after the budget has been met, so without a reserve a // fold could land on target and then overshoot it by what the lines cost. // // One line covers a whole run of adjacent turns, and only a turn carrying // nothing but calls is droppable, so prose cannot fragment a run into a line // per call: the total is bounded by how often prose interrupts the tool // traffic, not by how many calls were made. What is reserved here is therefore // a per-run cost charged once, where a run is what survives between two pieces // of prose. const DROP_LINE_RESERVE_BYTES = 64; // The share of the prompt a fold has to free to be worth the cache miss it // costs. See worthFolding. const MIN_FOLD_FRACTION = 0.2; // Bound an error kept in a folded call. Both ends are kept because a long // failure states its conclusion in either place: a rejected write leads with // the reason, a test or lint run ends with it. const ERROR_HEAD_BYTES = 1000; const ERROR_TAIL_BYTES = 500; export interface Limits { /** The prompt size that triggers folding. */ ceiling: number; /** * Folded down to once the ceiling is crossed, so the next call has room to * grow into rather than crossing the ceiling again immediately. Overshooting * is what makes cache invalidation rare. */ lowWater: number; /** * The size past which any saving is worth its cache miss, because the turn * itself is at risk. Absent means a margin above the ceiling. */ urgent?: number; /** * The image payload a prompt may carry, in the bytes that go on the wire. * Absent means {@link DEFAULT_IMAGE_BYTES_CEILING}. * * A second budget is needed because the first one cannot see this: an image * is charged by its pixels, so the estimate counts a screenshot at what the * provider bills for it — around a hundredth of the base64 that carries it. * A session that navigates by screenshot therefore accumulates megabytes the * token ceiling reads as kilobytes, re-uploads all of them every turn, and * fails on the request body limit long before anything folds. */ imageCeiling?: number; /** * The image payload a pass over {@link imageCeiling} folds back down to. * Absent means {@link DEFAULT_IMAGE_BYTES_LOW_WATER}. */ imageLowWater?: number; } /** * Tools whose newest result is never folded. * * A skill document is operating guidance the agent believes is in effect, not * a lookup it can repeat: folding it away silently changes how the agent works * with nothing to show for it. An older load of the same skill still folds — * only the newest of each is pinned. */ export const PINNED_TOOLS = new Set(["load_skill"]); /** * Tools no call of which is ever folded, however old. * * What the user decided is not a lookup that can be repeated: the session * cannot ask again, and an agent that has lost the answer proceeds on its own * guess instead. The question folds with it — an answer naming an option means * nothing without the options — so the pair is held whole. It is affordable * because it is rare and small: a long session's asks weigh a few kilobytes * between them, against the megabytes of tool output around them. */ export const NEVER_FOLDED_TOOLS = new Set(["ask_user"]); export interface FoldResult { /** Estimated prompt size after folding, in tokens. */ tokens: number; /** Estimated prompt size before folding, in tokens. */ tokensBefore: number; /** Calls standing below verbatim after this pass. */ folded: number; /** Calls this pass moved a tier, which is what a rewrite is made of. */ promoted: number; /** * The oldest message this pass rewrote, or -1 when it rewrote none. * * Everything from here on is a cache miss, so this is what an unexplained * re-bill is traced with: a pass that moved message 300 of 840 re-bought two * thirds of the prompt, whatever it freed. */ rewroteFrom: number; /** Whether the prompt was over the ceiling and folding was held back anyway. */ held: boolean; /** Calls this pass took the images off, leaving the rest of their result. */ imagesFolded: number; /** Image payload still in the prompt afterwards, in wire bytes. */ imageBytes: number; } interface Call { id: string; /** Index into messages of the assistant message carrying the tool call. */ callMessage: number; /** Index into that message's content array. */ callPart: number; /** Index into messages of the paired tool result. */ resultMessage: number; /** * The result's content as it arrived. Promoting a call twice in one pass * would otherwise measure and clamp the notice left by the first promotion * instead of the output it stood for. */ resultContent: any[]; tier: Tier; } /** * Remembers how far each call has been folded, so a call that has been folded * stays folded even when the budget briefly widens. Monotonicity is what keeps * the prompt's prefix stable between calls, and a stable prefix is what the * provider's cache is keyed on; re-expanding a call would invalidate everything * after it. */ export class FoldState { private tiers = new Map(); /** * How far orphan reasoning has been blanked, in message indices. * * Held here rather than recomputed because the range it belongs to is not a * function of the conversation alone: it ends at the last user message, and * that end moves forward with every user turn. Without the watermark a pass * that promoted nothing would still strip whatever the newest user turn had * left unprotected, and rewrite the prompt for it. */ private reasoningThrough = 0; /** * Calls whose result has had its images taken off while the rest of it * stayed. Remembered for the same reason a tier is: the host hands over a * fresh copy of the conversation every request, so an image not stripped * again would come back and move the prefix it sits in. */ private imagesStripped = new Set(); tierOf(callId: string): Tier { return this.tiers.get(callId) ?? Tier.Verbatim; } strippedImages(callId: string): void { this.imagesStripped.add(callId); } hasStrippedImages(callId: string): boolean { return this.imagesStripped.has(callId); } promote(callId: string, to: Tier): void { if (to > this.tierOf(callId)) this.tiers.set(callId, to); } get reasoningStrippedThrough(): number { return this.reasoningThrough; } reachedReasoning(through: number): void { if (through > this.reasoningThrough) this.reasoningThrough = through; } clear(): void { this.tiers.clear(); this.imagesStripped.clear(); this.reasoningThrough = 0; } get size(): number { return this.tiers.size; } } /** * Folds old tool traffic in `messages` until the prompt fits, mutating in * place, and reports the estimate it settled on. * * The host hands each `context` handler a deep clone of the conversation, so * editing it changes what the model is sent and nothing else — the session * store keeps every byte, which is what lets recall hand a dropped result back. */ export function fold( messages: AgentMessage[], fixedBytes: number, limits: Limits, state: FoldState, ratio = 1, imageTokens = DEFAULT_IMAGE_TOKENS, ): FoldResult { const toTokens = (bytes: number) => tokensOf(bytes, ratio); let bytes = fixedBytes + messagesBytes(messages, imageTokens); const tokensBefore = toTokens(bytes); const calls = indexCalls(messages); const protectedFrom = lastUserMessageIndex(messages); // The oldest message this pass rewrites. Re-applying a remembered tier does // not count: it reproduces what the last request already sent, so the prefix // is where it was. Anything else that moves a byte does. let rewroteFrom = -1; let promoted = 0; const moved = (index: number): void => { if (rewroteFrom < 0 || index < rewroteFrom) rewroteFrom = index; }; // How far the orphan reasoning below has been blanked. It follows the calls: // a turn that made none is folded when the tool traffic around it is. // // The replay is clamped to how far an earlier pass reached, because the // boundary it stops at is the last user message and that boundary moves: a // new user turn leaves the previous turn's reasoning strippable, and // stripping it would rewrite the prompt's tail for a handful of bytes // nobody asked to spend a cache miss on. let reasoningFrom = 0; const followReasoning = (through: number, replaying: boolean): void => { const limit = replaying ? Math.min(through, state.reasoningStrippedThrough) : through; bytes -= stripOrphanReasoning(messages, reasoningFrom, Math.min(limit, protectedFrom), replaying ? undefined : moved); reasoningFrom = Math.max(reasoningFrom, Math.min(limit, protectedFrom)); if (!replaying) state.reachedReasoning(reasoningFrom); }; // Re-apply what earlier requests already decided before measuring against the // budget. The host hands over a fresh copy of the stored conversation every // time, so without this a call folded ten requests ago would come back whole // and move the prompt's prefix. for (const call of calls) { const remembered = state.tierOf(call.id); if (remembered === Tier.Verbatim) { if (state.hasStrippedImages(call.id)) bytes -= stripResultImages(messages, call, imageTokens); continue; } // A dropped call is removed structurally in one pass once every tier is // settled, so here it only has to stop counting towards the prompt. bytes -= remembered >= Tier.Drop ? dropSavings(messages, call, imageTokens) + stripReasoning(messages, call, protectedFrom) : applyTier(messages, call, protectedFrom, imageTokens); call.tier = remembered; followReasoning(call.callMessage, true); } // Remembered drops are charged their lines here; newly promoted ones add // theirs as they are chosen. bytes += dropLineReserve(messages, calls.filter((call) => call.tier >= Tier.Drop)); // The image budget is settled before the token one, and outside the question // of whether folding pays for itself. That question is asked in tokens, and // in tokens an image is worth almost nothing; what is at stake here is the // size of the request body, which a provider rejects outright rather than // bills for. let imagesFolded = 0; let imageBytes = imagePayloadBytes(messages); // A call already dropped still stands in `messages`: it is removed in one // pass at the end, once every tier is settled. Its images are therefore // counted by the measure above and unreachable by the loop below, which // together would strip surviving captures to make room for bytes that were // never going to be sent. for (const call of calls) { if (call.tier >= Tier.Drop) imageBytes -= resultImageBytes(messages, call); } const imageLowWater = limits.imageLowWater ?? DEFAULT_IMAGE_BYTES_LOW_WATER; if (imageBytes > (limits.imageCeiling ?? DEFAULT_IMAGE_BYTES_CEILING)) { for (const call of calls) { if (imageBytes <= imageLowWater) break; if (call.tier !== Tier.Verbatim) continue; const carried = resultImageBytes(messages, call); if (carried === 0) continue; bytes -= stripResultImages(messages, call, imageTokens); state.strippedImages(call.id); imageBytes -= carried; imagesFolded++; // The call itself keeps every byte it had: only its result moved, and // naming the call message instead would overstate how much of the prompt // this re-bills. moved(call.resultMessage); } } const over = toTokens(bytes) > limits.ceiling; const worth = over && worthFolding(messages, calls, bytes, limits, protectedFrom, ratio, imageTokens); const rewrote = (call: Call): void => { promoted++; moved(call.callMessage); }; if (worth) { // The low-water mark already clears the floor by construction — it is a // share of the span between floor and ceiling — so an unreachable target // cannot be chased here. const target = limits.lowWater; for (const call of calls) { if (toTokens(bytes) <= target) break; if (call.tier >= Tier.Folded) continue; bytes -= applyTier(messages, call, protectedFrom, imageTokens); call.tier = Tier.Folded; state.promote(call.id, Tier.Folded); rewrote(call); followReasoning(call.callMessage, false); } const droppable = droppableCalls(messages, calls, protectedFrom); const newlyDropped: Call[] = []; for (const call of droppable) { if (toTokens(bytes - dropLineReserve(messages, newlyDropped)) <= target) break; if (call.tier >= Tier.Drop) continue; bytes -= dropSavings(messages, call, imageTokens) + stripReasoning(messages, call, protectedFrom); call.tier = Tier.Drop; state.promote(call.id, Tier.Drop); rewrote(call); newlyDropped.push(call); } bytes += dropLineReserve(messages, newlyDropped); } const dropped = calls.filter((call) => call.tier >= Tier.Drop); if (dropped.length > 0) { // The reserve stood in for these lines while the budget was being met; what // they actually cost replaces it now they exist. bytes -= dropLineReserve(messages, dropped); bytes += applyDrops(messages, dropped); } return { tokens: toTokens(bytes), tokensBefore, folded: calls.filter((call) => call.tier !== Tier.Verbatim).length, promoted, rewroteFrom, held: over && !worth, imagesFolded, imageBytes: imagePayloadBytes(messages), }; } /** * The image payload of a whole conversation, in the base64 bytes that travel. * * Everything is counted, including images a user attached and those of the * newest results, because what this is measured against is the size of the * request — not the size of the part folding is allowed to touch. */ function imagePayloadBytes(messages: AgentMessage[]): number { let bytes = 0; for (const message of messages) { if (!Array.isArray(message?.content)) continue; for (const part of message.content) { if (part?.type === "image") bytes += (part.data ?? "").length; } } return bytes; } function resultImageBytes(messages: AgentMessage[], call: Call): number { const content = messages[call.resultMessage]?.content; if (!Array.isArray(content)) return 0; let bytes = 0; for (const part of content) { if (part?.type === "image") bytes += (part.data ?? "").length; } return bytes; } /** * Replaces the images of one result with the notice that stands for any folded * output, keeping whatever text came with them, and reports the bytes it saved. * * The text stays because it is not what costs anything: a tool that returns a * screenshot alongside a description loses the expensive half and keeps the * half the model can still read. The image itself stays in the session store, * so recall hands it back by the id the result block carries. */ function stripResultImages(messages: AgentMessage[], call: Call, imageTokens: number): number { const result = messages[call.resultMessage]; if (!Array.isArray(result?.content)) return 0; const before = messagesBytes([result], imageTokens); result.content = result.content.map((part: any) => part?.type === "image" ? { type: "text", text: omission((part.data ?? "").length) } : part, ); return before - messagesBytes([result], imageTokens); } /** * Whether a fold would buy more than the cache miss it costs. * * Folding rewrites the prompt where the oldest unfolded call sits, and a * provider re-bills every token after the first one that moved — so a pass is * charged the whole conversation whatever it frees. Sessions on disk show what * that means in practice: passes that re-bought three hundred thousand tokens * to free a few hundred, because the budget was a little over and a little was * all that was left to give. * * So a pass has to free a real share of the prompt, measured against what is * still foldable rather than against the target, which no amount of folding * might reach. The exception is a prompt already past the point where the turn * is in danger: there a small saving still beats not being sent. */ function worthFolding( messages: AgentMessage[], calls: Call[], bytes: number, limits: Limits, protectedFrom: number, ratio: number, imageTokens: number, ): boolean { const tokens = tokensOf(bytes, ratio); if (tokens > (limits.urgent ?? limits.ceiling * OVERFLOW_MARGIN)) return true; const reachable = tokensOf(bytes - foldableBytes(messages, calls, protectedFrom, imageTokens), ratio); return tokens - Math.max(limits.lowWater, reachable) >= tokens * MIN_FOLD_FRACTION; } /** * Blanks the reasoning that produced a call, reporting the bytes it freed. * * The reasoning that produced a call is what the model needed to make it, not * something it re-reads twenty calls later — so it goes at the same moment, * except in the tail the provider requires intact. */ function stripReasoning(messages: AgentMessage[], call: Call, protectedFrom: number): number { if (call.callMessage >= protectedFrom) return 0; return blankReasoning(messages[call.callMessage]); } /** * Blanks the reasoning of the assistant turns in `[from, through)` that made no * tool call, reporting the bytes it freed. * * Such a turn is unreachable through any call — a reply to the user, or a turn * the provider cut off mid-thought — so without this its reasoning is the one * thing in the prompt that nothing can ever remove. It weighs more than it * looks: the signature is the whole of the model's thinking in encrypted form, * and a gateway that replays it charges for every token of it, so one turn * truncated at its output limit can cost tens of thousands of tokens on every * request for the rest of the session. * * The range follows the calls rather than the clock, so a turn's reasoning goes * when the tool traffic it sits among goes and not before: a conversation small * enough never to fold keeps all of it. */ function stripOrphanReasoning(messages: AgentMessage[], from: number, through: number, onBlank?: (index: number) => void): number { let saved = 0; for (let m = from; m < through; m++) { const message = messages[m]; if (message?.role !== "assistant" || !Array.isArray(message.content)) continue; if (message.content.some((part: any) => part?.type === "toolCall")) continue; const blanked = blankReasoning(message); if (blanked > 0) onBlank?.(m); saved += blanked; } return saved; } /** * Empties every reasoning block of one message, reporting the bytes it freed. * * Emptying the text rather than removing the block keeps every later index * valid; the provider adapter drops a thinking block whose text is blank, and * an assistant turn left with nothing else is dropped whole. */ function blankReasoning(message: AgentMessage | undefined): number { const content = message?.content; if (!Array.isArray(content)) return 0; let saved = 0; for (const block of content) { if (block?.type !== "thinking" || block.redacted) continue; if (!block.thinking && !block.thinkingSignature) continue; saved += byteLength(block.thinking ?? "") + byteLength(block.thinkingSignature ?? ""); block.thinking = ""; // The signature is dropped with the text it certifies. The adapter already // discards a block whose text is empty, so this changes no request; it is // what keeps the estimate saying the same thing the wire does, and the // signature is the larger half of what goes. block.thinkingSignature = ""; } return saved; } /** * The oldest calls that can leave without stranding the turn they belong to. * * A turn goes whole or not at all. Dropping only some of an assistant message's * calls leaves the message standing while the results answering them go, and a * turn left standing with no results beside it ends up adjacent to another * assistant turn — the shape the provider rejects. So a call is droppable only * when every call in its message is droppable too and the message carries * nothing else: text the model would lose, or reasoning in the tail that has to * be replayed intact and so cannot be blanked. */ function droppableCalls(messages: AgentMessage[], calls: Call[], protectedFrom: number): Call[] { const candidates = calls.slice(0, Math.max(0, calls.length - DROP_KEEP_RECENT)); const byMessage = new Map(); for (const call of candidates) { const group = byMessage.get(call.callMessage); if (group) group.push(call); else byMessage.set(call.callMessage, [call]); } const droppable: Call[] = []; for (const [index, group] of byMessage) { const content = messages[index]?.content; if (!Array.isArray(content)) continue; let callParts = 0; let carriesMore = false; for (const part of content) { if (part?.type === "toolCall") { callParts++; } else if (part?.type === "thinking") { if (index >= protectedFrom && (part.thinking || part.thinkingSignature)) carriesMore = true; } else { carriesMore = true; } } // A pinned call is kept out of `calls` entirely, so a message holding one // fails this count and stays — which is what pinning is for. if (carriesMore || callParts !== group.length) continue; droppable.push(...group); } return droppable; } /** What removing a call and the result answering it would take off the prompt. */ function dropSavings(messages: AgentMessage[], call: Call, imageTokens: number): number { let saved = 0; const part = callPart(messages, call); if (part) saved += toolCallBytes(part); const result = messages[call.resultMessage]; if (result) saved += messagesBytes([result], imageTokens); return saved; } /** * What the lines standing for `dropped` will cost, charged once per run. * * Two dropped turns belong to the same run when nothing but the results they * are losing lies between them, so a prefix of pure tool traffic collapses to a * single line however many calls it held. */ function dropLineReserve(messages: AgentMessage[], dropped: Call[]): number { const turns = [...new Set(dropped.map((call) => call.callMessage))].sort((a, b) => a - b); let runs = 0; let previous = -1; for (const turn of turns) { if (previous < 0 || !onlyResultsBetween(messages, previous, turn)) runs++; previous = turn; } return runs * DROP_LINE_RESERVE_BYTES; } function onlyResultsBetween(messages: AgentMessage[], from: number, to: number): boolean { for (let m = from + 1; m < to; m++) { if (messages[m]?.role !== "toolResult") return false; } return true; } /** * Removes dropped calls and their results, leaving one line per run naming what * went, and reports what those lines cost. * * The line lands on the next surviving assistant message rather than in one of * its own: an assistant turn and the results answering it leave together, so * deleting them keeps the alternation the provider expects, while inserting a * message between two assistant turns would break it. */ function applyDrops(messages: AgentMessage[], dropped: Call[]): number { const ids = new Set(dropped.map((call) => call.id)); const kept: AgentMessage[] = []; let tally = new Map(); let cost = 0; const flushInto = (message: AgentMessage): void => { if (tally.size === 0) return; const text = dropLine(tally); (message.content as any[]).unshift({ type: "text", text }); cost += byteLength(text); tally = new Map(); }; for (const message of messages) { if (message?.role === "toolResult") { if (!ids.has(message.toolCallId)) kept.push(message); continue; } const content = message?.content; if (!Array.isArray(content)) { kept.push(message); continue; } const survivors = content.filter((part: any) => { if (part?.type !== "toolCall" || !ids.has(part.id)) return true; tally.set(part.name ?? "", (tally.get(part.name ?? "") ?? 0) + 1); return false; }); // A turn whose calls have all gone is left holding nothing the model can // read: its reasoning was blanked when the calls were folded, and a block // with neither text nor signature is already discarded by the adapter. // Keeping it would leave two assistant turns adjacent with no result // between them, which is the shape the provider rejects. const speaks = survivors.some((part: any) => part?.type !== "thinking" || part.thinking || part.thinkingSignature); if (!speaks) continue; message.content = survivors; if (message.role === "assistant") flushInto(message); kept.push(message); } // A run reaching the end of the conversation has no later assistant turn to // land on, so it goes on the last one there is. if (tally.size > 0) { for (let m = kept.length - 1; m >= 0; m--) { if (kept[m]?.role !== "assistant" || !Array.isArray(kept[m].content)) continue; flushInto(kept[m]); break; } } messages.length = 0; messages.push(...kept); return cost; } /** Names what a run of dropped calls was, commonest tool first. */ function dropLine(tally: Map): string { const ranked = [...tally.entries()].sort((a, b) => b[1] - a[1]); const total = ranked.reduce((sum, [, count]) => sum + count, 0); const listed = ranked.slice(0, DROP_NAMES_LISTED).map(([name, count]) => `${name} \u00d7${count}`); const rest = ranked.length - listed.length; if (rest > 0) listed.push(`+${rest} more`); return `[dropped ${total} earlier calls: ${listed.join(", ")}]`; } /** * What the prompt would still cost with every tool call folded as far as it * goes: the prose, the fixed cost, and what the calls too recent to drop are * asking for. Nothing can bring a prompt below this, so it is what the budget * has to be set around. */ export function incompressibleTokens(messages: AgentMessage[], fixedBytes: number, ratio = 1, imageTokens = DEFAULT_IMAGE_TOKENS): number { const calls = indexCalls(messages); const total = fixedBytes + messagesBytes(messages, imageTokens); return tokensOf(total - foldableBytes(messages, calls, lastUserMessageIndex(messages), imageTokens), ratio); } export function tokensOf(bytes: number, ratio = 1): number { return Math.floor(Math.floor(Math.max(0, bytes) / BYTES_PER_TOKEN) * ratio); } /** * Pairs each tool call with its result, oldest first. A call still awaiting its * result belongs to the turn in flight — the one whose tool is running, or * whose confirmation the user is looking at — and is left out rather than * folded: the model is about to act on it. */ function indexCalls(messages: AgentMessage[]): Call[] { const pending = new Map(); const calls: Call[] = []; for (let m = 0; m < messages.length; m++) { const message = messages[m]; if (message?.role === "assistant") { const content = Array.isArray(message.content) ? message.content : []; for (let p = 0; p < content.length; p++) { const part = content[p]; if (part?.type !== "toolCall" || typeof part.id !== "string" || !part.id) continue; pending.set(part.id, { id: part.id, callMessage: m, callPart: p, resultMessage: -1, resultContent: [], tier: Tier.Verbatim }); } continue; } if (message?.role !== "toolResult") continue; const call = pending.get(message.toolCallId); if (!call) continue; pending.delete(message.toolCallId); call.resultMessage = m; call.resultContent = Array.isArray(message.content) ? message.content : []; calls.push(call); } // Ordered by where the call was made, not by when its result arrived: calls // issued together are answered in whatever order they finish, and folding // oldest-first has to mean oldest as the conversation reads. calls.sort((a, b) => a.callMessage - b.callMessage || a.callPart - b.callPart); return dropPinned(messages, calls); } /** Drops the calls folding may not touch: every call of a never-folded tool, * and the newest call of each pinned tool by the object it addressed. */ function dropPinned(messages: AgentMessage[], calls: Call[]): Call[] { const pinned = new Set(); const newest = new Map(); for (const call of calls) { const part = callPart(messages, call); if (!part) continue; if (NEVER_FOLDED_TOOLS.has(part.name)) pinned.add(call); else if (PINNED_TOOLS.has(part.name)) newest.set(`${part.name}:${jsonText(part.arguments)}`, call); } for (const call of newest.values()) pinned.add(call); if (pinned.size === 0) return calls; return calls.filter((call) => !pinned.has(call)); } /** * The index of the newest user message. Reasoning after it is left alone: * Anthropic requires the thinking blocks of an assistant turn in the last * position to be replayed complete when that turn used tools, and OpenAI asks * for every reasoning item since the last user message. */ function lastUserMessageIndex(messages: AgentMessage[]): number { for (let m = messages.length - 1; m >= 0; m--) { if (messages[m]?.role === "user") return m; } return messages.length; } /** * What full folding would still be able to remove from here. * * Drop is part of "full", so a call old enough to leave counts for everything it * weighs rather than for the call and arguments it would otherwise be stuck at. * This is what keeps the floor — and with it the ceiling standing above the * floor — from rising with every call a long session makes. */ function foldableBytes(messages: AgentMessage[], calls: Call[], protectedFrom: number, imageTokens: number): number { let savings = 0; // Parallel calls share one assistant message, and its reasoning goes with // whichever of them is promoted first, so counting it once per call would // overstate what is left to remove and put the floor below where folding can // actually land. const countedThinking = new Set(); const droppable = new Set(droppableCalls(messages, calls, protectedFrom).map((call) => call.id)); const droppedHere: Call[] = []; for (const call of calls) { if (call.tier >= Tier.Drop) continue; if (call.callMessage < protectedFrom && !countedThinking.has(call.callMessage)) { countedThinking.add(call.callMessage); savings += thinkingBytes(messages[call.callMessage]); } if (droppable.has(call.id)) { savings += dropSavings(messages, call, imageTokens); droppedHere.push(call); continue; } if (call.tier >= Tier.Folded) continue; const result = messages[call.resultMessage]; if (!result) continue; // A failed call keeps both ends of its text once folded, and no tier below // this one takes them away, so estimating it as a bare notice would promise // room that folding cannot deliver. const folded = byteLength(result.toolName ?? "") + byteLength(result.toolCallId ?? "") + byteLength(collapseResult(call, result.isError === true)[0].text); savings += Math.max(0, messagesBytes([result], imageTokens) - folded); } // What full folding leaves behind includes the lines standing for what left. // A turn that made no call is folded with the traffic around it, so the // reasoning full folding reaches is what lies below the newest call. savings += orphanReasoningBytes(messages, reasoningThrough(calls, protectedFrom)); return Math.max(0, savings - dropLineReserve(messages, droppedHere)); } /** How far into the conversation a full fold blanks reasoning of its own. */ function reasoningThrough(calls: Call[], protectedFrom: number): number { if (calls.length === 0) return 0; return Math.min(calls[calls.length - 1].callMessage, protectedFrom); } /** What the reasoning of the callless assistant turns below `through` weighs. */ function orphanReasoningBytes(messages: AgentMessage[], through: number): number { let size = 0; for (let m = 0; m < through; m++) { const message = messages[m]; if (message?.role !== "assistant" || !Array.isArray(message.content)) continue; if (message.content.some((part: any) => part?.type === "toolCall")) continue; size += thinkingBytes(message); } return size; } /** * What a message's reasoning weighs, signatures included. * * The signature is opaque base64 several times the size of the text it * certifies, and it goes when the text goes: a block left with a signature and * no text is dropped whole by the provider adapter, so counting only the text * as foldable leaves the larger half sitting in the floor as though nothing * could ever move it. * * Redacted reasoning is the exception and is left alone: there the signature is * the payload the provider replays, not a certificate attached to text. */ function thinkingBytes(message: AgentMessage | undefined): number { const content = message?.content; if (!Array.isArray(content)) return 0; let size = 0; for (const block of content) { if (block?.type !== "thinking" || block.redacted) continue; size += byteLength(block.thinking ?? "") + byteLength(block.thinkingSignature ?? ""); } return size; } /** Folds one call's result and its reasoning, returning the bytes it saved. */ function applyTier(messages: AgentMessage[], call: Call, protectedFrom: number, imageTokens: number): number { let saved = 0; saved += stripReasoning(messages, call, protectedFrom); const result = messages[call.resultMessage]; if (result) { const before = messagesBytes([result], imageTokens); result.content = collapseResult(call, result.isError === true); saved += before - messagesBytes([result], imageTokens); } return saved; } function callPart(messages: AgentMessage[], call: Call): any { const content = messages[call.callMessage]?.content; if (!Array.isArray(content)) return undefined; const part = content[call.callPart]; return part?.type === "toolCall" && part.id === call.id ? part : undefined; } /** * Replaces a tool result with a notice naming what it weighed. * * The notice names nothing else. What the call was is on the call above, which * folding leaves whole, and the id to ask for the result back is on the result * block itself, where the provider requires it. The wording is not repeated * here either: the system prompt explains once what the notice means, which is * worth thousands of tokens across a session that folds hundreds of calls. */ function collapseResult(call: Call, isError: boolean): any[] { let text = ""; let bytes = 0; for (const part of call.resultContent) { if (part?.type === "text") { text += part.text ?? ""; bytes += byteLength(part.text ?? ""); continue; } // An image's payload is most of what a result of this kind costs, so a // notice that counted only its caption would understate what went. if (part?.type === "image") bytes += (part.data ?? "").length; } // A failure is worth more than a success of the same size: it is what stops // the model repeating a call that cannot work, so a folded one keeps both // ends of its text where a folded success keeps none. if (isError) return [{ type: "text", text: clampEnds(text) }]; return [{ type: "text", text: omission(bytes) }]; } export function omission(size: number): string { return `[omitted: ${size}B]`; } /** Keeps the beginning and the end of `text`, naming what was dropped between. */ function clampEnds(text: string): string { const size = byteLength(text); if (size <= ERROR_HEAD_BYTES + ERROR_TAIL_BYTES) return text; const head = cutBytes(text, ERROR_HEAD_BYTES); const tail = cutBytesFromEnd(text, ERROR_TAIL_BYTES); return `${head}\n${omission(size - byteLength(head) - byteLength(tail))}\n${tail}`; } function jsonText(value: unknown): string { if (value === undefined || value === null) return ""; if (typeof value === "string") return value; try { return JSON.stringify(value) ?? String(value); } catch { return String(value); } } /** The first `n` bytes of `text`, without splitting a character. */ export function cutBytes(text: string, n: number): string { if (byteLength(text) <= n) return text; let cut = text.slice(0, n); while (cut.length > 0 && byteLength(cut) > n) cut = cut.slice(0, -1); return cut; } /** The last `n` bytes of `text`, without splitting a character. */ export function cutBytesFromEnd(text: string, n: number): string { if (byteLength(text) <= n) return text; let cut = text.slice(-n); while (cut.length > 0 && byteLength(cut) > n) cut = cut.slice(1); return cut; }