import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; import { danglingSignatureBytes, Estimator, fixedBytes, messagesBytes } from "./estimate.js"; import { fold, FoldState, incompressibleTokens, tokensOf, type AgentMessage } from "./fold.js"; import { limitsFor, imageTokensFor, OVERFLOW_MARGIN, type PromptcapSettings } from "./limits.js"; export interface GuardHost { settings(): PromptcapSettings; notify?(message: string, level: "info" | "warning" | "error"): void; log?(event: Record, message: string): void; } /** * The guard answering for a session, found by the session manager it folds for. * * Subagents run in the same process as the session that spawned them, each with * its own guard, so a request refused for its size has to reach the one that * built it and no other. */ const guards = new WeakMap(); /** * Tells the guard that built the last prompt for `session` that the provider * refused it for its size, and reports whether anything will come of it. * * False means a retry would resend what was just rejected: either no guard * folds for this session, or the last emergency already took every image the * conversation had. */ export function noteOversizedRequest(session: any): boolean { const manager = session?.sessionManager; if (!manager) return false; return guards.get(manager)?.noteOversizedRequest() ?? false; } /** * Holds each prompt inside the limits its model was given, and learns what its * estimate is worth from what the provider charges. * * Folding runs on the copy the host hands the `context` event, which is the * last thing to touch the conversation before it becomes an LLM request. The * stored session is never edited, so the transcript, the UI and recall all see * every byte. */ export class PromptGuard { private readonly estimator = new Estimator(); private readonly folds = new FoldState(); private predicted: { modelKey: string; tokens: number; rawTokens: number; danglingSignatures: number } | null = null; /** The size the last fold settled on, for the footer and the menu. */ lastTokens: number | null = null; lastCeiling: number | null = null; /** Set by a request the provider refused for its size, cleared by the fold it asks for. */ private imageEmergency = false; /** Set once an emergency found no image left to take, so the next one promises nothing. */ private emergencyExhausted = false; constructor(private readonly host: GuardHost) {} /** Called when the conversation is replaced wholesale, not merely extended. */ reset(): void { this.folds.clear(); this.predicted = null; this.lastTokens = null; this.lastCeiling = null; this.imageEmergency = false; this.emergencyExhausted = false; } noteOversizedRequest(): boolean { // A guard that is not folding cannot make the next attempt any smaller, and // a retry it promised would resend exactly what was refused. if (this.emergencyExhausted || !this.host.settings().enabled) return false; this.imageEmergency = true; return true; } apply(messages: AgentMessage[], ctx: any, tools: unknown[]): AgentMessage[] { const settings = this.host.settings(); applyCacheRetention(settings); if (ctx?.sessionManager) guards.set(ctx.sessionManager, this); const modelKey = modelKeyOf(ctx); const systemPrompt = typeof ctx?.getSystemPrompt === "function" ? ctx.getSystemPrompt() : undefined; const fixed = fixedBytes(systemPrompt, tools); const ratio = this.estimator.ratioFor(modelKey) ?? 1; const window = ctx?.getContextUsage?.()?.contextWindow; const imageTokens = imageTokensFor(settings, modelKey); const floor = incompressibleTokens(messages, fixed, ratio, imageTokens); const limits = limitsFor(settings, modelKey, floor, typeof window === "number" ? window : undefined); // A body the provider refused is not a budget to be trimmed to but one to // be emptied: a ceiling of a single byte with nothing kept under it takes // every image folding may touch, which is the largest thing this can do to // a request between one attempt and the next. const emergency = this.imageEmergency; this.imageEmergency = false; if (emergency) { limits.imageCeiling = 1; limits.imageLowWater = 0; } if (!settings.enabled) { this.lastTokens = null; this.lastCeiling = null; return messages; } const result = fold(messages, fixed, limits, this.folds, ratio, imageTokens); if (emergency) { const freedNothing = result.imagesFolded === 0; // Another attempt at this can only help while an image is still there to // be taken, so a conversation already emptied of them stops promising a // retry that would resend exactly what was refused. this.emergencyExhausted = freedNothing || result.imageBytes === 0; this.host.log?.( { s: "promptcap", model: modelKey, imagesFolded: result.imagesFolded, imageBytes: result.imageBytes }, "the provider refused the request for its size; took the images off what it may", ); if (freedNothing) { this.host.notify?.( "The provider refused this request as too large and there was no image left to drop. Start a new session, or raise the request size limit on the gateway.", "error", ); } } // Carried to calibration so one real turn can answer whether reasoning // blocks left with a signature and no text reach the provider: the adapters // disagree, and this is the difference the answer would show up as. Scaled // by the same learned ratio as the prediction it will be subtracted from, // or the two would be in different units and the comparison would mean // nothing on any model that has learned a ratio. const dangling = tokensOf(danglingSignatureBytes(messages), ratio); // The raw count is what the next charge is measured against: a ratio // learned from a prediction the last ratio already scaled would fold its // own correction back in. this.predicted = { modelKey, tokens: result.tokens, rawTokens: tokensOf(fixed + messagesBytes(messages, imageTokens), 1), danglingSignatures: dangling, }; this.lastTokens = result.tokens; this.lastCeiling = limits.ceiling; if (result.rewroteFrom >= 0) { // Logged whatever it saved, and naming the message it moved: everything // after that message is re-billed, so a fold that freed little is a // charge to account for rather than a quiet success. this.host.log?.( { s: "promptcap", model: modelKey, before: result.tokensBefore, after: result.tokens, ceiling: limits.ceiling, lowWater: limits.lowWater, floor, folded: result.folded, promoted: result.promoted, imagesFolded: result.imagesFolded, imageBytes: result.imageBytes, rewroteFrom: result.rewroteFrom, messages: messages.length, }, "folded old tool calls, re-billing the prompt from the message it rewrote", ); } else if (result.held) { this.host.log?.( { s: "promptcap", model: modelKey, tokens: result.tokens, ceiling: limits.ceiling, floor }, "over the ceiling, but folding would not free enough to pay for the cache miss", ); } // Prose is never folded, so a conversation can outgrow its window on prose // alone. Warning rather than refusing keeps the turn the provider might // still accept: the estimate is an approximation, and the alternative is a // dead session with no way out but a new one. if (result.tokens > limits.ceiling * OVERFLOW_MARGIN) { this.host.log?.({ s: "promptcap", model: modelKey, tokens: result.tokens, ceiling: limits.ceiling }, "the conversation does not fit even fully folded"); this.host.notify?.( `This conversation no longer fits the model's window even with old tool calls folded away (~${Math.round(result.tokens / 1000)}K of ~${Math.round(limits.ceiling / 1000)}K). Start a new session, or switch to a model with a larger window.`, "warning", ); } return messages; } /** Teaches the estimator what the prompt it sized actually cost. */ calibrate(charged: number, modelKey: string): void { const predicted = this.predicted; this.predicted = null; if (!predicted || charged <= 0) return; // Something was billed for, so a prompt was accepted, and whatever the // provider refused before is no longer what this session is up against. // Read from the charge rather than from the turn ending, because a turn // ends on a refusal too — carrying no charge, and carrying the very // refusal this would be forgetting. this.emergencyExhausted = false; // A turn answered by a model other than the one the prompt was sized for // teaches the wrong estimator: the fallback path switches providers between // the fold and the response. if (predicted.modelKey !== modelKey) return; if (predicted.danglingSignatures > 0) { this.host.log?.( { s: "promptcap", model: modelKey, predicted: predicted.tokens, charged, danglingSignatures: predicted.danglingSignatures, predictedWithout: predicted.tokens - predicted.danglingSignatures, }, "signature-only reasoning blocks were in the prompt: charged sits at whichever prediction it matches", ); } this.estimator.observe(modelKey, predicted.rawTokens, charged); } ratioFor(modelKey: string): number | undefined { return this.estimator.ratioFor(modelKey); } } export function modelKeyOf(ctx: any): string { const provider = ctx?.model?.provider; const id = ctx?.model?.id; if (provider && id) return `${provider}/${id}`; return id ?? ""; } /** * Wires a guard into a session. `context` fires immediately before every LLM * call with a copy of the conversation; `turn_end` reports what the provider * charged for the prompt that copy became. */ export function registerPromptGuard(pi: ExtensionAPI, guard: PromptGuard): void { pi.on("context", (event, ctx) => { const messages = event?.messages as AgentMessage[] | undefined; if (!Array.isArray(messages)) return undefined; return { messages: guard.apply(messages, ctx, activeTools(pi)) as any }; }); pi.on("turn_end", (event: any, ctx) => { const usage = event?.message?.usage; if (!usage) return; const charged = (usage.input ?? 0) + (usage.cacheRead ?? 0) + (usage.cacheWrite ?? 0); guard.calibrate(charged, modelKeyOf(ctx)); }); } /** * Asks the Anthropic adapters for the hour-long prompt cache. * * The adapter reads this off the environment, so the reach is the process and * every Anthropic-shaped provider in it. An operator who has already stated a * preference keeps it; one who withdraws this setting mid-session gets the * provider's own default back, which is why what was set here is remembered * rather than assumed. */ function applyCacheRetention(settings: PromptcapSettings): void { if (settings.longCacheRetention === false) { if (appliedRetention && process.env.PI_CACHE_RETENTION === appliedRetention) delete process.env.PI_CACHE_RETENTION; appliedRetention = null; return; } if (process.env.PI_CACHE_RETENTION) return; process.env.PI_CACHE_RETENTION = "long"; appliedRetention = "long"; } /** What this process asked for, so withdrawing the setting can undo it. */ let appliedRetention: string | null = null; function activeTools(pi: ExtensionAPI): unknown[] { if (typeof pi.getAllTools !== "function") return []; const active = new Set(typeof pi.getActiveTools === "function" ? pi.getActiveTools() : []); return pi .getAllTools() .filter((tool) => active.size === 0 || active.has(tool.name)) .map((tool) => ({ name: tool.name, description: tool.description, parameters: tool.parameters })); }