/** * Custom Compaction Extension * * Replaces the default compaction behavior with a full summary of the entire context. * Instead of keeping the last 20k tokens of conversation turns, this extension: * 1. Summarizes ALL messages (messagesToSummarize + turnPrefixMessages) * 2. Discards all old turns completely, keeping only the summary * * This example also demonstrates using a different model (Gemini Flash) for summarization, * which can be cheaper/faster than the main conversation model. * * Usage: * pi --extension examples/extensions/custom-compaction.ts */ import type { ExtensionAPI } from "@caupulican/pi-adaptative"; import { convertToLlm, serializeConversation } from "@caupulican/pi-adaptative"; import { complete } from "@caupulican/pi-ai"; export default function (pi: ExtensionAPI) { pi.on("session_before_compact", async (event, ctx) => { ctx.ui.notify("Custom compaction extension triggered", "info"); const { preparation, branchEntries: _, signal } = event; const { messagesToSummarize, turnPrefixMessages, tokensBefore, firstKeptEntryId, previousSummary } = preparation; // Use Gemini Flash for summarization (cheaper/faster than most conversation models) const model = ctx.modelRegistry.find("google", "gemini-2.5-flash"); if (!model) { ctx.ui.notify(`Could not find Gemini Flash model, using default compaction`, "warning"); return; } // Resolve request auth for the summarization model const auth = await ctx.modelRegistry.getApiKeyAndHeaders(model); if (!auth.ok) { ctx.ui.notify(`Compaction auth failed: ${auth.error}`, "warning"); return; } if (!auth.apiKey) { ctx.ui.notify(`No API key for ${model.provider}, using default compaction`, "warning"); return; } // Combine all messages for full summary const allMessages = [...messagesToSummarize, ...turnPrefixMessages]; ctx.ui.notify( `Custom compaction: summarizing ${allMessages.length} messages (${tokensBefore.toLocaleString()} tokens) with ${model.id}...`, "info", ); // Convert messages to readable text format const conversationText = serializeConversation(convertToLlm(allMessages)); // Build messages that ask for a comprehensive summary const summaryMessages = [ { role: "user" as const, content: [ { type: "text" as const, text: `Create complete replacement checkpoint. Preserve goals, mandatory rules, decisions/rationale, code/file details, current state, blockers/open questions, next steps. Omit resolved/transient noise. Structured Markdown, concise but sufficient to resume. OLD CHECKPOINT ${previousSummary ?? "(none)"} CHAT ${conversationText}`, }, ], timestamp: Date.now(), }, ]; try { // Pass signal to honor abort requests (e.g., user cancels compaction) const response = await complete( model, { messages: summaryMessages }, { apiKey: auth.apiKey, headers: auth.headers, maxTokens: 8192, signal, }, ); const summary = response.content .filter((c): c is { type: "text"; text: string } => c.type === "text") .map((c) => c.text) .join("\n"); if (!summary.trim()) { if (!signal.aborted) ctx.ui.notify("Compaction summary was empty, using default compaction", "warning"); return; } // Return compaction content - SessionManager adds id/parentId // Use firstKeptEntryId from preparation to keep recent messages return { compaction: { summary, firstKeptEntryId, tokensBefore, }, }; } catch (error) { const message = error instanceof Error ? error.message : String(error); ctx.ui.notify(`Compaction failed: ${message}`, "error"); // Fall back to default compaction on error return; } }); }