import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; import { calculateCost, createAssistantMessageEventStream, openAICodexResponsesApi, type Api, type AssistantMessage, type AssistantMessageEvent, type AssistantMessageEventStream, type Context, type Model, type SimpleStreamOptions, type TextContent, type ThinkingContent, type ToolCall, type Usage, } from "@earendil-works/pi-ai/compat"; const STEP = 518; const MIN_N = 1; const MAX_N = 0; // 0 = no cap const MAX_CONTINUE = 3; const MARKER_TEXT = "Continue thinking..."; const builtinCodex = openAICodexResponsesApi(); type CodexModel = Model<"openai-codex-responses">; type FinalBlock = TextContent | ToolCall; interface RoundResult { message: AssistantMessage; reasoningBlocks: ThinkingContent[]; finalBlocks: FinalBlock[]; } function emptyUsage(): Usage { return { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, reasoning: 0, totalTokens: 0, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, }; } function cloneUsage(usage?: Usage): Usage { return { input: usage?.input ?? 0, output: usage?.output ?? 0, cacheRead: usage?.cacheRead ?? 0, cacheWrite: usage?.cacheWrite ?? 0, reasoning: usage?.reasoning ?? 0, totalTokens: usage?.totalTokens ?? 0, cost: { input: usage?.cost?.input ?? 0, output: usage?.cost?.output ?? 0, cacheRead: usage?.cost?.cacheRead ?? 0, cacheWrite: usage?.cost?.cacheWrite ?? 0, total: usage?.cost?.total ?? 0, }, }; } function addUsage(acc: Usage, usage?: Usage) { if (!usage) return; acc.input += usage.input ?? 0; acc.output += usage.output ?? 0; acc.cacheRead += usage.cacheRead ?? 0; acc.cacheWrite += usage.cacheWrite ?? 0; acc.reasoning = (acc.reasoning ?? 0) + (usage.reasoning ?? 0); acc.totalTokens += usage.totalTokens ?? 0; acc.cost.input += usage.cost?.input ?? 0; acc.cost.output += usage.cost?.output ?? 0; acc.cost.cacheRead += usage.cost?.cacheRead ?? 0; acc.cost.cacheWrite += usage.cost?.cacheWrite ?? 0; acc.cost.total += usage.cost?.total ?? 0; } function tierN(tokens: number | undefined): number | undefined { if (tokens === undefined || tokens < STEP - 2 || (tokens + 2) % STEP !== 0) return undefined; return (tokens + 2) / STEP; } function inContinueWindow(n: number | undefined): boolean { return n !== undefined && n >= MIN_N && (MAX_N === 0 || n <= MAX_N); } function parseReasoningItem(block: ThinkingContent): Record | undefined { if (!block.thinkingSignature) return undefined; try { const parsed = JSON.parse(block.thinkingSignature); return parsed && typeof parsed === "object" ? (parsed as Record) : undefined; } catch { return undefined; } } function hasEncryptedContent(block: ThinkingContent | undefined): boolean { return Boolean(block && parseReasoningItem(block)?.encrypted_content); } function cloneThinking(block: ThinkingContent): ThinkingContent { return { type: "thinking", thinking: block.thinking, ...(block.thinkingSignature ? { thinkingSignature: block.thinkingSignature } : {}), ...(block.redacted ? { redacted: block.redacted } : {}), }; } function cloneText(block: TextContent): TextContent { return { type: "text", text: block.text, ...(block.textSignature ? { textSignature: block.textSignature } : {}), }; } function cloneToolCall(block: ToolCall): ToolCall { return { type: "toolCall", id: block.id, name: block.name, arguments: { ...block.arguments }, ...(block.thoughtSignature ? { thoughtSignature: block.thoughtSignature } : {}), }; } function continuationMessage(model: CodexModel, reasoningBlocks: ThinkingContent[], roundNo: number): AssistantMessage { const textSignature = JSON.stringify({ v: 1, id: `msg_pi_gpt55_continue_${roundNo}`, phase: "commentary", }); return { role: "assistant", content: [ ...reasoningBlocks.map(cloneThinking), { type: "text", text: MARKER_TEXT, textSignature }, ], api: model.api, provider: model.provider, model: model.id, usage: emptyUsage(), stopReason: "stop", timestamp: Date.now(), }; } function contextWithReplay(context: Context, replay: AssistantMessage[]): Context { if (replay.length === 0) return context; return { ...context, messages: [...context.messages, ...replay] }; } function buildOuterMessage(model: CodexModel): AssistantMessage { return { role: "assistant", content: [], api: model.api, provider: model.provider, model: model.id, usage: emptyUsage(), stopReason: "pending", timestamp: Date.now(), }; } function syncReasoningFromInner(target: ThinkingContent, inner: unknown) { if (!inner || typeof inner !== "object" || (inner as { type?: string }).type !== "thinking") return; const block = inner as ThinkingContent; target.thinking = block.thinking; if (block.thinkingSignature) target.thinkingSignature = block.thinkingSignature; if (block.redacted) target.redacted = block.redacted; } function consumeRound( model: CodexModel, context: Context, options: SimpleStreamOptions | undefined, outer: AssistantMessage, stream: AssistantMessageEventStream, ): Promise { return (async () => { const inner = builtinCodex.streamSimple(model, context, options); const contentIndexMap = new Map(); const reasoningBlocks: ThinkingContent[] = []; for await (const event of inner) { if (event.type === "start") { if (!outer.responseId && event.partial.responseId) outer.responseId = event.partial.responseId; continue; } if (event.type === "thinking_start") { const block: ThinkingContent = { type: "thinking", thinking: "" }; outer.content.push(block); const outerIndex = outer.content.length - 1; contentIndexMap.set(event.contentIndex, outerIndex); reasoningBlocks.push(block); stream.push({ type: "thinking_start", contentIndex: outerIndex, partial: outer }); continue; } if (event.type === "thinking_delta") { const outerIndex = contentIndexMap.get(event.contentIndex); const block = outerIndex === undefined ? undefined : outer.content[outerIndex]; if (!block || block.type !== "thinking") continue; block.thinking += event.delta; stream.push({ type: "thinking_delta", contentIndex: outerIndex, delta: event.delta, partial: outer }); continue; } if (event.type === "thinking_end") { const outerIndex = contentIndexMap.get(event.contentIndex); const block = outerIndex === undefined ? undefined : outer.content[outerIndex]; if (!block || block.type !== "thinking") continue; syncReasoningFromInner(block, event.partial.content[event.contentIndex]); stream.push({ type: "thinking_end", contentIndex: outerIndex, content: block.thinking, partial: outer }); continue; } if (event.type === "done") { if (!outer.responseId && event.message.responseId) outer.responseId = event.message.responseId; // In case the provider emitted a final signature only on the done partial, // backfill the already-streamed outer reasoning blocks by order. const innerReasoning = event.message.content.filter((block): block is ThinkingContent => block.type === "thinking"); for (let i = 0; i < reasoningBlocks.length; i++) syncReasoningFromInner(reasoningBlocks[i], innerReasoning[i]); return { message: event.message, reasoningBlocks, finalBlocks: event.message.content.filter((block): block is FinalBlock => block.type !== "thinking"), }; } if (event.type === "error") { throw new Error(event.error.errorMessage || "Codex provider stream failed"); } // Text and tool-call events are tentative until this round is known clean. // They are replayed from the final message by flushFinalBlocks(). } throw new Error("Codex provider stream ended before a terminal event"); })(); } function flushFinalBlocks(blocks: FinalBlock[], outer: AssistantMessage, stream: AssistantMessageEventStream) { for (const block of blocks) { if (block.type === "text") { const out = cloneText(block); outer.content.push(out); const contentIndex = outer.content.length - 1; stream.push({ type: "text_start", contentIndex, partial: outer }); if (out.text.length > 0) { stream.push({ type: "text_delta", contentIndex, delta: out.text, partial: outer }); } stream.push({ type: "text_end", contentIndex, content: out.text, partial: outer }); } else if (block.type === "toolCall") { const out = cloneToolCall(block); outer.content.push(out); const contentIndex = outer.content.length - 1; stream.push({ type: "toolcall_start", contentIndex, partial: outer }); const args = JSON.stringify(out.arguments ?? {}); if (args.length > 0) { stream.push({ type: "toolcall_delta", contentIndex, delta: args, partial: outer }); } stream.push({ type: "toolcall_end", contentIndex, toolCall: out, partial: outer }); } } } function foldedAgentUsage(model: CodexModel, firstUsage: Usage | undefined, summedUsage: Usage, finalRoundUsage: Usage): Usage { const first = cloneUsage(firstUsage); const reasoning = summedUsage.reasoning ?? 0; const finalNonReasoning = Math.max(0, (finalRoundUsage.output ?? 0) - (finalRoundUsage.reasoning ?? 0)); const usage: Usage = { input: first.input, cacheRead: first.cacheRead, cacheWrite: first.cacheWrite, output: reasoning + finalNonReasoning, reasoning, totalTokens: first.input + first.cacheRead + first.cacheWrite + reasoning + finalNonReasoning, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, }; calculateCost(model, usage); return usage; } function finishWithError( stream: AssistantMessageEventStream, outer: AssistantMessage, error: unknown, rounds: unknown[], summedUsage: Usage, ) { outer.stopReason = "error"; outer.errorMessage = error instanceof Error ? error.message : String(error); outer.usage = summedUsage; outer.diagnostics = [ ...(outer.diagnostics ?? []), { type: "gpt55_reasoning_fold_error", timestamp: Date.now(), details: { rounds, billedUsage: summedUsage }, }, ]; stream.push({ type: "error", reason: "error", error: outer }); stream.end(outer); } function streamFoldedGpt55( model: CodexModel, context: Context, options?: SimpleStreamOptions, ): AssistantMessageEventStream { const stream = createAssistantMessageEventStream(); (async () => { const outer = buildOuterMessage(model); const replay: AssistantMessage[] = []; const rounds: Array<{ round: number; reasoningTokens: number | undefined; n: number | undefined }> = []; const summedUsage = emptyUsage(); let firstUsage: Usage | undefined; stream.push({ type: "start", partial: outer }); try { for (let roundNo = 1; ; roundNo++) { const result = await consumeRound(model, contextWithReplay(context, replay), options, outer, stream); const usage = result.message.usage; if (!firstUsage) firstUsage = cloneUsage(usage); addUsage(summedUsage, usage); const reasoningTokens = usage.reasoning; const n = tierN(reasoningTokens); rounds.push({ round: roundNo, reasoningTokens, n }); const lastReasoning = result.reasoningBlocks[result.reasoningBlocks.length - 1]; const hasEnc = hasEncryptedContent(lastReasoning); const doContinue = inContinueWindow(n) && hasEnc && roundNo <= MAX_CONTINUE; if (doContinue) { replay.push(continuationMessage(model, result.reasoningBlocks, roundNo)); continue; } const stoppedReason = n !== undefined ? !hasEnc ? "no_encrypted_content" : roundNo > MAX_CONTINUE ? "max_continue" : "tier_out_of_window" : undefined; flushFinalBlocks(result.finalBlocks, outer, stream); outer.usage = foldedAgentUsage(model, firstUsage, summedUsage, usage); outer.stopReason = result.message.stopReason; outer.rawStopReason = result.message.rawStopReason; outer.responseModel = result.message.responseModel; outer.diagnostics = [ ...(outer.diagnostics ?? []), { type: "gpt55_reasoning_fold", timestamp: Date.now(), details: { rounds, billedUsage: summedUsage, stoppedReason, }, }, ]; if (outer.stopReason === "error" || outer.stopReason === "aborted") { stream.push({ type: "error", reason: outer.stopReason, error: outer }); } else { stream.push({ type: "done", reason: outer.stopReason, message: outer }); } stream.end(outer); return; } } catch (error) { finishWithError(stream, outer, error, rounds, summedUsage); } })(); return stream; } function streamGpt55Complete( model: Model, context: Context, options?: SimpleStreamOptions, ): AssistantMessageEventStream { if (model.provider === "openai-codex" && model.id === "gpt-5.5" && model.api === "openai-codex-responses") { return streamFoldedGpt55(model as CodexModel, context, options); } return builtinCodex.streamSimple(model as CodexModel, context, options); } export default function gpt55Complete(pi: ExtensionAPI) { pi.registerProvider("openai-codex", { api: "openai-codex-responses", streamSimple: streamGpt55Complete, }); pi.on("model_select", (event, ctx) => { if (event.model.provider === "openai-codex" && event.model.id === "gpt-5.5") { ctx.ui.setStatus("gpt55-complete", "gpt-5.5 fold: on"); } else { ctx.ui.setStatus("gpt55-complete", ""); } }); }