/** * pi-vision-bridge * * Gives text-only models (e.g. DeepSeek, Llama, etc.) the ability to "see" * images by delegating image analysis to a vision-capable model configured * in pi's model registry. * * Features: * 1. `describe_image` tool — the agent calls it mid-task when it needs to * understand an image (screenshot, UI, chart, photo). The image is sent * to the configured vision model and the description is returned as text. * The active model never changes. * 2. Automatic fallback — when the user pastes/attaches an image while the * active model has no vision capability, the image is transparently * converted to a text description before reaching the model, so the * conversation keeps working without errors. * * The vision model is resolved from pi's model registry (models.json), so any * provider that pi can talk to works: Google Gemini, Qwen VL, GLM, OpenAI, * local Ollama vision models, etc. No hardcoded keys, URLs, or proxies. */ import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent"; import type { ImageContent, Model, Api } from "@earendil-works/pi-ai"; import { Type } from "typebox"; import { existsSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { connect as tlsConnect } from "node:tls"; import { request as httpRequest } from "node:http"; import { request as httpsRequest } from "node:https"; // --------------------------------------------------------------------------- // Vision model resolution // --------------------------------------------------------------------------- /** * Score a model for auto-discovery. Higher is better. * Prefers standard chat-completions / generateContent formats over * special-purpose models (deep-research, computer-use, live, tts, image-gen). */ function visionModelScore(m: Model): number { const id = m.id.toLowerCase(); let score = 0; // Strongly prefer the two most common API formats if (m.api === "openai-completions") score += 20; else if (m.api === "google-generative-ai") score += 10; else if (m.api === "anthropic-messages") score += 5; // Penalize special-purpose models that usually can't do ad-hoc image chat if (/deep-research|computer-use|realtime|live-preview|\btts\b|\bimage\b|video|audio/.test(id)) score -= 30; // Slight preference for preview/stable flash-class models if (id.includes("flash") || id.includes("-vl") || id.includes("vision")) score += 2; if (id.includes("preview")) score += 1; return score; } /** * Resolve candidate vision models. * * Returns a list ordered by preference. With explicit env config the list has * a single entry; otherwise it is auto-discovered and sorted by score. The * caller should try candidates in order, falling back on failure. */ async function resolveVisionModels(ctx: ExtensionContext): Promise[]> { // 1. Explicit configuration via env vars const provider = process.env.PI_VISION_PROVIDER; const modelId = process.env.PI_VISION_MODEL; if (provider && modelId) { const model = ctx.modelRegistry.find(provider, modelId); return model ? [model] : []; } // 2. Auto-discovery. const all = ctx.modelRegistry.getAll(); const visionCapable = (m: Model) => m.input?.includes("image") && ctx.modelRegistry.hasConfiguredAuth(m); // 2a. Same provider as the active model first const currentProvider = ctx.model?.provider; const sameProvider = currentProvider ? all.filter((m) => m.provider === currentProvider && visionCapable(m)).sort((a, b) => visionModelScore(b) - visionModelScore(a)) : []; // 2b. Best-scoring vision-capable models with configured auth, capped at 5 const best = all.filter(visionCapable).sort((a, b) => visionModelScore(b) - visionModelScore(a)).slice(0, 5); // Dedupe while preserving order: sameProvider first, then the rest const seen = new Set(); return [...sameProvider, ...best].filter((m) => { const key = `${m.provider}/${m.id}`; if (seen.has(key)) return false; seen.add(key); return true; }); } // --------------------------------------------------------------------------- // HTTP transport (zero-dependency, honors standard proxy env vars) // --------------------------------------------------------------------------- /** * Resolve a proxy URL from standard environment variables, if any. * Honors: PI_VISION_PROXY > HTTPS_PROXY > HTTP_PROXY (and lowercase variants). * Returns undefined when no proxy is configured (direct connection). */ function resolveProxyEnv(): string | undefined { const candidates = [ process.env.PI_VISION_PROXY, process.env.HTTPS_PROXY, process.env.https_proxy, process.env.HTTP_PROXY, process.env.http_proxy, ]; for (const c of candidates) { if (c && c.trim()) return c.trim(); } return undefined; } /** Establish a TLS connection through an HTTP CONNECT proxy. */ function tunnelThroughProxy( proxyUrl: string, targetHost: string, targetPort: number, signal?: AbortSignal, ): Promise { let proxy: URL; try { proxy = new URL(proxyUrl); } catch { return Promise.reject(new Error(`Invalid proxy URL: ${proxyUrl}`)); } const proxyHost = proxy.hostname; const proxyPort = Number(proxy.port || 80); return new Promise((resolve, reject) => { const req = httpRequest({ host: proxyHost, port: proxyPort, method: "CONNECT", path: `${targetHost}:${targetPort}`, }); const onAbort = () => { req.destroy(new Error("aborted")); }; if (signal?.aborted) { reject(new Error("aborted")); return; } signal?.addEventListener("abort", onAbort, { once: true }); req.on("connect", (res, socket) => { if (res.statusCode !== 200) { socket.destroy(); return reject(new Error(`Proxy CONNECT failed: ${res.statusCode}`)); } const tls = tlsConnect({ socket, servername: targetHost }); tls.on("secureConnect", () => { signal?.removeEventListener("abort", onAbort); resolve(tls); }); tls.on("error", (e) => { signal?.removeEventListener("abort", onAbort); reject(e); }); }); req.on("error", (e) => { signal?.removeEventListener("abort", onAbort); reject(e); }); req.end(); }); } /** * POST JSON to an HTTPS URL. Uses the standard proxy env vars when set * (via a CONNECT tunnel), otherwise connects directly. */ async function postJson( url: string, body: string, headers: Record, signal?: AbortSignal, ): Promise<{ ok: boolean; status: number; text: string; json: any }> { const proxy = resolveProxyEnv(); if (!proxy) { // Direct connection with global fetch const res = await fetch(url, { method: "POST", headers, body, signal }); const text = await res.text().catch(() => ""); let json: any = undefined; try { json = JSON.parse(text); } catch { /* non-JSON response */ } return { ok: res.ok, status: res.status, text, json }; } // Proxied connection via CONNECT tunnel const target = new URL(url); const port = Number(target.port || 443); const socket = await tunnelThroughProxy(proxy, target.hostname, port, signal); const fullHeaders: Record = { "content-type": "application/json", "content-length": String(Buffer.byteLength(body)), host: target.host, ...headers, }; return new Promise((resolve, reject) => { const req = httpsRequest( { createConnection: () => socket, host: target.hostname, path: target.pathname + target.search, method: "POST", headers: fullHeaders, }, (res) => { let data = ""; res.on("data", (c) => (data += c)); res.on("end", () => { let json: any = undefined; try { json = JSON.parse(data); } catch { /* non-JSON */ } resolve({ ok: res.statusCode! >= 200 && res.statusCode! < 300, status: res.statusCode!, text: data, json }); }); }); req.on("error", reject); if (signal?.aborted) { req.destroy(new Error("aborted")); } else { signal?.addEventListener("abort", () => req.destroy(new Error("aborted")), { once: true }); } req.setTimeout(90000, () => req.destroy(new Error("Request timed out after 90s"))); req.end(body); }); } // --------------------------------------------------------------------------- // Vision call // --------------------------------------------------------------------------- /** Run the vision model with an image + prompt, returning the text response. */ async function runVision( ctx: ExtensionContext, model: Model, image: ImageContent, prompt: string, signal?: AbortSignal, ): Promise { const auth = await ctx.modelRegistry.getApiKeyAndHeaders(model); if (!auth.ok) { throw new Error(`Unable to resolve API key for "${model.provider}": ${auth.error}`); } const apiKey = auth.apiKey; const baseUrl = (model.baseUrl || "").replace(/\/+$/, ""); const systemPrompt = [ "You are an expert vision analysis assistant.", "Examine the provided image and respond to the request precisely.", "Be factual and do not invent details that are not visible in the image.", "Structure your response with markdown when appropriate.", ].join("\n"); const api = model.api; let response: { ok: boolean; status: number; text: string; json: any }; if (api === "google-generative-ai") { // Gemini native API const url = `${baseUrl}/models/${model.id}:generateContent`; const body = JSON.stringify({ contents: [ { parts: [ { text: `${systemPrompt}\n\n${prompt}` }, { inline_data: { mime_type: image.mimeType, data: image.data } }, ], }, ], }); const headers: Record = { ...(auth.headers ?? {}), }; if (apiKey && !headers["x-goog-api-key"]) headers["x-goog-api-key"] = apiKey; response = await postJson(url, body, headers, signal); if (!response.ok) { throw new Error(`Vision model returned ${response.status}: ${response.text.slice(0, 500)}`); } const text = (response.json?.candidates?.[0]?.content?.parts ?? []) .map((p: any) => p.text ?? "") .join(""); const trimmed = text.trim(); if (!trimmed) throw new Error(`Vision model "${model.provider}/${model.id}" returned no text.`); return trimmed; } // OpenAI-compatible API (covers openai, qwen, glm, ollama, openrouter, etc.) const url = `${baseUrl}/chat/completions`; const body = JSON.stringify({ model: model.id, temperature: 0, max_tokens: 4096, messages: [ { role: "system", content: systemPrompt }, { role: "user", content: [ { type: "image_url", image_url: { url: `data:${image.mimeType};base64,${image.data}` }, }, { type: "text", text: prompt }, ], }, ], }); const headers: Record = { ...(auth.headers ?? {}), }; if (apiKey && !headers["Authorization"]) headers["Authorization"] = `Bearer ${apiKey}`; response = await postJson(url, body, headers, signal); if (!response.ok) { throw new Error(`Vision model returned ${response.status}: ${response.text.slice(0, 500)}`); } const msg = response.json?.choices?.[0]?.message?.content?.trim(); if (!msg) throw new Error(`Vision model "${model.provider}/${model.id}" returned no text.`); return msg; } // --------------------------------------------------------------------------- // Image handling // --------------------------------------------------------------------------- /** Magic-byte detection for common image formats. */ function detectMime(buf: Buffer): string { if (buf.length > 8 && buf.subarray(0, 8).equals(Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]))) return "image/png"; if (buf.length > 3 && buf[0] === 0xff && buf[1] === 0xd8 && buf[2] === 0xff) return "image/jpeg"; if ( buf.length > 11 && buf.subarray(0, 4).toString() === "RIFF" && buf.subarray(8, 12).toString() === "WEBP" ) return "image/webp"; if (buf.length > 5 && (buf.subarray(0, 6).toString() === "GIF87a" || buf.subarray(0, 6).toString() === "GIF89a")) return "image/gif"; return "image/png"; } const MAX_IMAGE_BYTES = 20 * 1024 * 1024; // Most vision APIs cap single images at ~20MB /** Load an image file into an ImageContent block. */ function imageFromFile(absPath: string): ImageContent { if (!existsSync(absPath)) { throw new Error(`Image file not found: ${absPath}`); } const buf = readFileSync(absPath); if (buf.length === 0) { throw new Error(`Image file is empty: ${absPath}`); } if (buf.length > MAX_IMAGE_BYTES) { throw new Error(`Image exceeds the 20MB limit (${(buf.length / 1024 / 1024).toFixed(1)}MB): ${absPath}`); } return { type: "image", mimeType: detectMime(buf), data: buf.toString("base64") }; } /** Normalize a path argument (some models prepend "@" to paths). */ function normalizePath(p: string, cwd: string): string { const cleaned = p.startsWith("@") ? p.slice(1) : p; // Windows drive-letter absolute paths (D:\\foo, D:/foo) and UNC paths (\\\\server\\share) const isWindowsAbs = /^[A-Za-z]:[\\/]/.test(cleaned) || cleaned.startsWith("\\\\"); return cleaned.startsWith("/") || isWindowsAbs ? cleaned : join(cwd, cleaned); } const DEFAULT_PROMPT = "Describe this image in detail, including main elements, any text, layout, and notable details."; /** Shared implementation for the tool and the automatic fallback. */ async function describeImage( ctx: ExtensionContext, image: ImageContent, question: string | undefined, signal?: AbortSignal, ): Promise { const models = await resolveVisionModels(ctx); if (models.length === 0) { throw new Error( [ "No vision-capable model found in pi's model registry.", "Configure one in ~/.pi/agent/models.json (e.g. google/gemini-*, qwen-vl, glm-4v) with \"input\": [\"text\", \"image\"].", "Or set PI_VISION_PROVIDER and PI_VISION_MODEL environment variables.", ].join("\n"), ); } const prompt = question?.trim() ? question.trim() : DEFAULT_PROMPT; // Try candidates in order, falling back to the next on failure. const errors: string[] = []; for (const model of models) { try { return await runVision(ctx, model, image, prompt, signal); } catch (e) { const msg = e instanceof Error ? e.message : String(e); errors.push(`${model.provider}/${model.id}: ${msg}`); } } throw new Error(errors.join("\n")); } // --------------------------------------------------------------------------- // Extension entry // --------------------------------------------------------------------------- export default function piVisionBridge(pi: ExtensionAPI) { // Tool: agent calls this mid-task to understand an image file pi.registerTool({ name: "describe_image", label: "Describe Image", description: "Send an image file to a vision model and return a text description. Use when the task requires understanding an image (screenshot, UI, chart, photo) but the current model has no vision capability.", promptSnippet: "describe_image(path, question) — describe an image via a vision model, returns text", promptGuidelines: [ "Use describe_image when a task requires understanding an image (screenshot, UI mockup, chart, photo) and your current model has no vision capability.", "Use describe_image when the user asks you to look at an image file on disk or a screenshot you took.", ], parameters: Type.Object({ path: Type.String({ description: "Path to the image file (relative or absolute)" }), question: Type.Optional( Type.String({ description: "Optional: specific question about the image, e.g. \"What error messages are shown?\"" }), ), }), async execute(_toolCallId, params, signal, onUpdate, ctx) { const absPath = normalizePath(params.path, ctx.cwd); onUpdate?.({ content: [{ type: "text", text: `Analyzing image with vision model: ${absPath}...` }] }); const image = imageFromFile(absPath); const text = await describeImage(ctx, image, params.question, signal); return { content: [{ type: "text", text }], details: { path: absPath } }; }, }); // Automatic fallback: user attaches images while the active model is text-only pi.on("input", async (event, ctx) => { const images = event.images; if (!images || images.length === 0) return { action: "continue" }; // If the active model already supports images, let it handle them natively if (ctx.model?.input?.includes("image")) return { action: "continue" }; const descriptions: string[] = []; try { const models = await resolveVisionModels(ctx); if (models.length === 0) { return { action: "transform", text: `${event.text ? event.text + "\n\n" : ""}` + "[vision-bridge] Detected image attachments, but no vision-capable model is configured. " + "Set PI_VISION_PROVIDER/PI_VISION_MODEL or add a vision model to models.json.", images: [], }; } for (let i = 0; i < images.length; i++) { const img = images[i]; const text = await describeImage(ctx, img, DEFAULT_PROMPT); descriptions.push(`[Image ${i + 1}] ${text}`); } const injected = `[vision-bridge] Your active model has no vision capability, so the ${images.length} attached image(s) ` + `were described by a vision model instead:\n\n${descriptions.join("\n\n")}`; return { action: "transform", text: event.text ? `${event.text}\n\n${injected}` : injected, images: [], }; } catch (e) { const msg = e instanceof Error ? e.message : String(e); return { action: "transform", text: `${event.text ? event.text + "\n\n" : ""}[vision-bridge] Image description failed: ${msg}`, images: [], }; } }); }