/** * Provider capability manifests. * * The extension does NOT integrate with every GPU vendor. It ships one * first-class adapter — RunPod (runpodctl) — as a small capability table: * how to check auth, provision, stop, start, destroy, list, expressed as * command templates with structured output. * * Adding a provider = one table entry here (authCheck/provision/destroy/list) * + an install/auth recipe in the lvrged-factory-setup skill + a cheat sheet * in lvrged-factory-provider-adapters. No release needed — the extension * stays dumb, the agent learns the provider from the skills. * * All commands run via execFile (no shell interpolation of user input). */ import { execFile } from "node:child_process"; export interface CommandSpec { cmd: string; args: (vars: Record) => string[]; } export interface ProviderManifest { id: string; name: string; kind: "instance"; cli?: string; installHint: string; authCheck?: CommandSpec; provision?: CommandSpec; // returns instance id via parseInstanceId stop?: CommandSpec; // pause: GPU billing off, instance kept start?: CommandSpec; // resume a stopped instance destroy?: CommandSpec; list?: CommandSpec; get?: CommandSpec; // instance detail (SSH host/port, runtime status) notes: string; } /** * RunPod GPU ids are exact strings and NOT what `gpu list` shows as * displayName. Friendly name → exact `--gpu-id` value. (Session-verified * 2026-08-12: "RTX PRO 6000" displayName resolves to the Blackwell Server * Edition id below.) */ export const RUNPOD_GPU_IDS: Record = { "RTX PRO 6000": "NVIDIA RTX PRO 6000 Blackwell Server Edition", "RTX 5090": "NVIDIA GeForce RTX 5090", "RTX 4090": "NVIDIA GeForce RTX 4090", "RTX PRO 4500": "NVIDIA RTX PRO 4500", "H100 SXM": "NVIDIA H100 80GB SXM", "A100 SXM": "NVIDIA A100 80GB SXM", }; export function runpodGpuId(friendly: string): string { const key = Object.keys(RUNPOD_GPU_IDS).find((k) => friendly.toUpperCase().includes(k.toUpperCase())); return key ? RUNPOD_GPU_IDS[key] : friendly; } /** * The H3 lane image: the official RunPod ComfyUI CUDA 13 build. * Live-verified 2026-08-12: runpod-slim layout (/workspace/runpod-slim/ComfyUI * + comfyui_args.txt), torch 2.10.0+cu130 (in a venv confusingly named * .venv-cu128), hf CLI and nvcc present. Ships NO sageattention — the install * script builds real SageAttention2 (nvcc is on the image). * The cu12.8 sibling runs the H3 INT8 convrot path ~2x slower — do not use it * for benchmarks or production H3. * Alternative: hearmeman/comfyui-wan-template:v25-cuda13 claims prebuilt * Sage but its layout is unverified — only reach for it if the Sage build * on the official image fails. */ export const H3_IMAGE = "runpod/comfyui:cuda13.0"; export const H3_IMAGE_ALT = "hearmeman/comfyui-wan-template:v25-cuda13"; export const H3_CONTAINER_DISK_GB = 80; // image + ~41GB H3 weights + headroom /** * Datacenter rotation order for capacity fallback (US first for latency, * then EU). `--data-center-ids` only honors the FIRST id despite accepting a * comma list — callers must loop one DC per create attempt. */ export const RUNPOD_DCS = [ "US-NC-1", "US-NC-2", "US-PA-1", "US-MO-2", "US-KS-2", "US-NE-1", "CA-MTL-3", "EU-CZ-1", "EU-NL-1", "EU-RO-1", "EUR-IS-1", "EUR-IS-2", ]; export const PROVIDERS: Record = { runpod: { id: "runpod", name: "RunPod", kind: "instance", cli: "runpodctl", installHint: "Non-root: download the GitHub release binary to ~/.local/bin (see skills/lvrged-factory-setup). The cli.runpod.net installer needs root.", authCheck: { cmd: "runpodctl", args: () => ["pod", "list", "--output", "json"] }, provision: { cmd: "runpodctl", args: (v) => { const args = ["pod", "create", "--name", v.name || "lvrged-h3"]; // image-based create (the H3 lane) vs template-based (generic) if (v.image) args.push("--image", v.image); else if (v.template) args.push("--template-id", v.template); args.push( "--gpu-id", v.gpu, "--cloud-type", v.cloud || "COMMUNITY", "--container-disk-in-gb", v.disk || String(H3_CONTAINER_DISK_GB), "--volume-in-gb", "0", // custom images publish NO ports by default — without this the pod // comes up with SSH unreachable and needs a container-restarting // `pod update --ports` afterwards "--ports", "22/tcp,8188/http", ); if (v.dc) args.push("--data-center-ids", v.dc); return args; }, }, stop: { cmd: "runpodctl", args: (v) => ["pod", "stop", v.id] }, start: { cmd: "runpodctl", args: (v) => ["pod", "start", v.id] }, destroy: { cmd: "runpodctl", args: (v) => ["pod", "delete", v.id] }, list: { cmd: "runpodctl", args: () => ["pod", "list", "--output", "json"] }, get: { cmd: "runpodctl", args: (v) => ["pod", "get", v.id, "--output", "json"] }, notes: "Community vs Secure pool; stock status from `runpodctl gpu list` overstates availability — expect create rejections and rotate DCs. Weights on container disk (no network volume: volumes DC-lock the install). Stopped (EXITED) pods drop out of `pod list` — use `pod get `. Full flow in docs/providers/runpod.md.", }, }; export interface ExecResult { ok: boolean; stdout: string; stderr: string; error?: string; } export function runSpec(spec: CommandSpec, vars: Record, timeoutMs = 60000): Promise { return new Promise((resolve) => { execFile(spec.cmd, spec.args(vars), { timeout: timeoutMs, maxBuffer: 8 * 1024 * 1024 }, (err, stdout, stderr) => { if (err) { resolve({ ok: false, stdout: String(stdout || ""), stderr: String(stderr || ""), error: (err as Error).message }); } else { resolve({ ok: true, stdout: String(stdout || ""), stderr: String(stderr || "") }); } }); }); } export function cliExists(bin: string): Promise { return new Promise((resolve) => { execFile("sh", ["-c", `command -v ${bin} >/dev/null 2>&1 && ${bin} --version 2>/dev/null | head -1 || true`], { timeout: 10000 }, (err, stdout) => { resolve(!err && stdout.trim().length > 0); }); }); } /** Best-effort instance id extraction from a provider's provision output. */ export function parseInstanceId(provider: string, stdout: string): string | null { try { const j = JSON.parse(stdout); const candidates = [j.id, j.pod?.id, j.instance?.id, j.data?.id, j.instance_id]; for (const c of candidates) if (typeof c === "string" || typeof c === "number") return String(c); if (Array.isArray(j)) return String(j[0]?.id ?? ""); return null; } catch { const m = stdout.match(/(?:pod|instance)[_-]?id["':=\s]+([a-zA-Z0-9_-]{6,})/i); return m ? m[1] : null; } } /** * Classify a RunPod create failure. The two graphql errors mean different * things (session-verified): * "no longer any instances available" → GPU+cloud stock is gone in that DC → try the next DC / other pool * "does not have the resources" → the machine pool exists but the spec (usually container disk) doesn't fit → shrink disk, retry same DC */ export type CreateFailure = "no_stock" | "spec_too_big" | "other"; export function classifyCreateFailure(output: string): CreateFailure { const s = output.toLowerCase(); if (s.includes("no longer any instances available")) return "no_stock"; if (s.includes("does not have the resources")) return "spec_too_big"; return "other"; }