{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../../src/commands/models.ts"],"names":[],"mappings":"AA0EA;;GAEG;AACH,eAAO,MAAM,UAAU;;;;;;mBA+UtB,CAAC;AAEF;;GAEG;AACH,eAAO,MAAM,SAAS;;mBA0BrB,CAAC;AAEF;;GAEG;AACH,eAAO,MAAM,aAAa;;mBA2BzB,CAAC;AAEF;;GAEG;AACH,eAAO,MAAM,UAAU;;mBAwEtB,CAAC;AAEF;;GAEG;AACH,eAAO,MAAM,QAAQ;;mBA+BpB,CAAC;AAEF;;GAEG;AACH,eAAO,MAAM,eAAe,qBA+J3B,CAAC","sourcesContent":["import chalk from \"chalk\";\nimport { spawn } from \"child_process\";\nimport { readFileSync } from \"fs\";\nimport { dirname, join } from \"path\";\nimport { fileURLToPath } from \"url\";\nimport { getActivePod, loadConfig, saveConfig } from \"../config.js\";\nimport { getModelConfig, getModelName, isKnownModel } from \"../model-configs.js\";\nimport { sshExec } from \"../ssh.js\";\nimport type { Pod } from \"../types.js\";\n\n/**\n * Get the pod to use (active or override)\n */\nconst getPod = (podOverride?: string): { name: string; pod: Pod } => {\n\tif (podOverride) {\n\t\tconst config = loadConfig();\n\t\tconst pod = config.pods[podOverride];\n\t\tif (!pod) {\n\t\t\tconsole.error(chalk.red(`Pod '${podOverride}' not found`));\n\t\t\tprocess.exit(1);\n\t\t}\n\t\treturn { name: podOverride, pod };\n\t}\n\n\tconst active = getActivePod();\n\tif (!active) {\n\t\tconsole.error(chalk.red(\"No active pod. Use 'pi pods active <name>' to set one.\"));\n\t\tprocess.exit(1);\n\t}\n\treturn active;\n};\n\n/**\n * Find next available port starting from 8001\n */\nconst getNextPort = (pod: Pod): number => {\n\tconst usedPorts = Object.values(pod.models).map((m) => m.port);\n\tlet port = 8001;\n\twhile (usedPorts.includes(port)) {\n\t\tport++;\n\t}\n\treturn port;\n};\n\n/**\n * Select GPUs for model deployment (round-robin)\n */\nconst selectGPUs = (pod: Pod, count: number = 1): number[] => {\n\tif (count === pod.gpus.length) {\n\t\t// Use all GPUs\n\t\treturn pod.gpus.map((g) => g.id);\n\t}\n\n\t// Count GPU usage across all models\n\tconst gpuUsage = new Map<number, number>();\n\tfor (const gpu of pod.gpus) {\n\t\tgpuUsage.set(gpu.id, 0);\n\t}\n\n\tfor (const model of Object.values(pod.models)) {\n\t\tfor (const gpuId of model.gpu) {\n\t\t\tgpuUsage.set(gpuId, (gpuUsage.get(gpuId) || 0) + 1);\n\t\t}\n\t}\n\n\t// Sort GPUs by usage (least used first)\n\tconst sortedGPUs = Array.from(gpuUsage.entries())\n\t\t.sort((a, b) => a[1] - b[1])\n\t\t.map((entry) => entry[0]);\n\n\t// Return the least used GPUs\n\treturn sortedGPUs.slice(0, count);\n};\n\n/**\n * Start a model\n */\nexport const startModel = async (\n\tmodelId: string,\n\tname: string,\n\toptions: {\n\t\tpod?: string;\n\t\tvllmArgs?: string[];\n\t\tmemory?: string;\n\t\tcontext?: string;\n\t\tgpus?: number;\n\t},\n) => {\n\tconst { name: podName, pod } = getPod(options.pod);\n\n\t// Validation\n\tif (!pod.modelsPath) {\n\t\tconsole.error(chalk.red(\"Pod does not have a models path configured\"));\n\t\tprocess.exit(1);\n\t}\n\tif (pod.models[name]) {\n\t\tconsole.error(chalk.red(`Model '${name}' already exists on pod '${podName}'`));\n\t\tprocess.exit(1);\n\t}\n\n\tconst port = getNextPort(pod);\n\n\t// Determine GPU allocation and vLLM args\n\tlet gpus: number[] = [];\n\tlet vllmArgs: string[] = [];\n\tlet modelConfig = null;\n\n\tif (options.vllmArgs?.length) {\n\t\t// Custom args override everything\n\t\tvllmArgs = options.vllmArgs;\n\t\tconsole.log(chalk.gray(\"Using custom vLLM args, GPU allocation managed by vLLM\"));\n\t} else if (isKnownModel(modelId)) {\n\t\t// Handle --gpus parameter for known models\n\t\tif (options.gpus) {\n\t\t\t// Validate GPU count\n\t\t\tif (options.gpus > pod.gpus.length) {\n\t\t\t\tconsole.error(chalk.red(`Error: Requested ${options.gpus} GPUs but pod only has ${pod.gpus.length}`));\n\t\t\t\tprocess.exit(1);\n\t\t\t}\n\n\t\t\t// Try to find config for requested GPU count\n\t\t\tmodelConfig = getModelConfig(modelId, pod.gpus, options.gpus);\n\t\t\tif (modelConfig) {\n\t\t\t\tgpus = selectGPUs(pod, options.gpus);\n\t\t\t\tvllmArgs = [...(modelConfig.args || [])];\n\t\t\t} else {\n\t\t\t\tconsole.error(\n\t\t\t\t\tchalk.red(`Model '${getModelName(modelId)}' does not have a configuration for ${options.gpus} GPU(s)`),\n\t\t\t\t);\n\t\t\t\tconsole.error(chalk.yellow(\"Available configurations:\"));\n\n\t\t\t\t// Show available configurations\n\t\t\t\tfor (let gpuCount = 1; gpuCount <= pod.gpus.length; gpuCount++) {\n\t\t\t\t\tconst config = getModelConfig(modelId, pod.gpus, gpuCount);\n\t\t\t\t\tif (config) {\n\t\t\t\t\t\tconsole.error(chalk.gray(`  - ${gpuCount} GPU(s)`));\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t\tprocess.exit(1);\n\t\t\t}\n\t\t} else {\n\t\t\t// Find best config for this hardware (original behavior)\n\t\t\tfor (let gpuCount = pod.gpus.length; gpuCount >= 1; gpuCount--) {\n\t\t\t\tmodelConfig = getModelConfig(modelId, pod.gpus, gpuCount);\n\t\t\t\tif (modelConfig) {\n\t\t\t\t\tgpus = selectGPUs(pod, gpuCount);\n\t\t\t\t\tvllmArgs = [...(modelConfig.args || [])];\n\t\t\t\t\tbreak;\n\t\t\t\t}\n\t\t\t}\n\t\t\tif (!modelConfig) {\n\t\t\t\tconsole.error(chalk.red(`Model '${getModelName(modelId)}' not compatible with this pod's GPUs`));\n\t\t\t\tprocess.exit(1);\n\t\t\t}\n\t\t}\n\t} else {\n\t\t// Unknown model\n\t\tif (options.gpus) {\n\t\t\tconsole.error(chalk.red(\"Error: --gpus can only be used with predefined models\"));\n\t\t\tconsole.error(chalk.yellow(\"For custom models, use --vllm with tensor-parallel-size or similar arguments\"));\n\t\t\tprocess.exit(1);\n\t\t}\n\t\t// Single GPU default\n\t\tgpus = selectGPUs(pod, 1);\n\t\tconsole.log(chalk.gray(\"Unknown model, defaulting to single GPU\"));\n\t}\n\n\t// Apply memory/context overrides\n\tif (!options.vllmArgs?.length) {\n\t\tif (options.memory) {\n\t\t\tconst fraction = parseFloat(options.memory.replace(\"%\", \"\")) / 100;\n\t\t\tvllmArgs = vllmArgs.filter((arg) => !arg.includes(\"gpu-memory-utilization\"));\n\t\t\tvllmArgs.push(\"--gpu-memory-utilization\", String(fraction));\n\t\t}\n\t\tif (options.context) {\n\t\t\tconst contextSizes: Record<string, number> = {\n\t\t\t\t\"4k\": 4096,\n\t\t\t\t\"8k\": 8192,\n\t\t\t\t\"16k\": 16384,\n\t\t\t\t\"32k\": 32768,\n\t\t\t\t\"64k\": 65536,\n\t\t\t\t\"128k\": 131072,\n\t\t\t};\n\t\t\tconst maxTokens = contextSizes[options.context.toLowerCase()] || parseInt(options.context, 10);\n\t\t\tvllmArgs = vllmArgs.filter((arg) => !arg.includes(\"max-model-len\"));\n\t\t\tvllmArgs.push(\"--max-model-len\", String(maxTokens));\n\t\t}\n\t}\n\n\t// Show what we're doing\n\tconsole.log(chalk.green(`Starting model '${name}' on pod '${podName}'...`));\n\tconsole.log(`Model: ${modelId}`);\n\tconsole.log(`Port: ${port}`);\n\tconsole.log(`GPU(s): ${gpus.length ? gpus.join(\", \") : \"Managed by vLLM\"}`);\n\tif (modelConfig?.notes) console.log(chalk.yellow(`Note: ${modelConfig.notes}`));\n\tconsole.log(\"\");\n\n\t// Read and customize model_run.sh script with our values\n\tconst scriptPath = join(dirname(fileURLToPath(import.meta.url)), \"../../scripts/model_run.sh\");\n\tlet scriptContent = readFileSync(scriptPath, \"utf-8\");\n\n\t// Replace placeholders - no escaping needed, heredoc with 'EOF' is literal\n\tscriptContent = scriptContent\n\t\t.replace(\"{{MODEL_ID}}\", modelId)\n\t\t.replace(\"{{NAME}}\", name)\n\t\t.replace(\"{{PORT}}\", String(port))\n\t\t.replace(\"{{VLLM_ARGS}}\", vllmArgs.join(\" \"));\n\n\t// Upload customized script\n\tawait sshExec(\n\t\tpod.ssh,\n\t\t`cat > /tmp/model_run_${name}.sh << 'EOF'\n${scriptContent}\nEOF\nchmod +x /tmp/model_run_${name}.sh`,\n\t);\n\n\t// Prepare environment\n\tconst env = [\n\t\t`HF_TOKEN='${process.env.HF_TOKEN}'`,\n\t\t`PI_API_KEY='${process.env.PI_API_KEY}'`,\n\t\t`HF_HUB_ENABLE_HF_TRANSFER=1`,\n\t\t`VLLM_NO_USAGE_STATS=1`,\n\t\t`PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True`,\n\t\t`FORCE_COLOR=1`,\n\t\t`TERM=xterm-256color`,\n\t\t...(gpus.length === 1 ? [`CUDA_VISIBLE_DEVICES=${gpus[0]}`] : []),\n\t\t...Object.entries(modelConfig?.env || {}).map(([k, v]) => `${k}='${v}'`),\n\t]\n\t\t.map((e) => `export ${e}`)\n\t\t.join(\"\\n\");\n\n\t// Start the model runner with script command for pseudo-TTY (preserves colors)\n\t// Note: We use script to preserve colors and create a log file\n\t// setsid creates a new session so it survives SSH disconnection\n\tconst startCmd = `\n\t\t${env}\n\t\tmkdir -p ~/.vllm_logs\n\t\t# Create a wrapper that monitors the script command\n\t\tcat > /tmp/model_wrapper_${name}.sh << 'WRAPPER'\n#!/bin/bash\nscript -q -f -c \"/tmp/model_run_${name}.sh\" ~/.vllm_logs/${name}.log\nexit_code=$?\necho \"Script exited with code $exit_code\" >> ~/.vllm_logs/${name}.log\nexit $exit_code\nWRAPPER\n\t\tchmod +x /tmp/model_wrapper_${name}.sh\n\t\tsetsid /tmp/model_wrapper_${name}.sh </dev/null >/dev/null 2>&1 &\n\t\techo $!\n\t\texit 0\n\t`;\n\n\tconst pidResult = await sshExec(pod.ssh, startCmd);\n\tconst pid = parseInt(pidResult.stdout.trim(), 10);\n\tif (!pid) {\n\t\tconsole.error(chalk.red(\"Failed to start model runner\"));\n\t\tprocess.exit(1);\n\t}\n\n\t// Save to config\n\tconst config = loadConfig();\n\tconfig.pods[podName].models[name] = { model: modelId, port, gpu: gpus, pid };\n\tsaveConfig(config);\n\n\tconsole.log(`Model runner started with PID: ${pid}`);\n\tconsole.log(\"Streaming logs... (waiting for startup)\\n\");\n\n\t// Small delay to ensure log file is created\n\tawait new Promise((resolve) => setTimeout(resolve, 500));\n\n\t// Stream logs with color support, watching for startup complete\n\tconst sshParts = pod.ssh.split(\" \");\n\tconst sshCommand = sshParts[0]; // \"ssh\"\n\tconst sshArgs = sshParts.slice(1); // [\"root@86.38.238.55\"]\n\tconst host = sshArgs[0].split(\"@\")[1] || \"localhost\";\n\tconst tailCmd = `tail -f ~/.vllm_logs/${name}.log`;\n\n\t// Build the full args array for spawn\n\tconst fullArgs = [...sshArgs, tailCmd];\n\n\tconst logProcess = spawn(sshCommand, fullArgs, {\n\t\tstdio: [\"inherit\", \"pipe\", \"pipe\"], // capture stdout and stderr\n\t\tenv: { ...process.env, FORCE_COLOR: \"1\" },\n\t});\n\n\tlet interrupted = false;\n\tlet startupComplete = false;\n\tlet startupFailed = false;\n\tlet failureReason = \"\";\n\n\t// Handle Ctrl+C\n\tconst sigintHandler = () => {\n\t\tinterrupted = true;\n\t\tlogProcess.kill();\n\t};\n\tprocess.on(\"SIGINT\", sigintHandler);\n\n\t// Process log output line by line\n\tconst processOutput = (data: Buffer) => {\n\t\tconst lines = data.toString().split(\"\\n\");\n\t\tfor (const line of lines) {\n\t\t\tif (line) {\n\t\t\t\tconsole.log(line); // Echo the line to console\n\n\t\t\t\t// Check for startup complete message\n\t\t\t\tif (line.includes(\"Application startup complete\")) {\n\t\t\t\t\tstartupComplete = true;\n\t\t\t\t\tlogProcess.kill(); // Stop tailing logs\n\t\t\t\t}\n\n\t\t\t\t// Check for failure indicators\n\t\t\t\tif (line.includes(\"Model runner exiting with code\") && !line.includes(\"code 0\")) {\n\t\t\t\t\tstartupFailed = true;\n\t\t\t\t\tfailureReason = \"Model runner failed to start\";\n\t\t\t\t\tlogProcess.kill();\n\t\t\t\t}\n\t\t\t\tif (line.includes(\"Script exited with code\") && !line.includes(\"code 0\")) {\n\t\t\t\t\tstartupFailed = true;\n\t\t\t\t\tfailureReason = \"Script failed to execute\";\n\t\t\t\t\tlogProcess.kill();\n\t\t\t\t}\n\t\t\t\tif (line.includes(\"torch.OutOfMemoryError\") || line.includes(\"CUDA out of memory\")) {\n\t\t\t\t\tstartupFailed = true;\n\t\t\t\t\tfailureReason = \"Out of GPU memory (OOM)\";\n\t\t\t\t\t// Don't kill immediately - let it show more error context\n\t\t\t\t}\n\t\t\t\tif (line.includes(\"RuntimeError: Engine core initialization failed\")) {\n\t\t\t\t\tstartupFailed = true;\n\t\t\t\t\tfailureReason = \"vLLM engine initialization failed\";\n\t\t\t\t\tlogProcess.kill();\n\t\t\t\t}\n\t\t\t}\n\t\t}\n\t};\n\n\tlogProcess.stdout?.on(\"data\", processOutput);\n\tlogProcess.stderr?.on(\"data\", processOutput);\n\n\tawait new Promise<void>((resolve) => logProcess.on(\"exit\", resolve));\n\tprocess.removeListener(\"SIGINT\", sigintHandler);\n\n\tif (startupFailed) {\n\t\t// Model failed to start - clean up and report error\n\t\tconsole.log(`\\n${chalk.red(`✗ Model failed to start: ${failureReason}`)}`);\n\n\t\t// Remove the failed model from config\n\t\tconst config = loadConfig();\n\t\tdelete config.pods[podName].models[name];\n\t\tsaveConfig(config);\n\n\t\tconsole.log(chalk.yellow(\"\\nModel has been removed from configuration.\"));\n\n\t\t// Provide helpful suggestions based on failure reason\n\t\tif (failureReason.includes(\"OOM\") || failureReason.includes(\"memory\")) {\n\t\t\tconsole.log(`\\n${chalk.bold(\"Suggestions:\")}`);\n\t\t\tconsole.log(\"  • Try reducing GPU memory utilization: --memory 50%\");\n\t\t\tconsole.log(\"  • Use a smaller context window: --context 4k\");\n\t\t\tconsole.log(\"  • Use a quantized version of the model (e.g., FP8)\");\n\t\t\tconsole.log(\"  • Use more GPUs with tensor parallelism\");\n\t\t\tconsole.log(\"  • Try a smaller model variant\");\n\t\t}\n\n\t\tconsole.log(`\\n${chalk.cyan(`Check full logs: pi ssh \"tail -100 ~/.vllm_logs/${name}.log\"`)}`);\n\t\tprocess.exit(1);\n\t} else if (startupComplete) {\n\t\t// Model started successfully - output connection details\n\t\tconsole.log(`\\n${chalk.green(\"✓ Model started successfully!\")}`);\n\t\tconsole.log(`\\n${chalk.bold(\"Connection Details:\")}`);\n\t\tconsole.log(chalk.cyan(\"─\".repeat(50)));\n\t\tconsole.log(chalk.white(\"Base URL:    \") + chalk.yellow(`http://${host}:${port}/v1`));\n\t\tconsole.log(chalk.white(\"Model:       \") + chalk.yellow(modelId));\n\t\tconsole.log(chalk.white(\"API Key:     \") + chalk.yellow(process.env.PI_API_KEY || \"(not set)\"));\n\t\tconsole.log(chalk.cyan(\"─\".repeat(50)));\n\n\t\tconsole.log(`\\n${chalk.bold(\"Export for shell:\")}`);\n\t\tconsole.log(chalk.gray(`export OPENAI_BASE_URL=\"http://${host}:${port}/v1\"`));\n\t\tconsole.log(chalk.gray(`export OPENAI_API_KEY=\"${process.env.PI_API_KEY || \"your-api-key\"}\"`));\n\t\tconsole.log(chalk.gray(`export OPENAI_MODEL=\"${modelId}\"`));\n\n\t\tconsole.log(`\\n${chalk.bold(\"Example usage:\")}`);\n\t\tconsole.log(\n\t\t\tchalk.gray(`\n  # Python\n  from openai import OpenAI\n  client = OpenAI()  # Uses env vars\n  response = client.chat.completions.create(\n      model=\"${modelId}\",\n      messages=[{\"role\": \"user\", \"content\": \"Hello!\"}]\n  )\n\n  # CLI\n  curl $OPENAI_BASE_URL/chat/completions \\\\\n    -H \"Authorization: Bearer $OPENAI_API_KEY\" \\\\\n    -H \"Content-Type: application/json\" \\\\\n    -d '{\"model\":\"${modelId}\",\"messages\":[{\"role\":\"user\",\"content\":\"Hi\"}]}'`),\n\t\t);\n\t\tconsole.log(\"\");\n\t\tconsole.log(chalk.cyan(`Chat with model:  pi agent ${name} \"Your message\"`));\n\t\tconsole.log(chalk.cyan(`Interactive mode: pi agent ${name} -i`));\n\t\tconsole.log(chalk.cyan(`Monitor logs:     pi logs ${name}`));\n\t\tconsole.log(chalk.cyan(`Stop model:       pi stop ${name}`));\n\t} else if (interrupted) {\n\t\tconsole.log(chalk.yellow(\"\\n\\nStopped monitoring. Model deployment continues in background.\"));\n\t\tconsole.log(chalk.cyan(`Chat with model: pi agent ${name} \"Your message\"`));\n\t\tconsole.log(chalk.cyan(`Check status: pi logs ${name}`));\n\t\tconsole.log(chalk.cyan(`Stop model: pi stop ${name}`));\n\t} else {\n\t\tconsole.log(chalk.yellow(\"\\n\\nLog stream ended. Model may still be running.\"));\n\t\tconsole.log(chalk.cyan(`Chat with model: pi agent ${name} \"Your message\"`));\n\t\tconsole.log(chalk.cyan(`Check status: pi logs ${name}`));\n\t\tconsole.log(chalk.cyan(`Stop model: pi stop ${name}`));\n\t}\n};\n\n/**\n * Stop a model\n */\nexport const stopModel = async (name: string, options: { pod?: string }) => {\n\tconst { name: podName, pod } = getPod(options.pod);\n\n\tconst model = pod.models[name];\n\tif (!model) {\n\t\tconsole.error(chalk.red(`Model '${name}' not found on pod '${podName}'`));\n\t\tprocess.exit(1);\n\t}\n\n\tconsole.log(chalk.yellow(`Stopping model '${name}' on pod '${podName}'...`));\n\n\t// Kill the script process and all its children\n\t// Using pkill to kill the process and all children\n\tconst killCmd = `\n\t\t# Kill the script process and all its children\n\t\tpkill -TERM -P ${model.pid} 2>/dev/null || true\n\t\tkill ${model.pid} 2>/dev/null || true\n\t`;\n\tawait sshExec(pod.ssh, killCmd);\n\n\t// Remove from config\n\tconst config = loadConfig();\n\tdelete config.pods[podName].models[name];\n\tsaveConfig(config);\n\n\tconsole.log(chalk.green(`✓ Model '${name}' stopped`));\n};\n\n/**\n * Stop all models on a pod\n */\nexport const stopAllModels = async (options: { pod?: string }) => {\n\tconst { name: podName, pod } = getPod(options.pod);\n\n\tconst modelNames = Object.keys(pod.models);\n\tif (modelNames.length === 0) {\n\t\tconsole.log(`No models running on pod '${podName}'`);\n\t\treturn;\n\t}\n\n\tconsole.log(chalk.yellow(`Stopping ${modelNames.length} model(s) on pod '${podName}'...`));\n\n\t// Kill all script processes and their children\n\tconst pids = Object.values(pod.models).map((m) => m.pid);\n\tconst killCmd = `\n\t\tfor PID in ${pids.join(\" \")}; do\n\t\t\tpkill -TERM -P $PID 2>/dev/null || true\n\t\t\tkill $PID 2>/dev/null || true\n\t\tdone\n\t`;\n\tawait sshExec(pod.ssh, killCmd);\n\n\t// Clear all models from config\n\tconst config = loadConfig();\n\tconfig.pods[podName].models = {};\n\tsaveConfig(config);\n\n\tconsole.log(chalk.green(`✓ Stopped all models: ${modelNames.join(\", \")}`));\n};\n\n/**\n * List all models\n */\nexport const listModels = async (options: { pod?: string }) => {\n\tconst { name: podName, pod } = getPod(options.pod);\n\n\tconst modelNames = Object.keys(pod.models);\n\tif (modelNames.length === 0) {\n\t\tconsole.log(`No models running on pod '${podName}'`);\n\t\treturn;\n\t}\n\n\t// Get pod SSH host for URL display\n\tconst sshParts = pod.ssh.split(\" \");\n\tconst host = sshParts.find((p) => p.includes(\"@\"))?.split(\"@\")[1] || \"unknown\";\n\n\tconsole.log(`Models on pod '${chalk.bold(podName)}':`);\n\tfor (const name of modelNames) {\n\t\tconst model = pod.models[name];\n\t\tconst gpuStr =\n\t\t\tmodel.gpu.length > 1\n\t\t\t\t? `GPUs ${model.gpu.join(\",\")}`\n\t\t\t\t: model.gpu.length === 1\n\t\t\t\t\t? `GPU ${model.gpu[0]}`\n\t\t\t\t\t: \"GPU unknown\";\n\t\tconsole.log(`  ${chalk.green(name)} - Port ${model.port} - ${gpuStr} - PID ${model.pid}`);\n\t\tconsole.log(`    Model: ${chalk.gray(model.model)}`);\n\t\tconsole.log(`    URL: ${chalk.cyan(`http://${host}:${model.port}/v1`)}`);\n\t}\n\n\t// Optionally verify processes are still running\n\tconsole.log(\"\");\n\tconsole.log(\"Verifying processes...\");\n\tlet anyDead = false;\n\tfor (const name of modelNames) {\n\t\tconst model = pod.models[name];\n\t\t// Check both the wrapper process and if vLLM is responding\n\t\tconst checkCmd = `\n\t\t\t# Check if wrapper process exists\n\t\t\tif ps -p ${model.pid} > /dev/null 2>&1; then\n\t\t\t\t# Process exists, now check if vLLM is responding\n\t\t\t\tif curl -s -f http://localhost:${model.port}/health > /dev/null 2>&1; then\n\t\t\t\t\techo \"running\"\n\t\t\t\telse\n\t\t\t\t\t# Check if it's still starting up\n\t\t\t\t\tif tail -n 20 ~/.vllm_logs/${name}.log 2>/dev/null | grep -q \"ERROR\\\\|Failed\\\\|Cuda error\\\\|died\"; then\n\t\t\t\t\t\techo \"crashed\"\n\t\t\t\t\telse\n\t\t\t\t\t\techo \"starting\"\n\t\t\t\t\tfi\n\t\t\t\tfi\n\t\t\telse\n\t\t\t\techo \"dead\"\n\t\t\tfi\n\t\t`;\n\t\tconst result = await sshExec(pod.ssh, checkCmd);\n\t\tconst status = result.stdout.trim();\n\t\tif (status === \"dead\") {\n\t\t\tconsole.log(chalk.red(`  ${name}: Process ${model.pid} is not running`));\n\t\t\tanyDead = true;\n\t\t} else if (status === \"crashed\") {\n\t\t\tconsole.log(chalk.red(`  ${name}: vLLM crashed (check logs with 'pi logs ${name}')`));\n\t\t\tanyDead = true;\n\t\t} else if (status === \"starting\") {\n\t\t\tconsole.log(chalk.yellow(`  ${name}: Still starting up...`));\n\t\t}\n\t}\n\n\tif (anyDead) {\n\t\tconsole.log(\"\");\n\t\tconsole.log(chalk.yellow(\"Some models are not running. Clean up with:\"));\n\t\tconsole.log(chalk.cyan(\"  pi stop <name>\"));\n\t} else {\n\t\tconsole.log(chalk.green(\"✓ All processes verified\"));\n\t}\n};\n\n/**\n * View model logs\n */\nexport const viewLogs = async (name: string, options: { pod?: string }) => {\n\tconst { name: podName, pod } = getPod(options.pod);\n\n\tconst model = pod.models[name];\n\tif (!model) {\n\t\tconsole.error(chalk.red(`Model '${name}' not found on pod '${podName}'`));\n\t\tprocess.exit(1);\n\t}\n\n\tconsole.log(chalk.green(`Streaming logs for '${name}' on pod '${podName}'...`));\n\tconsole.log(chalk.gray(\"Press Ctrl+C to stop\"));\n\tconsole.log(\"\");\n\n\t// Stream logs with color preservation\n\tconst sshParts = pod.ssh.split(\" \");\n\tconst sshCommand = sshParts[0]; // \"ssh\"\n\tconst sshArgs = sshParts.slice(1); // [\"root@86.38.238.55\"]\n\tconst tailCmd = `tail -f ~/.vllm_logs/${name}.log`;\n\n\tconst logProcess = spawn(sshCommand, [...sshArgs, tailCmd], {\n\t\tstdio: \"inherit\",\n\t\tenv: {\n\t\t\t...process.env,\n\t\t\tFORCE_COLOR: \"1\",\n\t\t},\n\t});\n\n\t// Wait for process to exit\n\tawait new Promise<void>((resolve) => {\n\t\tlogProcess.on(\"exit\", () => resolve());\n\t});\n};\n\n/**\n * Show known models and their hardware requirements\n */\nexport const showKnownModels = async () => {\n\tconst __filename = fileURLToPath(import.meta.url);\n\tconst __dirname = dirname(__filename);\n\tconst modelsJsonPath = join(__dirname, \"..\", \"models.json\");\n\tconst modelsJson = JSON.parse(readFileSync(modelsJsonPath, \"utf-8\"));\n\tconst models = modelsJson.models;\n\n\t// Get active pod info if available\n\tconst activePod = getActivePod();\n\tlet podGpuCount = 0;\n\tlet podGpuType = \"\";\n\n\tif (activePod) {\n\t\tpodGpuCount = activePod.pod.gpus.length;\n\t\t// Extract GPU type from name (e.g., \"NVIDIA H200\" -> \"H200\")\n\t\tpodGpuType = activePod.pod.gpus[0]?.name?.replace(\"NVIDIA\", \"\")?.trim()?.split(\" \")[0] || \"\";\n\n\t\tconsole.log(chalk.bold(`Known Models for ${activePod.name} (${podGpuCount}x ${podGpuType || \"GPU\"}):\\n`));\n\t} else {\n\t\tconsole.log(chalk.bold(\"Known Models:\\n\"));\n\t\tconsole.log(chalk.yellow(\"No active pod. Use 'pi pods active <name>' to filter compatible models.\\n\"));\n\t}\n\n\tconsole.log(\"Usage: pi start <model> --name <name> [options]\\n\");\n\n\t// Group models by compatibility and family\n\tconst compatible: Record<string, Array<{ id: string; name: string; config: string; notes?: string }>> = {};\n\tconst incompatible: Record<string, Array<{ id: string; name: string; minGpu: string; notes?: string }>> = {};\n\n\tfor (const [modelId, info] of Object.entries(models)) {\n\t\tconst modelInfo = info as any;\n\t\tconst family = modelInfo.name.split(\"-\")[0] || \"Other\";\n\n\t\tlet isCompatible = false;\n\t\tlet compatibleConfig = \"\";\n\t\tlet minGpu = \"Unknown\";\n\t\tlet minNotes: string | undefined;\n\n\t\tif (modelInfo.configs && modelInfo.configs.length > 0) {\n\t\t\t// Sort configs by GPU count to find minimum\n\t\t\tconst sortedConfigs = [...modelInfo.configs].sort((a: any, b: any) => (a.gpuCount || 1) - (b.gpuCount || 1));\n\n\t\t\t// Find minimum requirements\n\t\t\tconst minConfig = sortedConfigs[0];\n\t\t\tconst minGpuCount = minConfig.gpuCount || 1;\n\t\t\tconst gpuTypes = minConfig.gpuTypes?.join(\"/\") || \"H100/H200\";\n\n\t\t\tif (minGpuCount === 1) {\n\t\t\t\tminGpu = `1x ${gpuTypes}`;\n\t\t\t} else {\n\t\t\t\tminGpu = `${minGpuCount}x ${gpuTypes}`;\n\t\t\t}\n\n\t\t\tminNotes = minConfig.notes || modelInfo.notes;\n\n\t\t\t// Check compatibility with active pod\n\t\t\tif (activePod && podGpuCount > 0) {\n\t\t\t\t// Find best matching config for this pod\n\t\t\t\tfor (const config of sortedConfigs) {\n\t\t\t\t\tconst configGpuCount = config.gpuCount || 1;\n\t\t\t\t\tconst configGpuTypes = config.gpuTypes || [];\n\n\t\t\t\t\t// Check if we have enough GPUs\n\t\t\t\t\tif (configGpuCount <= podGpuCount) {\n\t\t\t\t\t\t// Check if GPU type matches (if specified)\n\t\t\t\t\t\tif (\n\t\t\t\t\t\t\tconfigGpuTypes.length === 0 ||\n\t\t\t\t\t\t\tconfigGpuTypes.some((type: string) => podGpuType.includes(type) || type.includes(podGpuType))\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\tisCompatible = true;\n\t\t\t\t\t\t\tif (configGpuCount === 1) {\n\t\t\t\t\t\t\t\tcompatibleConfig = `1x ${podGpuType}`;\n\t\t\t\t\t\t\t} else {\n\t\t\t\t\t\t\t\tcompatibleConfig = `${configGpuCount}x ${podGpuType}`;\n\t\t\t\t\t\t\t}\n\t\t\t\t\t\t\tminNotes = config.notes || modelInfo.notes;\n\t\t\t\t\t\t\tbreak;\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\t\t}\n\n\t\tconst modelEntry = {\n\t\t\tid: modelId,\n\t\t\tname: modelInfo.name,\n\t\t\tnotes: minNotes,\n\t\t};\n\n\t\tif (activePod && isCompatible) {\n\t\t\tif (!compatible[family]) {\n\t\t\t\tcompatible[family] = [];\n\t\t\t}\n\t\t\tcompatible[family].push({ ...modelEntry, config: compatibleConfig });\n\t\t} else {\n\t\t\tif (!incompatible[family]) {\n\t\t\t\tincompatible[family] = [];\n\t\t\t}\n\t\t\tincompatible[family].push({ ...modelEntry, minGpu });\n\t\t}\n\t}\n\n\t// Display compatible models first\n\tif (activePod && Object.keys(compatible).length > 0) {\n\t\tconsole.log(chalk.green.bold(\"✓ Compatible Models:\\n\"));\n\n\t\tconst sortedFamilies = Object.keys(compatible).sort();\n\t\tfor (const family of sortedFamilies) {\n\t\t\tconsole.log(chalk.cyan(`${family} Models:`));\n\n\t\t\tconst modelList = compatible[family].sort((a, b) => a.name.localeCompare(b.name));\n\n\t\t\tfor (const model of modelList) {\n\t\t\t\tconsole.log(`  ${chalk.green(model.id)}`);\n\t\t\t\tconsole.log(`    Name: ${model.name}`);\n\t\t\t\tconsole.log(`    Config: ${model.config}`);\n\t\t\t\tif (model.notes) {\n\t\t\t\t\tconsole.log(chalk.gray(`    Note: ${model.notes}`));\n\t\t\t\t}\n\t\t\t\tconsole.log(\"\");\n\t\t\t}\n\t\t}\n\t}\n\n\t// Display incompatible models\n\tif (Object.keys(incompatible).length > 0) {\n\t\tif (activePod && Object.keys(compatible).length > 0) {\n\t\t\tconsole.log(chalk.red.bold(\"✗ Incompatible Models (need more/different GPUs):\\n\"));\n\t\t}\n\n\t\tconst sortedFamilies = Object.keys(incompatible).sort();\n\t\tfor (const family of sortedFamilies) {\n\t\t\tif (!activePod) {\n\t\t\t\tconsole.log(chalk.cyan(`${family} Models:`));\n\t\t\t} else {\n\t\t\t\tconsole.log(chalk.gray(`${family} Models:`));\n\t\t\t}\n\n\t\t\tconst modelList = incompatible[family].sort((a, b) => a.name.localeCompare(b.name));\n\n\t\t\tfor (const model of modelList) {\n\t\t\t\tconst color = activePod ? chalk.gray : chalk.green;\n\t\t\t\tconsole.log(`  ${color(model.id)}`);\n\t\t\t\tconsole.log(chalk.gray(`    Name: ${model.name}`));\n\t\t\t\tconsole.log(chalk.gray(`    Min Hardware: ${model.minGpu}`));\n\t\t\t\tif (model.notes && !activePod) {\n\t\t\t\t\tconsole.log(chalk.gray(`    Note: ${model.notes}`));\n\t\t\t\t}\n\t\t\t\tif (activePod) {\n\t\t\t\t\tconsole.log(\"\"); // Less verbose for incompatible models when filtered\n\t\t\t\t} else {\n\t\t\t\t\tconsole.log(\"\");\n\t\t\t\t}\n\t\t\t}\n\t\t}\n\t}\n\n\tconsole.log(chalk.gray(\"\\nFor unknown models, defaults to single GPU deployment.\"));\n\tconsole.log(chalk.gray(\"Use --vllm to pass custom arguments to vLLM.\"));\n};\n"]}