#!/usr/bin/env bash
# h3-pod-up.sh — provision the H3 lane pod with the capacity-fallback protocol.
#
#   RTX PRO 6000 · runpod/comfyui:cuda13.0 (official cu130 build, verified
#   layout; h3-install.sh builds SageAttention2) · 80GB container disk ·
#   no network volume · ports 22+8188.
#
# Tries COMMUNITY across the DC rotation, then SECURE; shrinks the container
# disk once when a machine reports "does not have the resources". Failed
# creates don't bill. Prints the pod id on success and waits for SSH.
#
# Usage: h3-pod-up.sh [--secure-only] [--disk GB] [--name NAME]
set -euo pipefail

IMAGE="runpod/comfyui:cuda13.0"
GPU_ID="NVIDIA RTX PRO 6000 Blackwell Server Edition"
DISK=80
NAME="lvrged-h3"
CLOUDS=(COMMUNITY SECURE)
DCS=(US-NC-1 US-NC-2 US-PA-1 US-MO-2 US-KS-2 US-NE-1 CA-MTL-3 EU-CZ-1 EU-NL-1 EU-RO-1 EUR-IS-1 EUR-IS-2)

while [[ $# -gt 0 ]]; do
  case "$1" in
    --secure-only) CLOUDS=(SECURE); shift ;;
    --disk) DISK="$2"; shift 2 ;;
    --name) NAME="$2"; shift 2 ;;
    *) echo "unknown arg: $1" >&2; exit 2 ;;
  esac
done

create() { # cloud dc disk → pod id on stdout, or returns 1
  local out
  out=$(runpodctl pod create --name "$NAME" --image "$IMAGE" \
    --gpu-id "$GPU_ID" --cloud-type "$1" --data-center-ids "$2" \
    --container-disk-in-gb "$3" --volume-in-gb 0 \
    --ports "22/tcp,8188/http" 2>&1) || true
  if echo "$out" | grep -q '"id"'; then
    echo "$out" | grep -o '"id": *"[^"]*"' | head -1 | sed 's/.*"id": *"//;s/"//'
    return 0
  fi
  LAST_ERR="$out"
  return 1
}

POD=""
for cloud in "${CLOUDS[@]}"; do
  disk=$DISK; shrunk=0
  for dc in "${DCS[@]}"; do
    echo "→ $cloud $dc disk=${disk}GB" >&2
    if POD=$(create "$cloud" "$dc" "$disk"); then break 2; fi
    if echo "$LAST_ERR" | grep -qi "does not have the resources" && [[ $shrunk -eq 0 && $disk -gt 60 ]]; then
      disk=60; shrunk=1
      echo "→ $cloud $dc disk=60GB (shrunk — machine pool exists but spec didn't fit)" >&2
      if POD=$(create "$cloud" "$dc" "$disk"); then break 2; fi
    fi
    if ! echo "$LAST_ERR" | grep -qiE "no longer any instances available|does not have the resources"; then
      echo "unexpected create error (not a capacity issue) — stopping:" >&2
      echo "$LAST_ERR" | head -3 >&2
      exit 1
    fi
    sleep 2
  done
done

if [[ -z "$POD" ]]; then
  echo "capacity protocol exhausted — nothing created, nothing billed." >&2
  echo "wait 10-30 min and retry, or check stock: runpodctl gpu list" >&2
  exit 1
fi

echo "pod created: $POD (billing started)" >&2
echo "waiting for SSH (image pull takes minutes; 'refused' while extracting is normal)..." >&2
for i in $(seq 1 40); do
  if runpodctl pod get "$POD" | grep -q '"ssh_command"'; then
    runpodctl pod get "$POD" | grep -o '"ssh_command": *"[^"]*"' | sed 's/.*: *"//;s/"//' >&2
    echo "$POD"
    exit 0
  fi
  sleep 15
done
echo "SSH never came up — check: curl -s \"https://api.runpod.io/v2/pods/$POD/logs?stream=false\" -H \"Authorization: Bearer \$RUNPOD_API_KEY\"" >&2
echo "$POD"
exit 1
