#!/usr/bin/env bash
# h3-install.sh — install the full H3 Turbo stack on a pod, then apply it the
# sanctioned way: comfyui_args.txt + pod restart (never hand-rolled daemons).
#
#   1. clone larryvrh/ComfyUI-MiniMax-H3-Turbo + xmarre/ComfyUI-Spectrum-MiniMax-H3
#   2. hf download the 5 weight files (~41GB, public repos, hf-transfer on)
#   3. append --use-sage-attention to the template's comfyui_args.txt
#   4. runpodctl pod restart, wait for boot (60-90s with 40GB of models), and
#      verify: sage active in the boot log + Turbo nodes in /object_info
#
# Usage: h3-install.sh <pod_id>
set -euo pipefail
POD="${1:?usage: h3-install.sh <pod_id>}"
KEY="$HOME/.runpod/ssh/runpodctl-ssh-key"

sshinfo=$(runpodctl pod get "$POD")
IP=$(echo "$sshinfo" | grep -o '"ip": *"[^"]*"' | head -1 | sed 's/.*: *"//;s/"//')
PORT=$(echo "$sshinfo" | grep -o '"port": *[0-9]*' | head -1 | grep -o '[0-9]*')
[[ -n "$IP" && -n "$PORT" ]] || { echo "pod $POD has no SSH info yet (still pulling image?)" >&2; exit 1; }
SSH=(ssh -i "$KEY" -o StrictHostKeyChecking=no -o ConnectTimeout=20 -p "$PORT" "root@$IP")

echo "→ locating ComfyUI layout on the pod" >&2
LAYOUT=$("${SSH[@]}" 'find /workspace /opt -maxdepth 4 -name comfyui_version.py 2>/dev/null | head -1')
[[ -n "$LAYOUT" ]] || { echo "no ComfyUI found on pod — wrong image?" >&2; exit 1; }
CDIR=$(dirname "$LAYOUT")
ARGS_FILE=$("${SSH[@]}" "ls $(dirname "$CDIR")/comfyui_args.txt 2>/dev/null || find /workspace -maxdepth 3 -name comfyui_args.txt 2>/dev/null | head -1")
echo "  ComfyUI: $CDIR · args: ${ARGS_FILE:-NOT FOUND}" >&2

echo "→ cloning H3 nodes + downloading weights (~41GB; hf-transfer)" >&2
"${SSH[@]}" "set -e
cd '$CDIR/custom_nodes'
[ -d ComfyUI-MiniMax-H3-Turbo ] || git clone -q https://github.com/larryvrh/ComfyUI-MiniMax-H3-Turbo
[ -d ComfyUI-Spectrum-MiniMax-H3 ] || git clone -q https://github.com/xmarre/ComfyUI-Spectrum-MiniMax-H3
M='$CDIR/models'
mkdir -p \"\$M/diffusion_models\" \"\$M/text_encoders\" \"\$M/vae\" \"\$M/loras\"
export HF_HUB_ENABLE_HF_TRANSFER=1
hf download Comfy-Org/MiniMax-H3 diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors --local-dir \"\$M\"
hf download Comfy-Org/MiniMax-H3 text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors --local-dir \"\$M\"
hf download Comfy-Org/MiniMax-H3 vae/minimax_h3_audio_vae_fp32.safetensors --local-dir \"\$M\"
hf download Comfy-Org/MiniMax-H3 vae/minimax_h3_video_vae_fp16.safetensors --local-dir \"\$M\"
hf download larryvrh/MiniMax-H3-Turbo-Lora minimax_h3_turbo_v4_step600_ema.safetensors --local-dir \"\$M/loras\"
du -sh \"\$M\"/diffusion_models \"\$M\"/text_encoders \"\$M\"/vae \"\$M\"/loras"

echo "→ ensuring real SageAttention2 in the venv (the official cu13 image ships NONE;" >&2
echo "  'pip install sageattention' would install a useless 20kB stub — building from source instead)" >&2
SAGE_OK=0
if "${SSH[@]}" "set -e
  PY=\$(ls -d '$CDIR'/.venv-*/bin/python 2>/dev/null | head -1)
  [ -n \"\$PY\" ] || PY=/usr/bin/python3
  # real Sage2 = compiled kernels: require a .so in the package, not just an import
  if \$PY - <<'CHECK'
import glob, os, sys
try:
    import sageattention
except ImportError:
    sys.exit(1)
d = os.path.dirname(sageattention.__file__)
sys.exit(0 if glob.glob(os.path.join(d, '**', '*.so'), recursive=True) else 1)
CHECK
  then echo SAGE_PRESENT; exit 0; fi
  export PATH=/usr/local/cuda/bin:\$PATH
  command -v nvcc >/dev/null || { echo NO_NVCC; exit 3; }
  rm -rf /tmp/SageAttention && git clone -q https://github.com/thu-ml/SageAttention /tmp/SageAttention
  cd /tmp/SageAttention && \${PY%python}pip install -q . && echo SAGE_BUILT" 2>&1 | tail -2 | grep -qE "SAGE_PRESENT|SAGE_BUILT"; then
  SAGE_OK=1
else
  echo "!! SageAttention build failed — continuing WITHOUT it. Every timing from this pod is" >&2
  echo "!! stock-attention; label it so. Fallback image with prebuilt Sage (unverified layout):" >&2
  echo "!! hearmeman/comfyui-wan-template:v25-cuda13" >&2
fi

if [[ -n "$ARGS_FILE" && "$SAGE_OK" -eq 1 ]]; then
  echo "→ enabling --use-sage-attention via $ARGS_FILE (does nothing without the flag)" >&2
  "${SSH[@]}" "grep -q -- --use-sage-attention '$ARGS_FILE' || printf -- '--use-sage-attention\n' >> '$ARGS_FILE'"
elif [[ -z "$ARGS_FILE" ]]; then
  echo "!! no comfyui_args.txt found — add --use-sage-attention to this image's supervisor config by hand" >&2
fi

echo "→ restarting pod so the supervisor relaunches with new nodes + args" >&2
runpodctl pod restart "$POD" >/dev/null

echo "→ waiting for ComfyUI (60-90s boot with 40GB of models)" >&2
sleep 20
for i in $(seq 1 40); do
  # ssh port can change on restart — re-read
  sshinfo=$(runpodctl pod get "$POD"); IP2=$(echo "$sshinfo" | grep -o '"ip": *"[^"]*"' | head -1 | sed 's/.*: *"//;s/"//'); PORT2=$(echo "$sshinfo" | grep -o '"port": *[0-9]*' | head -1 | grep -o '[0-9]*')
  if [[ -n "$IP2" && -n "$PORT2" ]] && ssh -i "$KEY" -o StrictHostKeyChecking=no -o ConnectTimeout=8 -o BatchMode=yes -p "$PORT2" "root@$IP2" \
      'curl -s -m 3 http://127.0.0.1:8188/system_stats >/dev/null' 2>/dev/null; then
    SSH=(ssh -i "$KEY" -o StrictHostKeyChecking=no -p "$PORT2" "root@$IP2")
    echo "→ verifying: sage in boot log + Turbo nodes registered" >&2
    "${SSH[@]}" 'LOG=$(ls /workspace/*/comfyui.log /workspace/comfyui.log 2>/dev/null | head -1)
      [ -n "$LOG" ] && grep -i sage "$LOG" | head -2 || echo "(no comfyui.log found — check supervisor log path)"
      curl -s -m 20 http://127.0.0.1:8188/object_info | python3 -c "import json,sys; d=json.load(sys.stdin); print(\"TurboSampler:\", \"MiniMaxH3TurboSampler\" in d, \"· Spectrum:\", \"SpectrumApplyMiniMaxH3\" in d)"'
    echo "install done. tunnel next: ssh -i $KEY -f -N -L 8188:127.0.0.1:8188 root@$IP2 -p $PORT2" >&2
    exit 0
  fi
  sleep 10
done
echo "ComfyUI never answered — read the boot log on the pod" >&2
exit 1
