#!/usr/bin/env bash
# ON-PREM GPU box (office) = inference appliance ONLY. Serves Qwen3-Coder-30B-A3B on 127.0.0.1:8000
# and dials a REVERSE tunnel out to Vultr (it's behind office NAT — Vultr cannot dial in).
# The Slack bot + OpenHands live on the Vultr box — see vultr-setup.sh.
#
# Target: Ubuntu 24.04, AMD Ryzen 7 7700, 64GB RAM, RTX 5090 (Blackwell sm_120, 32GB), 2TB NVMe.
# Run:  sudo bash gpu-setup.sh
set -euo pipefail

RUN_USER="${RUN_USER:-ai-admin}"

echo "=== 0. Verify GPU (expect: RTX 5090 / 32GB / compute_cap 12.0) ==="
nvidia-smi --query-gpu=name,memory.total,driver_version,compute_cap --format=csv || {
  echo "!! No GPU / driver. Install the 570+ driver first (Blackwell needs it)."; exit 1; }

echo "=== 1. Disk — the Ubuntu installer only allocates ~100GB of the 2TB SSD to the LV ==="
# Model weights (~17GB) + HF cache fill 68GB fast. Extend the root LV over the whole VG.
df -h /
if vgs --noheadings -o vg_free --units g 2>/dev/null | grep -qvE '^\s*0(\.0+)?g\s*$'; then
  echo "    Free space found in the volume group — extending root LV to 100%..."
  LV_PATH="$(findmnt -no SOURCE /)"
  lvextend -l +100%FREE "$LV_PATH" || echo "    (already extended, or extend failed — check manually)"
  resize2fs "$LV_PATH" || echo "    (resize2fs skipped — xfs? use xfs_growfs /)"
  df -h /
else
  echo "    No unallocated space in the VG — nothing to extend."
fi

echo "=== 2. System packages ==="
apt-get update
apt-get install -y python3 python3-venv python3-pip git curl wget ninja-build autossh

echo "=== 3. CUDA toolkit 13.0 (nvcc) — vLLM JIT-compiles kernels at startup ==="
# The driver alone has no toolkit. Install the TOOLKIT only (never the 'cuda' meta package — it
# would replace your working 580 driver). MUST be >= 12.9 for Blackwell sm_120: a 12.8 toolkit
# makes kernel JIT crawl (~20 min model load) and mis-detects the card. 13.0 matches the cu130 torch.
if ! /usr/local/cuda/bin/nvcc --version 2>/dev/null | grep -qE 'release (12\.9|13\.)'; then
  cd /tmp
  wget -q https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64/cuda-keyring_1.1-1_all.deb
  dpkg -i cuda-keyring_1.1-1_all.deb
  apt-get update
  apt-get install -y cuda-toolkit-13-0
  ln -sfn /usr/local/cuda-13.0 /usr/local/cuda
fi
/usr/local/cuda/bin/nvcc --version || echo "!! nvcc not found — check the toolkit install"

echo "=== 4. vLLM in its own venv (/opt/vllm) — MUST be a current build for Blackwell ==="
# sm_120 is NOT supported by older vLLM/PyTorch. Do NOT pin an old version here: the previous
# 'vllm>=0.6.0' pin (from the L40S/sm_89 box) will not run on this card. Install latest.
mkdir -p /opt/vllm/hf
python3 -m venv /opt/vllm/venv
source /opt/vllm/venv/bin/activate     # activate before pip
pip install --upgrade pip
pip install --upgrade vllm             # latest — brings a CUDA 12.8+ torch
pip install ninja                      # FlashInfer's nvcc JIT build needs it
echo "--- verifying PyTorch sees Blackwell (expect capability (12, 0)) ---"
python - <<'PY'
import torch
print("torch:", torch.__version__, "| cuda:", torch.version.cuda)
print("device:", torch.cuda.get_device_name(0))
cap = torch.cuda.get_device_capability(0)
print("capability:", cap)
assert cap[0] >= 12, "!! PyTorch does not support sm_120 — need a cu128+ build of torch/vLLM"
PY
deactivate
chown -R "$RUN_USER:$RUN_USER" /opt/vllm

echo "=== 5. SSH key for the REVERSE tunnel (authorize this on the VULTR box) ==="
sudo -u "$RUN_USER" ssh-keygen -t ed25519 -N "" -f "/home/$RUN_USER/.ssh/id_ed25519" <<<y >/dev/null 2>&1 || true
echo "    >>> Copy this pubkey into VULTR's ~clonebot/.ssh/authorized_keys:"
cat "/home/$RUN_USER/.ssh/id_ed25519.pub"

echo "=== 6. Install services ==="
echo "    Set the Vultr IP:  sed -i 's/VULTR_IP/<vultr-ip>/' infra/onprem-tunnel.service"
cp "$(dirname "$0")/vllm-coder.service"   /etc/systemd/system/
cp "$(dirname "$0")/onprem-tunnel.service" /etc/systemd/system/
systemctl daemon-reload
systemctl enable vllm-coder onprem-tunnel
systemctl start vllm-coder
echo "    First boot downloads ~17GB. Watch: journalctl -u vllm-coder -f"

echo "=== 7. Wait for the endpoint ==="
for i in $(seq 1 120); do
  curl -fsS http://127.0.0.1:8000/v1/models >/dev/null 2>&1 && { echo "    vLLM up."; break; }
  sleep 15
done
curl -fsS http://127.0.0.1:8000/v1/models || echo "    (not up yet — keep watching the journal)"

cat <<'EOF'

=== On-prem GPU box done ===
  vLLM is bound to 127.0.0.1:8000 — do NOT port-forward it on the office router.
  Next:
    1. Put the Vultr IP into /etc/systemd/system/onprem-tunnel.service, then: systemctl daemon-reload
    2. Authorize the printed pubkey on Vultr (~clonebot/.ssh/authorized_keys)
    3. systemctl start onprem-tunnel
    4. From VULTR, confirm the reverse tunnel:
         curl -fsS http://127.0.0.1:8000/v1/models     (should list 'devstral' / 'qwen3-coder')
EOF
