[Unit] # REVERSE tunnel — runs on the ON-PREM GPU box (office), NOT on Vultr. # # Why reversed: the old clone-tunnel.service had Vultr SSH *into* the GPU box, which worked when # the GPU was an EC2 instance with a public IP. The on-prem box sits behind office NAT with no # public IP and no port-forward, so Vultr cannot dial in. Instead this box dials OUT to Vultr and # publishes its local vLLM back through the tunnel. # # Net effect is identical for the bot: Vultr's 127.0.0.1:8000 still reaches vLLM, so LLM_BASE_URL # stays http://127.0.0.1:8000/v1 and nothing in the app changes. # # -R 127.0.0.1:8000:127.0.0.1:8000 = "publish MY localhost:8000 as VULTR's localhost:8000" # # Bound to 127.0.0.1 on the Vultr side, so the model is never exposed publicly (no GatewayPorts). Description=autossh reverse tunnel — publish on-prem vLLM 127.0.0.1:8000 to Vultr's localhost:8000 After=network-online.target vllm-coder.service Wants=network-online.target [Service] User=ai-admin Environment=AUTOSSH_GATETIME=0 # Replace VULTR_IP with the Vultr box's IP, and ensure this box's ai-admin SSH key is authorized # on Vultr (ssh-keygen here, then append the pubkey to Vultr's ~clonebot/.ssh/authorized_keys). # ExitOnForwardFailure=yes -> fail fast if the remote port is stale rather than sit connected # but not forwarding (that failure mode is what produced the silent 13-minute hangs before). # ServerAlive* + autossh -> reconnect across office ISP drops / dynamic IP changes. ExecStart=/usr/bin/autossh -M 0 -N \ -o "ServerAliveInterval=30" -o "ServerAliveCountMax=3" \ -o "ExitOnForwardFailure=yes" -o "StrictHostKeyChecking=accept-new" \ -R 127.0.0.1:8000:127.0.0.1:8000 clonebot@VULTR_IP Restart=always RestartSec=10 [Install] WantedBy=multi-user.target