From bf3a1ba523a13fa5e6e2560358930341ead7f50e Mon Sep 17 00:00:00 2001 From: root Date: Mon, 28 Sep 2026 07:12:57 +0000 Subject: [PATCH] fix(agent-health): use llmuser for .8 GPU health probe instead of root The rebuilt VM 101 (192.168.68.8) kept only one SSH key in root's authorized_keys, so the health check's root probe returns Permission denied and misreports the healthy host as UNREACHABLE. The llama-server runs as llmuser, so that user can see the :8080 pid via ss -tlnp. Add per-host user to GPU_HOSTS (default root, llmuser for .8) and pass it through check_gpu_ports() into all ssh() calls. Closes the gpu-unreachable:192.168.68.8 leg while leaving the root SSH security decision for the captain. --- scripts/agent-health-check.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index d59bca3..02c98b6 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -95,7 +95,7 @@ AGENTS = { # .110 rtx5070 (ocu-llm VM) -> llama-server.service (active) # .15 strixhalo (amdpve) -> strix-server.service (active) GPU_HOSTS = { - "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"}, + "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service", "user": "llmuser"}, "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"}, "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"}, } @@ -300,13 +300,14 @@ def check_gpu_ports(): host = gpu["host"] port = gpu["port"] svc = gpu["service"] + user = gpu.get("user", "root") # default root, overridden per-host where needed # `systemctl is-active` exits non-zero when the unit is inactive or # missing, which the ssh() helper would swallow as an SSH failure and # report as UNREACHABLE. `|| true` keeps the real state word so we can # tell "unit inactive" from "host unreachable". - svc_status = ssh(host, f"systemctl is-active {svc} || true") - port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1") + svc_status = ssh(host, f"systemctl is-active {svc} || true", user=user) + port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1", user=user) if not svc_status: print(f" ❌ {label}: UNREACHABLE") @@ -317,14 +318,14 @@ def check_gpu_ports(): print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})") FAIL.append(f"gpu-no-port:{label}") elif svc_status != "active": - svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2") + svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2", user=user) if svc_pid and port_owner != svc_pid: print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})") FAIL.append(f"gpu-ghost:{label}:{port_owner}") else: print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}") else: - health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health") + health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health", user=user) if health and '"status":"ok"' in health: print(f" ✅ {label}: healthy (pid={port_owner})") elif health and '"status":"no slot available"' in health: