Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e0c9852de8 | ||
|
|
bf3a1ba523 |
@@ -24,7 +24,7 @@ Runs every 4 hours (2, 6, 10, 14, 18, 22 UTC at :35) via cron (`35 2,6,10,14,18,
|
||||
## Requires
|
||||
|
||||
- **LiteLLM admin key** for key validation (retrieved from `/root/.pi/agent/env.sh`)
|
||||
- **SSH access** to GPU hosts (.8, .110, .15) and agent CTs (.122, .129, .114, .24)
|
||||
- **SSH access** to GPU hosts — `llmuser` on .8 (owns `llama-server`), `root` on .110 and .15 — and agent CTs (.122, .129, .114, .24)
|
||||
- **Python 3** for script execution
|
||||
- **Network access** to LiteLLM (:4000), GPU exporters (:9400), and gateway endpoints
|
||||
|
||||
|
||||
@@ -50,6 +50,11 @@ Changelog:
|
||||
(kagentz CT 105 on minipve, .14, dedicated `hermes` user) and is monitored
|
||||
from her side. This script must not probe mumuni or .24 — the v2 changelog
|
||||
roster line was the last reference still placing her at .24 / CT100.
|
||||
v6 (2026-09-28): .8 GPU health probe now runs as `llmuser` instead of `root`.
|
||||
Root SSH to .8 was lost when the guest was rebuilt, so every .8 leg read as
|
||||
UNREACHABLE for a healthy host. llmuser owns llama-server and can read
|
||||
`systemctl is-active`, `systemctl show -p MainPID`, and the :8080 pid.
|
||||
.110 and .15 keep the default `root` user.
|
||||
"""
|
||||
|
||||
import subprocess, json, sys, os, time, re, io, contextlib
|
||||
@@ -95,7 +100,7 @@ AGENTS = {
|
||||
# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active)
|
||||
# .15 strixhalo (amdpve) -> strix-server.service (active)
|
||||
GPU_HOSTS = {
|
||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"},
|
||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service", "user": "llmuser"},
|
||||
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"},
|
||||
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"},
|
||||
}
|
||||
@@ -300,13 +305,14 @@ def check_gpu_ports():
|
||||
host = gpu["host"]
|
||||
port = gpu["port"]
|
||||
svc = gpu["service"]
|
||||
user = gpu.get("user", "root") # default root, overridden per-host where needed
|
||||
|
||||
# `systemctl is-active` exits non-zero when the unit is inactive or
|
||||
# missing, which the ssh() helper would swallow as an SSH failure and
|
||||
# report as UNREACHABLE. `|| true` keeps the real state word so we can
|
||||
# tell "unit inactive" from "host unreachable".
|
||||
svc_status = ssh(host, f"systemctl is-active {svc} || true")
|
||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1")
|
||||
svc_status = ssh(host, f"systemctl is-active {svc} || true", user=user)
|
||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1", user=user)
|
||||
|
||||
if not svc_status:
|
||||
print(f" ❌ {label}: UNREACHABLE")
|
||||
@@ -317,14 +323,14 @@ def check_gpu_ports():
|
||||
print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})")
|
||||
FAIL.append(f"gpu-no-port:{label}")
|
||||
elif svc_status != "active":
|
||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2")
|
||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2", user=user)
|
||||
if svc_pid and port_owner != svc_pid:
|
||||
print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})")
|
||||
FAIL.append(f"gpu-ghost:{label}:{port_owner}")
|
||||
else:
|
||||
print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}")
|
||||
else:
|
||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health")
|
||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health", user=user)
|
||||
if health and '"status":"ok"' in health:
|
||||
print(f" ✅ {label}: healthy (pid={port_owner})")
|
||||
elif health and '"status":"no slot available"' in health:
|
||||
|
||||
@@ -124,6 +124,39 @@ def test_tanko_ct112_is_probed_on_minipve(ahc, monkeypatch, capsys):
|
||||
ahc.REPORT_ONLY.clear()
|
||||
|
||||
|
||||
def test_gpu_rtx3090_probe_uses_llmuser_not_root(ahc, monkeypatch, capsys):
|
||||
# 2026-09-28: root SSH to .8 was lost when the guest was rebuilt; llmuser
|
||||
# owns llama-server and can read systemctl status and the :8080 pid. A root
|
||||
# probe reads as UNREACHABLE for a healthy host (the reported bug). Execute
|
||||
# check_gpu_ports() against an SSH boundary that only accepts llmuser@.8 and
|
||||
# assert the .8 leg does not produce the false UNREACHABLE failure.
|
||||
seen = []
|
||||
|
||||
def fake_ssh(host, cmd, user="root"):
|
||||
seen.append((host, user))
|
||||
if host == "192.168.68.8" and user != "llmuser":
|
||||
return None # root SSH denied -> baseline false UNREACHABLE
|
||||
if cmd.startswith("systemctl is-active"):
|
||||
return "active"
|
||||
if cmd.startswith("ss -tlnp"):
|
||||
return "48351"
|
||||
if cmd.startswith("curl"):
|
||||
return '{"status":"ok"}'
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(ahc, "ssh", fake_ssh)
|
||||
ahc.FAIL.clear()
|
||||
try:
|
||||
ahc.check_gpu_ports()
|
||||
out = capsys.readouterr().out
|
||||
assert "gpu-unreachable:192.168.68.8" not in ahc.FAIL
|
||||
assert "\u2705 gpu-rtx3090 (.8): healthy" in out
|
||||
assert ("192.168.68.8", "llmuser") in seen
|
||||
assert not any(host == "192.168.68.8" and user == "root" for host, user in seen)
|
||||
finally:
|
||||
ahc.FAIL.clear()
|
||||
|
||||
|
||||
def test_report_only_legs_never_count_as_failures(ahc):
|
||||
for agent, report_only in (("koby", True), ("koonimo", False), ("tanko", False)):
|
||||
ahc.FAIL.clear()
|
||||
|
||||
Reference in New Issue
Block a user