Merge PR #143: fix(agent-health): use llmuser for .8 GPU health probe instead of root
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 17s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 14s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 22s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 11s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 17s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 14s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 22s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 11s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 3s
This commit is contained in:
@@ -24,7 +24,7 @@ Runs every 4 hours (2, 6, 10, 14, 18, 22 UTC at :35) via cron (`35 2,6,10,14,18,
|
|||||||
## Requires
|
## Requires
|
||||||
|
|
||||||
- **LiteLLM admin key** for key validation (retrieved from `/root/.pi/agent/env.sh`)
|
- **LiteLLM admin key** for key validation (retrieved from `/root/.pi/agent/env.sh`)
|
||||||
- **SSH access** to GPU hosts (.8, .110, .15) and agent CTs (.122, .129, .114, .24)
|
- **SSH access** to GPU hosts — `llmuser` on .8 (owns `llama-server`), `root` on .110 and .15 — and agent CTs (.122, .129, .114, .24)
|
||||||
- **Python 3** for script execution
|
- **Python 3** for script execution
|
||||||
- **Network access** to LiteLLM (:4000), GPU exporters (:9400), and gateway endpoints
|
- **Network access** to LiteLLM (:4000), GPU exporters (:9400), and gateway endpoints
|
||||||
|
|
||||||
|
|||||||
@@ -50,6 +50,11 @@ Changelog:
|
|||||||
(kagentz CT 105 on minipve, .14, dedicated `hermes` user) and is monitored
|
(kagentz CT 105 on minipve, .14, dedicated `hermes` user) and is monitored
|
||||||
from her side. This script must not probe mumuni or .24 — the v2 changelog
|
from her side. This script must not probe mumuni or .24 — the v2 changelog
|
||||||
roster line was the last reference still placing her at .24 / CT100.
|
roster line was the last reference still placing her at .24 / CT100.
|
||||||
|
v6 (2026-09-28): .8 GPU health probe now runs as `llmuser` instead of `root`.
|
||||||
|
Root SSH to .8 was lost when the guest was rebuilt, so every .8 leg read as
|
||||||
|
UNREACHABLE for a healthy host. llmuser owns llama-server and can read
|
||||||
|
`systemctl is-active`, `systemctl show -p MainPID`, and the :8080 pid.
|
||||||
|
.110 and .15 keep the default `root` user.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import subprocess, json, sys, os, time, re, io, contextlib
|
import subprocess, json, sys, os, time, re, io, contextlib
|
||||||
@@ -95,7 +100,7 @@ AGENTS = {
|
|||||||
# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active)
|
# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active)
|
||||||
# .15 strixhalo (amdpve) -> strix-server.service (active)
|
# .15 strixhalo (amdpve) -> strix-server.service (active)
|
||||||
GPU_HOSTS = {
|
GPU_HOSTS = {
|
||||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"},
|
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service", "user": "llmuser"},
|
||||||
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"},
|
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"},
|
||||||
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"},
|
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"},
|
||||||
}
|
}
|
||||||
@@ -300,13 +305,14 @@ def check_gpu_ports():
|
|||||||
host = gpu["host"]
|
host = gpu["host"]
|
||||||
port = gpu["port"]
|
port = gpu["port"]
|
||||||
svc = gpu["service"]
|
svc = gpu["service"]
|
||||||
|
user = gpu.get("user", "root") # default root, overridden per-host where needed
|
||||||
|
|
||||||
# `systemctl is-active` exits non-zero when the unit is inactive or
|
# `systemctl is-active` exits non-zero when the unit is inactive or
|
||||||
# missing, which the ssh() helper would swallow as an SSH failure and
|
# missing, which the ssh() helper would swallow as an SSH failure and
|
||||||
# report as UNREACHABLE. `|| true` keeps the real state word so we can
|
# report as UNREACHABLE. `|| true` keeps the real state word so we can
|
||||||
# tell "unit inactive" from "host unreachable".
|
# tell "unit inactive" from "host unreachable".
|
||||||
svc_status = ssh(host, f"systemctl is-active {svc} || true")
|
svc_status = ssh(host, f"systemctl is-active {svc} || true", user=user)
|
||||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1")
|
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1", user=user)
|
||||||
|
|
||||||
if not svc_status:
|
if not svc_status:
|
||||||
print(f" ❌ {label}: UNREACHABLE")
|
print(f" ❌ {label}: UNREACHABLE")
|
||||||
@@ -317,14 +323,14 @@ def check_gpu_ports():
|
|||||||
print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})")
|
print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})")
|
||||||
FAIL.append(f"gpu-no-port:{label}")
|
FAIL.append(f"gpu-no-port:{label}")
|
||||||
elif svc_status != "active":
|
elif svc_status != "active":
|
||||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2")
|
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2", user=user)
|
||||||
if svc_pid and port_owner != svc_pid:
|
if svc_pid and port_owner != svc_pid:
|
||||||
print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})")
|
print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})")
|
||||||
FAIL.append(f"gpu-ghost:{label}:{port_owner}")
|
FAIL.append(f"gpu-ghost:{label}:{port_owner}")
|
||||||
else:
|
else:
|
||||||
print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}")
|
print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}")
|
||||||
else:
|
else:
|
||||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health")
|
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health", user=user)
|
||||||
if health and '"status":"ok"' in health:
|
if health and '"status":"ok"' in health:
|
||||||
print(f" ✅ {label}: healthy (pid={port_owner})")
|
print(f" ✅ {label}: healthy (pid={port_owner})")
|
||||||
elif health and '"status":"no slot available"' in health:
|
elif health and '"status":"no slot available"' in health:
|
||||||
|
|||||||
@@ -124,6 +124,39 @@ def test_tanko_ct112_is_probed_on_minipve(ahc, monkeypatch, capsys):
|
|||||||
ahc.REPORT_ONLY.clear()
|
ahc.REPORT_ONLY.clear()
|
||||||
|
|
||||||
|
|
||||||
|
def test_gpu_rtx3090_probe_uses_llmuser_not_root(ahc, monkeypatch, capsys):
|
||||||
|
# 2026-09-28: root SSH to .8 was lost when the guest was rebuilt; llmuser
|
||||||
|
# owns llama-server and can read systemctl status and the :8080 pid. A root
|
||||||
|
# probe reads as UNREACHABLE for a healthy host (the reported bug). Execute
|
||||||
|
# check_gpu_ports() against an SSH boundary that only accepts llmuser@.8 and
|
||||||
|
# assert the .8 leg does not produce the false UNREACHABLE failure.
|
||||||
|
seen = []
|
||||||
|
|
||||||
|
def fake_ssh(host, cmd, user="root"):
|
||||||
|
seen.append((host, user))
|
||||||
|
if host == "192.168.68.8" and user != "llmuser":
|
||||||
|
return None # root SSH denied -> baseline false UNREACHABLE
|
||||||
|
if cmd.startswith("systemctl is-active"):
|
||||||
|
return "active"
|
||||||
|
if cmd.startswith("ss -tlnp"):
|
||||||
|
return "48351"
|
||||||
|
if cmd.startswith("curl"):
|
||||||
|
return '{"status":"ok"}'
|
||||||
|
return None
|
||||||
|
|
||||||
|
monkeypatch.setattr(ahc, "ssh", fake_ssh)
|
||||||
|
ahc.FAIL.clear()
|
||||||
|
try:
|
||||||
|
ahc.check_gpu_ports()
|
||||||
|
out = capsys.readouterr().out
|
||||||
|
assert "gpu-unreachable:192.168.68.8" not in ahc.FAIL
|
||||||
|
assert "\u2705 gpu-rtx3090 (.8): healthy" in out
|
||||||
|
assert ("192.168.68.8", "llmuser") in seen
|
||||||
|
assert not any(host == "192.168.68.8" and user == "root" for host, user in seen)
|
||||||
|
finally:
|
||||||
|
ahc.FAIL.clear()
|
||||||
|
|
||||||
|
|
||||||
def test_report_only_legs_never_count_as_failures(ahc):
|
def test_report_only_legs_never_count_as_failures(ahc):
|
||||||
for agent, report_only in (("koby", True), ("koonimo", False), ("tanko", False)):
|
for agent, report_only in (("koby", True), ("koonimo", False), ("tanko", False)):
|
||||||
ahc.FAIL.clear()
|
ahc.FAIL.clear()
|
||||||
|
|||||||
Reference in New Issue
Block a user