Single non-disruptive script replacing 7 scattered Zulip health checks. Checks every 10 min via cron, never restarts anything: - LiteLLM key validation for all 4 Hermes agents - GPU port conflict / ghost process detection - Agent gateway liveness + Zulip streaming status - Recent gateway error count Port conflict detection added to all 3 GPU wrappers: - .8 (qwen): wrapper detects ghost on port 8080 before starting - .110 (gemma): same pattern - .15 (ornith): port-cleanup.sh replaces blanket pkill -x llama-server
229 lines
9.7 KiB
Python
Executable File
229 lines
9.7 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
/root/scripts/agent-health-check.py — Consolidated Agent Health Verification
|
|
|
|
Single non-disruptive health check replacing 7 scattered scripts.
|
|
Verifies: LiteLLM keys, GPU port conflicts, agent Zulip streaming,
|
|
gateway liveness, and gateway log health. NEVER restarts anything.
|
|
|
|
Usage:
|
|
python3 /root/scripts/agent-health-check.py # Full check
|
|
python3 /root/scripts/agent-health-check.py --json # Machine-readable
|
|
python3 /root/scripts/agent-health-check.py --quiet # Only output on failure
|
|
|
|
Cron: */10 * * * * python3 /root/scripts/agent-health-check.py --quiet
|
|
"""
|
|
|
|
import subprocess, json, sys, os, time
|
|
from datetime import datetime
|
|
|
|
LITELLM = "http://192.168.68.116:80"
|
|
|
|
AGENTS = {
|
|
"tanko": {"ct": 112, "host": "192.168.68.122", "key": "sk-CggiHWlamQyShxWC3Hx6uw", "user": "jerome"},
|
|
"mumuni": {"ct": 114, "host": "192.168.68.123", "key": "sk-VrqCNlwUgzoNGOpikJ7nwQ", "user": "root"},
|
|
"tdunna": {"ct": 111, "host": None, "key": "sk-6sbCNjz2T6lTVDBdlNHXsA", "user": None},
|
|
"baggy": {"ct": 113, "host": None, "key": "sk-krnw_zGBwvvL5b7l2t-s-A", "user": None},
|
|
}
|
|
|
|
GPU_HOSTS = {
|
|
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-server"},
|
|
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server"},
|
|
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "ornith-server"},
|
|
}
|
|
|
|
FAIL = []
|
|
|
|
def ssh(host, cmd, user="root"):
|
|
"""Execute a command on a remote host, return stdout or None."""
|
|
try:
|
|
result = subprocess.run(
|
|
["ssh", "-o", "StrictHostKeyChecking=no", "-o", "ConnectTimeout=8",
|
|
f"{user}@{host}", cmd],
|
|
capture_output=True, text=True, timeout=15
|
|
)
|
|
return result.stdout.strip() if result.returncode == 0 else None
|
|
except:
|
|
return None
|
|
|
|
def http_get(url, headers=None, timeout=5):
|
|
"""Return HTTP status code as string."""
|
|
try:
|
|
cmd = ["curl", "-sfk", "--connect-timeout", str(timeout), "-o", "/dev/null", "-w", "%{http_code}"]
|
|
if headers:
|
|
for k, v in headers.items():
|
|
cmd.extend(["-H", f"{k}: {v}"])
|
|
cmd.append(url)
|
|
result = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout+3)
|
|
return result.stdout.strip() or "000"
|
|
except:
|
|
return "timeout"
|
|
|
|
def http_json(url, headers=None, timeout=5):
|
|
"""Return parsed JSON from URL, or None."""
|
|
try:
|
|
cmd = ["curl", "-sfk", "--connect-timeout", str(timeout)]
|
|
if headers:
|
|
for k, v in headers.items():
|
|
cmd.extend(["-H", f"{k}: {v}"])
|
|
cmd.append(url)
|
|
result = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout+3)
|
|
return json.loads(result.stdout) if result.returncode == 0 and result.stdout else None
|
|
except:
|
|
return None
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# CHECK 1: LiteLLM Key Validation
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
def check_keys():
|
|
for name, agent in AGENTS.items():
|
|
data = http_json(f"{LITELLM}/v1/models",
|
|
headers={"Authorization": f"Bearer {agent['key']}"})
|
|
if data and data.get("data"):
|
|
model = data["data"][0].get("id", "?")
|
|
print(f" ✅ {name}: key valid → {model}")
|
|
else:
|
|
print(f" ❌ {name}: KEY FAILURE — auth rejected or unreachable")
|
|
FAIL.append(f"key:{name}")
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# CHECK 2: GPU Port Conflict Detection
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
def check_gpu_ports():
|
|
for label, gpu in GPU_HOSTS.items():
|
|
host = gpu["host"]
|
|
port = gpu["port"]
|
|
svc = gpu["service"]
|
|
|
|
svc_status = ssh(host, f"systemctl is-active {svc}")
|
|
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1")
|
|
|
|
if not svc_status:
|
|
print(f" ❌ {label}: UNREACHABLE")
|
|
FAIL.append(f"gpu-unreachable:{host}")
|
|
continue
|
|
|
|
if not port_owner:
|
|
print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})")
|
|
FAIL.append(f"gpu-no-port:{label}")
|
|
elif svc_status != "active":
|
|
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2")
|
|
if svc_pid and port_owner != svc_pid:
|
|
print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})")
|
|
FAIL.append(f"gpu-ghost:{label}:{port_owner}")
|
|
else:
|
|
print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}")
|
|
else:
|
|
# Verify health endpoint
|
|
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health")
|
|
if health and '"status":"ok"' in health:
|
|
print(f" ✅ {label}: healthy (pid={port_owner})")
|
|
elif health and '"status":"no slot available"' in health:
|
|
print(f" ✅ {label}: healthy (loading, pid={port_owner})")
|
|
elif health and '"error"' in health.lower():
|
|
print(f" ⚠️ {label}: error response (pid={port_owner}): {health[:80]}")
|
|
else:
|
|
print(f" ⚠️ {label}: unknown health (pid={port_owner}): {str(health)[:80]}")
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# CHECK 3: Agent Gateway Liveness + Streaming
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
def check_agents():
|
|
for name, agent in AGENTS.items():
|
|
host = agent.get("host")
|
|
user = agent.get("user")
|
|
ct = agent["ct"]
|
|
|
|
if not host or not user:
|
|
print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check")
|
|
continue
|
|
|
|
# Gateway process
|
|
pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | head -1", user=user)
|
|
if not pid:
|
|
print(f" ❌ {name}: GATEWAY NOT RUNNING")
|
|
FAIL.append(f"gateway-down:{name}")
|
|
continue
|
|
|
|
# Gateway state file
|
|
state = ssh(host, "cat ~/.hermes/gateway_state.json 2>/dev/null", user=user)
|
|
if state:
|
|
try:
|
|
st = json.loads(state)
|
|
gw_state = st.get("gateway_state", "?")
|
|
zulip = st.get("platforms", {}).get("zulip", {}).get("state", "?")
|
|
except:
|
|
gw_state, zulip = "corrupt", "?"
|
|
else:
|
|
gw_state, zulip = "no-state-file", "?"
|
|
|
|
# Zulip streaming: does adapter have edit_message?
|
|
adapter_paths = [
|
|
"~/.hermes/plugins/zulip-platform/adapter.py",
|
|
"~/.hermes/plugins/platforms/zulip/adapter.py",
|
|
]
|
|
streaming = "no"
|
|
for p in adapter_paths:
|
|
has_edit = ssh(host, f"grep -c 'async def edit_message' {p} 2>/dev/null", user=user)
|
|
if has_edit and has_edit != "0":
|
|
streaming = "yes"
|
|
break
|
|
|
|
# Recent errors
|
|
recent_errors = ssh(host,
|
|
r"journalctl --user -u hermes-gateway --since '10 min ago' -o cat --no-pager 2>/dev/null "
|
|
r"| grep -ci 'error\|traceback\|exception\|401\|403\|500' || echo 0",
|
|
user=user)
|
|
recent_errors = (recent_errors or "0").strip().split("\n")[-1] # take last line
|
|
|
|
print(f" {'✅' if gw_state == 'running' and zulip == 'connected' else '⚠️'} "
|
|
f"{name}: gw={gw_state} zulip={zulip} streaming={streaming} "
|
|
f"errors_10m={recent_errors.strip() or '0'} pid={pid}")
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# MAIN
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
def main():
|
|
quiet = "--quiet" in sys.argv
|
|
as_json = "--json" in sys.argv
|
|
|
|
if not quiet:
|
|
print(f"🏥 Agent Health Check — {datetime.now().strftime('%Y-%m-%d %H:%M UTC')}")
|
|
print()
|
|
|
|
print("🔑 LiteLLM Keys:")
|
|
check_keys()
|
|
print()
|
|
|
|
print("🎮 GPU Port Health:")
|
|
check_gpu_ports()
|
|
print()
|
|
|
|
print("🤖 Agent Gateways:")
|
|
check_agents()
|
|
|
|
if FAIL:
|
|
print(f"\n❌ {len(FAIL)} FAILURE(S): {' | '.join(FAIL)}")
|
|
if quiet:
|
|
# In quiet mode, only print failures as a single alert line
|
|
print(f"ALERT agent-health:{','.join(FAIL)}")
|
|
elif not quiet:
|
|
print("\n✅ All checks passed")
|
|
|
|
if as_json:
|
|
print(json.dumps({"timestamp": datetime.now().isoformat(),
|
|
"failures": FAIL, "healthy": len(FAIL) == 0}))
|
|
|
|
sys.exit(1 if FAIL else 0)
|
|
|
|
if __name__ == "__main__":
|
|
main()
|