diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py new file mode 100755 index 0000000..13df095 --- /dev/null +++ b/scripts/agent-health-check.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +""" +/root/scripts/agent-health-check.py — Consolidated Agent Health Verification + +Single non-disruptive health check replacing 7 scattered scripts. +Verifies: LiteLLM keys, GPU port conflicts, agent Zulip streaming, +gateway liveness, and gateway log health. NEVER restarts anything. + +Usage: + python3 /root/scripts/agent-health-check.py # Full check + python3 /root/scripts/agent-health-check.py --json # Machine-readable + python3 /root/scripts/agent-health-check.py --quiet # Only output on failure + +Cron: */10 * * * * python3 /root/scripts/agent-health-check.py --quiet +""" + +import subprocess, json, sys, os, time +from datetime import datetime + +LITELLM = "http://192.168.68.116:80" + +AGENTS = { + "tanko": {"ct": 112, "host": "192.168.68.122", "key": "sk-CggiHWlamQyShxWC3Hx6uw", "user": "jerome"}, + "mumuni": {"ct": 114, "host": "192.168.68.123", "key": "sk-VrqCNlwUgzoNGOpikJ7nwQ", "user": "root"}, + "tdunna": {"ct": 111, "host": None, "key": "sk-6sbCNjz2T6lTVDBdlNHXsA", "user": None}, + "baggy": {"ct": 113, "host": None, "key": "sk-krnw_zGBwvvL5b7l2t-s-A", "user": None}, +} + +GPU_HOSTS = { + "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-server"}, + "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server"}, + "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "ornith-server"}, +} + +FAIL = [] + +def ssh(host, cmd, user="root"): + """Execute a command on a remote host, return stdout or None.""" + try: + result = subprocess.run( + ["ssh", "-o", "StrictHostKeyChecking=no", "-o", "ConnectTimeout=8", + f"{user}@{host}", cmd], + capture_output=True, text=True, timeout=15 + ) + return result.stdout.strip() if result.returncode == 0 else None + except: + return None + +def http_get(url, headers=None, timeout=5): + """Return HTTP status code as string.""" + try: + cmd = ["curl", "-sfk", "--connect-timeout", str(timeout), "-o", "/dev/null", "-w", "%{http_code}"] + if headers: + for k, v in headers.items(): + cmd.extend(["-H", f"{k}: {v}"]) + cmd.append(url) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout+3) + return result.stdout.strip() or "000" + except: + return "timeout" + +def http_json(url, headers=None, timeout=5): + """Return parsed JSON from URL, or None.""" + try: + cmd = ["curl", "-sfk", "--connect-timeout", str(timeout)] + if headers: + for k, v in headers.items(): + cmd.extend(["-H", f"{k}: {v}"]) + cmd.append(url) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout+3) + return json.loads(result.stdout) if result.returncode == 0 and result.stdout else None + except: + return None + + +# ═══════════════════════════════════════════════════════════════════ +# CHECK 1: LiteLLM Key Validation +# ═══════════════════════════════════════════════════════════════════ + +def check_keys(): + for name, agent in AGENTS.items(): + data = http_json(f"{LITELLM}/v1/models", + headers={"Authorization": f"Bearer {agent['key']}"}) + if data and data.get("data"): + model = data["data"][0].get("id", "?") + print(f" ✅ {name}: key valid → {model}") + else: + print(f" ❌ {name}: KEY FAILURE — auth rejected or unreachable") + FAIL.append(f"key:{name}") + + +# ═══════════════════════════════════════════════════════════════════ +# CHECK 2: GPU Port Conflict Detection +# ═══════════════════════════════════════════════════════════════════ + +def check_gpu_ports(): + for label, gpu in GPU_HOSTS.items(): + host = gpu["host"] + port = gpu["port"] + svc = gpu["service"] + + svc_status = ssh(host, f"systemctl is-active {svc}") + port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1") + + if not svc_status: + print(f" ❌ {label}: UNREACHABLE") + FAIL.append(f"gpu-unreachable:{host}") + continue + + if not port_owner: + print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})") + FAIL.append(f"gpu-no-port:{label}") + elif svc_status != "active": + svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2") + if svc_pid and port_owner != svc_pid: + print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})") + FAIL.append(f"gpu-ghost:{label}:{port_owner}") + else: + print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}") + else: + # Verify health endpoint + health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health") + if health and '"status":"ok"' in health: + print(f" ✅ {label}: healthy (pid={port_owner})") + elif health and '"status":"no slot available"' in health: + print(f" ✅ {label}: healthy (loading, pid={port_owner})") + elif health and '"error"' in health.lower(): + print(f" ⚠️ {label}: error response (pid={port_owner}): {health[:80]}") + else: + print(f" ⚠️ {label}: unknown health (pid={port_owner}): {str(health)[:80]}") + + +# ═══════════════════════════════════════════════════════════════════ +# CHECK 3: Agent Gateway Liveness + Streaming +# ═══════════════════════════════════════════════════════════════════ + +def check_agents(): + for name, agent in AGENTS.items(): + host = agent.get("host") + user = agent.get("user") + ct = agent["ct"] + + if not host or not user: + print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check") + continue + + # Gateway process + pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | head -1", user=user) + if not pid: + print(f" ❌ {name}: GATEWAY NOT RUNNING") + FAIL.append(f"gateway-down:{name}") + continue + + # Gateway state file + state = ssh(host, "cat ~/.hermes/gateway_state.json 2>/dev/null", user=user) + if state: + try: + st = json.loads(state) + gw_state = st.get("gateway_state", "?") + zulip = st.get("platforms", {}).get("zulip", {}).get("state", "?") + except: + gw_state, zulip = "corrupt", "?" + else: + gw_state, zulip = "no-state-file", "?" + + # Zulip streaming: does adapter have edit_message? + adapter_paths = [ + "~/.hermes/plugins/zulip-platform/adapter.py", + "~/.hermes/plugins/platforms/zulip/adapter.py", + ] + streaming = "no" + for p in adapter_paths: + has_edit = ssh(host, f"grep -c 'async def edit_message' {p} 2>/dev/null", user=user) + if has_edit and has_edit != "0": + streaming = "yes" + break + + # Recent errors + recent_errors = ssh(host, + r"journalctl --user -u hermes-gateway --since '10 min ago' -o cat --no-pager 2>/dev/null " + r"| grep -ci 'error\|traceback\|exception\|401\|403\|500' || echo 0", + user=user) + recent_errors = (recent_errors or "0").strip().split("\n")[-1] # take last line + + print(f" {'✅' if gw_state == 'running' and zulip == 'connected' else '⚠️'} " + f"{name}: gw={gw_state} zulip={zulip} streaming={streaming} " + f"errors_10m={recent_errors.strip() or '0'} pid={pid}") + + +# ═══════════════════════════════════════════════════════════════════ +# MAIN +# ═══════════════════════════════════════════════════════════════════ + +def main(): + quiet = "--quiet" in sys.argv + as_json = "--json" in sys.argv + + if not quiet: + print(f"🏥 Agent Health Check — {datetime.now().strftime('%Y-%m-%d %H:%M UTC')}") + print() + + print("🔑 LiteLLM Keys:") + check_keys() + print() + + print("🎮 GPU Port Health:") + check_gpu_ports() + print() + + print("🤖 Agent Gateways:") + check_agents() + + if FAIL: + print(f"\n❌ {len(FAIL)} FAILURE(S): {' | '.join(FAIL)}") + if quiet: + # In quiet mode, only print failures as a single alert line + print(f"ALERT agent-health:{','.join(FAIL)}") + elif not quiet: + print("\n✅ All checks passed") + + if as_json: + print(json.dumps({"timestamp": datetime.now().isoformat(), + "failures": FAIL, "healthy": len(FAIL) == 0})) + + sys.exit(1 if FAIL else 0) + +if __name__ == "__main__": + main()