From 2716e55c16fb1bfc30b1617ef034eefc2c0b673f Mon Sep 17 00:00:00 2001 From: root Date: Tue, 8 Sep 2026 11:08:28 +0000 Subject: [PATCH] fix(agent-health): repoint .8 GPU unit, fix pid UnboundLocalError, source abiba key from env.sh - GPU unit repoint verified live 2026-09-08: .8 rtx3090 probes llama-chat-api.service (stale llama-server unit read inactive -> false UNREACHABLE for a healthy process); .110 keeps llama-server.service (ocu-llm VM), .15 keeps strix-server.service. is-active no longer swallowed as SSH failure (|| true). - check_agents: bind pid before the summary f-string so the non-report-only path (abiba/koonimo) no longer raises UnboundLocalError (line 305 crash). - abiba key leg: read LITELLM_API_KEY from /root/.pi/agent/env.sh (#735 moved creds out of shared /root/.bashrc); 'abiba NO KEY' gone on healthy setup. --- scripts/agent-health-check.py | 84 ++++++++++++++++++++++++++++++----- 1 file changed, 72 insertions(+), 12 deletions(-) diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index 22a3b98..00c7051 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -19,6 +19,13 @@ Changelog: name format ({NAME}_LITELLM_API_KEY not LITELLM_API_KEY_{NAME}). Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). + v3 (2026-09-08): GPU unit repoint verified live (.8 llama-chat-api.service, + .110 llama-server.service, .15 strix-server.service) — .8 was probing a stale + llama-server unit that reads inactive, producing false UNREACHABLE legs. + systemctl is-active no longer swallows non-zero exit as SSH failure. + Fixed UnboundLocalError on the abiba/koonimo gateway leg (pid unbound in the + summary f-string). Abiba's LiteLLM key now comes from /root/.pi/agent/env.sh + (#735 agent separation; creds moved out of shared /root/.bashrc). """ import subprocess, json, sys, os, time @@ -40,15 +47,25 @@ PVE_NODES = { # Agent definitions: ct, host, user, pve_node, vault_key_name AGENTS = { "tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"}, - "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", "vault_key": None}, # Pi agent + Mumuni Zulip, no vault key + # abiba = pi agent (.24) — no vault key; its LiteLLM key is read from its + # local env file (key_env below), not from the shared vault or .bashrc. + "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", + "vault_key": None, + "key_env": {"file": "/root/.pi/agent/env.sh", "var": "LITELLM_API_KEY"}}, "koby": {"ct": 111, "host": "192.168.68.129", "user": "root", "pve": "amdpve", "vault_key": "KOBY_LITELLM_API_KEY"}, "koonimo": {"ct": 113, "host": "192.168.68.114", "user": "root", "pve": "amdpve", "vault_key": "KOONIMO_LITELLM_API_KEY"}, } +# Systemd units verified live 2026-09-08 (systemctl list-units on each host): +# .8 rtx3090 (gpu-dense) -> llama-chat-api.service (active; the old +# llama-server.service unit file is stale/inactive — probing it read as +# UNREACHABLE for a healthy process) +# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active) +# .15 strixhalo (amdpve) -> strix-server.service (active) GPU_HOSTS = { - "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-server"}, - "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server"}, - "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server"}, + "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"}, + "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"}, + "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"}, } FAIL = [] @@ -161,10 +178,41 @@ def _get_agent_key(agent_name, vault_key_name): return None -# Inject keys from vault for each agent + +def _read_env_export(path, var): + """Parse `export VAR=value` (or `VAR=value`) out of a local env file. + + #735 agent separation (2026-09-06): agent creds moved out of the shared + /root/.bashrc into per-agent env files under /root/.pi/agent/ (bashrc's + source line keeps abiba shells resolving them, but the file of record is + env.sh). Do NOT fall back to /root/.bashrc here: desktop (.200) SSH + sessions override LITELLM_API_KEY with mumuni's key, so sourcing bashrc + would validate the wrong identity. + """ + try: + with open(os.path.expanduser(path)) as _f: + for line in _f: + line = line.strip() + if not (line.startswith("export " + var + "=") or line.startswith(var + "=")): + continue + value = line.split("=", 1)[1].strip().strip('"').strip("'") + if value: + return value + except (OSError, UnicodeDecodeError): + pass + return None + + +# Inject keys for each agent: +# - vault-backed agents (tanko/koby/koonimo): {NAME}_LITELLM_API_KEY from +# Infisical (project 322fceab-39da-4854-a55a-568e76c0f13f, env prod). +# - abiba (pi agent, no vault key): LITELLM_API_KEY from its local env file +# /root/.pi/agent/env.sh (moved there from /root/.bashrc in #735). for agent_name in AGENTS: info = AGENTS[agent_name] key = _get_agent_key(agent_name, info.get("vault_key")) + if not key and info.get("key_env"): + key = _read_env_export(info["key_env"]["file"], info["key_env"]["var"]) AGENTS[agent_name]["key"] = key @@ -176,7 +224,7 @@ def check_keys(): for name, agent in AGENTS.items(): key = agent.get("key") if not key: - print(f" ❌ {name}: NO KEY FOUND (vault empty or unreachable)") + print(f" ❌ {name}: NO KEY FOUND (vault/env empty or unreachable)") FAIL.append(f"key:{name}:no-key") continue data = http_json(f"{LITELLM}/v1/models", @@ -190,7 +238,7 @@ def check_keys(): # ═══════════════════════════════════════════════════════════════════ -# CHECK 2: GPU Port Conflict Detection (unchanged) +# CHECK 2: GPU Port Conflict Detection (unit names verified live 2026-09-08) # ═══════════════════════════════════════════════════════════════════ def check_gpu_ports(): @@ -199,7 +247,11 @@ def check_gpu_ports(): port = gpu["port"] svc = gpu["service"] - svc_status = ssh(host, f"systemctl is-active {svc}") + # `systemctl is-active` exits non-zero when the unit is inactive or + # missing, which the ssh() helper would swallow as an SSH failure and + # report as UNREACHABLE. `|| true` keeps the real state word so we can + # tell "unit inactive" from "host unreachable". + svc_status = ssh(host, f"systemctl is-active {svc} || true") port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1") if not svc_status: @@ -254,14 +306,22 @@ def check_agents(): print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check") continue + # Resolve the Hermes gateway PID once, before the report-only branch: + # the summary line below renders `pid`, and it used to be bound only in + # the report-only path — leaving it unbound on the abiba/koonimo path + # raised UnboundLocalError and crashed the whole check. Agents without + # a gateway get pid=?. + pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) + if not pid: + pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) + if not pid: + pid = "?" + # ⛔ KOBY IS NEVER REPAIRED — diagnostic only if report_only: print(f" 🔍 {name}: REPORT-ONLY mode (diagnostic only, no repairs on .129)") # Still check gateway status for reporting purposes - pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) - if not pid: - pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) - if not pid: + if pid == "?": print(f" ⚠️ {name}: GATEWAY NOT RUNNING (reported only)") FAIL.append(f"gateway-down:{name}") continue