diff --git a/litellm-self-heal.prose.md b/litellm-self-heal.prose.md index c7680a8..66972b1 100644 --- a/litellm-self-heal.prose.md +++ b/litellm-self-heal.prose.md @@ -113,7 +113,7 @@ Preferred implementation: uncap shared pool, add capped alias for crew-only. - **Health-check script** (`/opt/inference-harness/scripts/litellm-health-check.sh` on CT 116): `gpu-fleet` check fails only on **critical** alerts (warnings are informational). Tests `strix-moe` (not `ornith-1.0-35b`). - **GPU monitor** (`/root/scripts/gpu-monitor-server.py` on pi .24): runs as **systemd unit `gpu-monitor.service`** (was bare `&` process). `gpu_count` includes Strix Halo (was 2, now 3). VRAM alert thresholds: warning 93%, critical 97% (raised from 90/95 — 128K context steady-state is ~70% on RTX 3090, not a fault). -- **Agent key monitor** (`/root/scripts/agent-health-check.py` on pi .24, cron `*/10`): v2 (2026-07-26) — reads each agent's **agent-specific** `{NAME}_LITELLM_API_KEY` from Infisical vault (not the shared master key). Covers: LiteLLM keys, GPU ports, agent gateways (all 5 agents now SSHa ble), CT liveness (pct status on PVE nodes), config.yaml YAML integrity, wrapper/CLI integrity, vault secret non-emptiness checks. Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). Legacy `tdunna`/`baggy` replaced with canonical agent hostnames. +- **Agent key monitor** (`/root/scripts/agent-health-check.py` on pi .24, cron `*/10`): v3 (2026-09-08) — vault-backed agents (tanko/koby/koonimo) read their **agent-specific** `{NAME}_LITELLM_API_KEY` from Infisical vault (not the shared master key); abiba (pi agent) reads `LITELLM_API_KEY` from its local `/root/.pi/agent/env.sh` (#735 — moved out of shared `/root/.bashrc`), not from the vault. Covers: LiteLLM keys, GPU ports, agent gateways (all 5 agents now SSHa ble), CT liveness (pct status on PVE nodes), config.yaml YAML integrity, wrapper/CLI integrity, vault secret non-emptiness checks. Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). Legacy `tdunna`/`baggy` replaced with canonical agent hostnames. - **Stale keys cleaned**: `daily-infra-report.py` SYNTHETIC_API_KEY was stale (`sk-U_ydi3B` → 401); now reads `LITELLM_MASTER_KEY` from env. Deprecated scripts (`router-original.py`, `router-phase0-backup.py`, `apply-fixes.py`) still reference `sk-syslog-local-master-key` but do not actively poll LiteLLM. ## Maintains diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index 22a3b98..ae94aa1 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -19,6 +19,13 @@ Changelog: name format ({NAME}_LITELLM_API_KEY not LITELLM_API_KEY_{NAME}). Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). + v3 (2026-09-08): GPU unit repoint verified live (.8 llama-chat-api.service, + .110 llama-server.service, .15 strix-server.service) — .8 was probing a stale + llama-server unit that reads inactive, producing false UNREACHABLE legs. + systemctl is-active no longer swallows non-zero exit as SSH failure. + Fixed UnboundLocalError on the abiba/koonimo gateway leg (pid unbound in the + summary f-string). Abiba's LiteLLM key now comes from /root/.pi/agent/env.sh + (#735 agent separation; creds moved out of shared /root/.bashrc). """ import subprocess, json, sys, os, time @@ -40,15 +47,25 @@ PVE_NODES = { # Agent definitions: ct, host, user, pve_node, vault_key_name AGENTS = { "tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"}, - "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", "vault_key": None}, # Pi agent + Mumuni Zulip, no vault key + # abiba = pi agent (.24) — no vault key; its LiteLLM key is read from its + # local env file (key_env below), not from the shared vault or .bashrc. + "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", + "vault_key": None, + "key_env": {"file": "/root/.pi/agent/env.sh", "var": "LITELLM_API_KEY"}}, "koby": {"ct": 111, "host": "192.168.68.129", "user": "root", "pve": "amdpve", "vault_key": "KOBY_LITELLM_API_KEY"}, "koonimo": {"ct": 113, "host": "192.168.68.114", "user": "root", "pve": "amdpve", "vault_key": "KOONIMO_LITELLM_API_KEY"}, } +# Systemd units verified live 2026-09-08 (systemctl list-units on each host): +# .8 rtx3090 (gpu-dense) -> llama-chat-api.service (active; the old +# llama-server.service unit file is stale/inactive — probing it read as +# UNREACHABLE for a healthy process) +# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active) +# .15 strixhalo (amdpve) -> strix-server.service (active) GPU_HOSTS = { - "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-server"}, - "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server"}, - "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server"}, + "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"}, + "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"}, + "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"}, } FAIL = [] @@ -161,10 +178,41 @@ def _get_agent_key(agent_name, vault_key_name): return None -# Inject keys from vault for each agent + +def _read_env_export(path, var): + """Parse `export VAR=value` (or `VAR=value`) out of a local env file. + + #735 agent separation (2026-09-06): agent creds moved out of the shared + /root/.bashrc into per-agent env files under /root/.pi/agent/ (bashrc's + source line keeps abiba shells resolving them, but the file of record is + env.sh). Do NOT fall back to /root/.bashrc here: desktop (.200) SSH + sessions override LITELLM_API_KEY with mumuni's key, so sourcing bashrc + would validate the wrong identity. + """ + try: + with open(os.path.expanduser(path)) as _f: + for line in _f: + line = line.strip() + if not (line.startswith("export " + var + "=") or line.startswith(var + "=")): + continue + value = line.split("=", 1)[1].strip().strip('"').strip("'") + if value: + return value + except (OSError, UnicodeDecodeError): + pass + return None + + +# Inject keys for each agent: +# - vault-backed agents (tanko/koby/koonimo): {NAME}_LITELLM_API_KEY from +# Infisical (project 322fceab-39da-4854-a55a-568e76c0f13f, env prod). +# - abiba (pi agent, no vault key): LITELLM_API_KEY from its local env file +# /root/.pi/agent/env.sh (moved there from /root/.bashrc in #735). for agent_name in AGENTS: info = AGENTS[agent_name] key = _get_agent_key(agent_name, info.get("vault_key")) + if not key and info.get("key_env"): + key = _read_env_export(info["key_env"]["file"], info["key_env"]["var"]) AGENTS[agent_name]["key"] = key @@ -176,7 +224,7 @@ def check_keys(): for name, agent in AGENTS.items(): key = agent.get("key") if not key: - print(f" ❌ {name}: NO KEY FOUND (vault empty or unreachable)") + print(f" ❌ {name}: NO KEY FOUND (vault/env empty or unreachable)") FAIL.append(f"key:{name}:no-key") continue data = http_json(f"{LITELLM}/v1/models", @@ -190,7 +238,7 @@ def check_keys(): # ═══════════════════════════════════════════════════════════════════ -# CHECK 2: GPU Port Conflict Detection (unchanged) +# CHECK 2: GPU Port Conflict Detection (unit names verified live 2026-09-08) # ═══════════════════════════════════════════════════════════════════ def check_gpu_ports(): @@ -199,7 +247,11 @@ def check_gpu_ports(): port = gpu["port"] svc = gpu["service"] - svc_status = ssh(host, f"systemctl is-active {svc}") + # `systemctl is-active` exits non-zero when the unit is inactive or + # missing, which the ssh() helper would swallow as an SSH failure and + # report as UNREACHABLE. `|| true` keeps the real state word so we can + # tell "unit inactive" from "host unreachable". + svc_status = ssh(host, f"systemctl is-active {svc} || true") port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1") if not svc_status: @@ -254,14 +306,22 @@ def check_agents(): print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check") continue + # Resolve the Hermes gateway PID once, before the report-only branch: + # the summary line below renders `pid`, and it used to be bound only in + # the report-only path — leaving it unbound on the abiba/koonimo path + # raised UnboundLocalError and crashed the whole check. Agents without + # a gateway get pid=?. + pid = ssh(host, "pgrep -f '[h]ermes_cli.main gateway run' | grep -v infisical | head -1", user=user) + if not pid: + pid = ssh(host, "pgrep -f '[h]ermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) + if not pid: + pid = "?" + # ⛔ KOBY IS NEVER REPAIRED — diagnostic only if report_only: print(f" 🔍 {name}: REPORT-ONLY mode (diagnostic only, no repairs on .129)") # Still check gateway status for reporting purposes - pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) - if not pid: - pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) - if not pid: + if pid == "?": print(f" ⚠️ {name}: GATEWAY NOT RUNNING (reported only)") FAIL.append(f"gateway-down:{name}") continue