From 2716e55c16fb1bfc30b1617ef034eefc2c0b673f Mon Sep 17 00:00:00 2001 From: root Date: Tue, 8 Sep 2026 11:08:28 +0000 Subject: [PATCH 1/4] fix(agent-health): repoint .8 GPU unit, fix pid UnboundLocalError, source abiba key from env.sh - GPU unit repoint verified live 2026-09-08: .8 rtx3090 probes llama-chat-api.service (stale llama-server unit read inactive -> false UNREACHABLE for a healthy process); .110 keeps llama-server.service (ocu-llm VM), .15 keeps strix-server.service. is-active no longer swallowed as SSH failure (|| true). - check_agents: bind pid before the summary f-string so the non-report-only path (abiba/koonimo) no longer raises UnboundLocalError (line 305 crash). - abiba key leg: read LITELLM_API_KEY from /root/.pi/agent/env.sh (#735 moved creds out of shared /root/.bashrc); 'abiba NO KEY' gone on healthy setup. --- scripts/agent-health-check.py | 84 ++++++++++++++++++++++++++++++----- 1 file changed, 72 insertions(+), 12 deletions(-) diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index 22a3b98..00c7051 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -19,6 +19,13 @@ Changelog: name format ({NAME}_LITELLM_API_KEY not LITELLM_API_KEY_{NAME}). Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). + v3 (2026-09-08): GPU unit repoint verified live (.8 llama-chat-api.service, + .110 llama-server.service, .15 strix-server.service) — .8 was probing a stale + llama-server unit that reads inactive, producing false UNREACHABLE legs. + systemctl is-active no longer swallows non-zero exit as SSH failure. + Fixed UnboundLocalError on the abiba/koonimo gateway leg (pid unbound in the + summary f-string). Abiba's LiteLLM key now comes from /root/.pi/agent/env.sh + (#735 agent separation; creds moved out of shared /root/.bashrc). """ import subprocess, json, sys, os, time @@ -40,15 +47,25 @@ PVE_NODES = { # Agent definitions: ct, host, user, pve_node, vault_key_name AGENTS = { "tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"}, - "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", "vault_key": None}, # Pi agent + Mumuni Zulip, no vault key + # abiba = pi agent (.24) — no vault key; its LiteLLM key is read from its + # local env file (key_env below), not from the shared vault or .bashrc. + "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", + "vault_key": None, + "key_env": {"file": "/root/.pi/agent/env.sh", "var": "LITELLM_API_KEY"}}, "koby": {"ct": 111, "host": "192.168.68.129", "user": "root", "pve": "amdpve", "vault_key": "KOBY_LITELLM_API_KEY"}, "koonimo": {"ct": 113, "host": "192.168.68.114", "user": "root", "pve": "amdpve", "vault_key": "KOONIMO_LITELLM_API_KEY"}, } +# Systemd units verified live 2026-09-08 (systemctl list-units on each host): +# .8 rtx3090 (gpu-dense) -> llama-chat-api.service (active; the old +# llama-server.service unit file is stale/inactive — probing it read as +# UNREACHABLE for a healthy process) +# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active) +# .15 strixhalo (amdpve) -> strix-server.service (active) GPU_HOSTS = { - "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-server"}, - "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server"}, - "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server"}, + "gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"}, + "gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"}, + "gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"}, } FAIL = [] @@ -161,10 +178,41 @@ def _get_agent_key(agent_name, vault_key_name): return None -# Inject keys from vault for each agent + +def _read_env_export(path, var): + """Parse `export VAR=value` (or `VAR=value`) out of a local env file. + + #735 agent separation (2026-09-06): agent creds moved out of the shared + /root/.bashrc into per-agent env files under /root/.pi/agent/ (bashrc's + source line keeps abiba shells resolving them, but the file of record is + env.sh). Do NOT fall back to /root/.bashrc here: desktop (.200) SSH + sessions override LITELLM_API_KEY with mumuni's key, so sourcing bashrc + would validate the wrong identity. + """ + try: + with open(os.path.expanduser(path)) as _f: + for line in _f: + line = line.strip() + if not (line.startswith("export " + var + "=") or line.startswith(var + "=")): + continue + value = line.split("=", 1)[1].strip().strip('"').strip("'") + if value: + return value + except (OSError, UnicodeDecodeError): + pass + return None + + +# Inject keys for each agent: +# - vault-backed agents (tanko/koby/koonimo): {NAME}_LITELLM_API_KEY from +# Infisical (project 322fceab-39da-4854-a55a-568e76c0f13f, env prod). +# - abiba (pi agent, no vault key): LITELLM_API_KEY from its local env file +# /root/.pi/agent/env.sh (moved there from /root/.bashrc in #735). for agent_name in AGENTS: info = AGENTS[agent_name] key = _get_agent_key(agent_name, info.get("vault_key")) + if not key and info.get("key_env"): + key = _read_env_export(info["key_env"]["file"], info["key_env"]["var"]) AGENTS[agent_name]["key"] = key @@ -176,7 +224,7 @@ def check_keys(): for name, agent in AGENTS.items(): key = agent.get("key") if not key: - print(f" ❌ {name}: NO KEY FOUND (vault empty or unreachable)") + print(f" ❌ {name}: NO KEY FOUND (vault/env empty or unreachable)") FAIL.append(f"key:{name}:no-key") continue data = http_json(f"{LITELLM}/v1/models", @@ -190,7 +238,7 @@ def check_keys(): # ═══════════════════════════════════════════════════════════════════ -# CHECK 2: GPU Port Conflict Detection (unchanged) +# CHECK 2: GPU Port Conflict Detection (unit names verified live 2026-09-08) # ═══════════════════════════════════════════════════════════════════ def check_gpu_ports(): @@ -199,7 +247,11 @@ def check_gpu_ports(): port = gpu["port"] svc = gpu["service"] - svc_status = ssh(host, f"systemctl is-active {svc}") + # `systemctl is-active` exits non-zero when the unit is inactive or + # missing, which the ssh() helper would swallow as an SSH failure and + # report as UNREACHABLE. `|| true` keeps the real state word so we can + # tell "unit inactive" from "host unreachable". + svc_status = ssh(host, f"systemctl is-active {svc} || true") port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1") if not svc_status: @@ -254,14 +306,22 @@ def check_agents(): print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check") continue + # Resolve the Hermes gateway PID once, before the report-only branch: + # the summary line below renders `pid`, and it used to be bound only in + # the report-only path — leaving it unbound on the abiba/koonimo path + # raised UnboundLocalError and crashed the whole check. Agents without + # a gateway get pid=?. + pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) + if not pid: + pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) + if not pid: + pid = "?" + # ⛔ KOBY IS NEVER REPAIRED — diagnostic only if report_only: print(f" 🔍 {name}: REPORT-ONLY mode (diagnostic only, no repairs on .129)") # Still check gateway status for reporting purposes - pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) - if not pid: - pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) - if not pid: + if pid == "?": print(f" ⚠️ {name}: GATEWAY NOT RUNNING (reported only)") FAIL.append(f"gateway-down:{name}") continue -- 2.54.0 From af3d3642421da174049a5d4cccc9cceda5eb9848 Mon Sep 17 00:00:00 2001 From: root Date: Tue, 8 Sep 2026 12:04:01 +0000 Subject: [PATCH 2/4] no-mistakes(review): Bracket pgrep patterns to stop ssh wrapper self-match --- scripts/agent-health-check.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index 00c7051..ae94aa1 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -311,9 +311,9 @@ def check_agents(): # the report-only path — leaving it unbound on the abiba/koonimo path # raised UnboundLocalError and crashed the whole check. Agents without # a gateway get pid=?. - pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) + pid = ssh(host, "pgrep -f '[h]ermes_cli.main gateway run' | grep -v infisical | head -1", user=user) if not pid: - pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) + pid = ssh(host, "pgrep -f '[h]ermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) if not pid: pid = "?" -- 2.54.0 From bbe9ee533f0449bea9060a6093c7d2b65c9bc27c Mon Sep 17 00:00:00 2001 From: root Date: Tue, 8 Sep 2026 12:18:14 +0000 Subject: [PATCH 3/4] no-mistakes(document): Docs updated for .8 unit repoint and abiba env.sh sourcing --- gpu-fleet.prose.md | 3 ++- infrastructure-control.prose.md | 2 +- litellm-self-heal.prose.md | 2 +- 3 files changed, 4 insertions(+), 3 deletions(-) diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index f3b4aab..847c589 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -220,7 +220,8 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. | `/root/scripts/gpu_benchmark.py` | pi (.24) | GPU inference benchmark module (tok/s tracking) | | `/root/scripts/gpu-saturation-watchdog.py` | pi (.24) | Auto-restart stuck llama-server | | `/root/dashboard/gpu-fleet.html` | pi (.24) | Live HTML dashboard | -| `/etc/systemd/system/llama-server.service` | .8, .110 | llama-server daemons (Nvidia GPUs) | +| `/etc/systemd/system/llama-chat-api.service` | .8 | RTX 3090 GPU daemon — active unit since 2026-09-06; the old `llama-server.service` file on .8 is stale/inactive | +| `/etc/systemd/system/llama-server.service` | .110 | llama-server daemon (RTX 5070) | | `/etc/systemd/system/strix-server.service` | .15 (amdpve) | llama-server daemon (Vulkan, Strix Halo) running unsloth/Qwen3.6-35B-A3B-MTP-GGUF. Note: `llama-server.service` and `llama-server@.service` are **masked** on .15 to prevent port 8080 collisions. | ## Prometheus & Grafana diff --git a/infrastructure-control.prose.md b/infrastructure-control.prose.md index fe3e350..65b87fe 100644 --- a/infrastructure-control.prose.md +++ b/infrastructure-control.prose.md @@ -593,7 +593,7 @@ curl -s -X POST http://192.168.68.116:9000/admin/roster/reload \ -H "Authorization: Bearer sk-admin-ee09fffd04978b61a1569ac670c68814" # Restart stuck GPU (saturation watchdog alternative) -ssh root@192.168.68.8 "systemctl restart llama-server" +ssh root@192.168.68.8 "systemctl restart llama-chat-api.service" # .8 active unit since 2026-09-06; llama-server.service file on .8 is stale/inactive ssh root@192.168.68.110 "systemctl restart llama-server" ``` diff --git a/litellm-self-heal.prose.md b/litellm-self-heal.prose.md index c7680a8..66972b1 100644 --- a/litellm-self-heal.prose.md +++ b/litellm-self-heal.prose.md @@ -113,7 +113,7 @@ Preferred implementation: uncap shared pool, add capped alias for crew-only. - **Health-check script** (`/opt/inference-harness/scripts/litellm-health-check.sh` on CT 116): `gpu-fleet` check fails only on **critical** alerts (warnings are informational). Tests `strix-moe` (not `ornith-1.0-35b`). - **GPU monitor** (`/root/scripts/gpu-monitor-server.py` on pi .24): runs as **systemd unit `gpu-monitor.service`** (was bare `&` process). `gpu_count` includes Strix Halo (was 2, now 3). VRAM alert thresholds: warning 93%, critical 97% (raised from 90/95 — 128K context steady-state is ~70% on RTX 3090, not a fault). -- **Agent key monitor** (`/root/scripts/agent-health-check.py` on pi .24, cron `*/10`): v2 (2026-07-26) — reads each agent's **agent-specific** `{NAME}_LITELLM_API_KEY` from Infisical vault (not the shared master key). Covers: LiteLLM keys, GPU ports, agent gateways (all 5 agents now SSHa ble), CT liveness (pct status on PVE nodes), config.yaml YAML integrity, wrapper/CLI integrity, vault secret non-emptiness checks. Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). Legacy `tdunna`/`baggy` replaced with canonical agent hostnames. +- **Agent key monitor** (`/root/scripts/agent-health-check.py` on pi .24, cron `*/10`): v3 (2026-09-08) — vault-backed agents (tanko/koby/koonimo) read their **agent-specific** `{NAME}_LITELLM_API_KEY` from Infisical vault (not the shared master key); abiba (pi agent) reads `LITELLM_API_KEY` from its local `/root/.pi/agent/env.sh` (#735 — moved out of shared `/root/.bashrc`), not from the vault. Covers: LiteLLM keys, GPU ports, agent gateways (all 5 agents now SSHa ble), CT liveness (pct status on PVE nodes), config.yaml YAML integrity, wrapper/CLI integrity, vault secret non-emptiness checks. Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114), abiba (.24). Legacy `tdunna`/`baggy` replaced with canonical agent hostnames. - **Stale keys cleaned**: `daily-infra-report.py` SYNTHETIC_API_KEY was stale (`sk-U_ydi3B` → 401); now reads `LITELLM_MASTER_KEY` from env. Deprecated scripts (`router-original.py`, `router-phase0-backup.py`, `apply-fixes.py`) still reference `sk-syslog-local-master-key` but do not actively poll LiteLLM. ## Maintains -- 2.54.0 From 2e0b737f2ddff7b06bc5d2cf32a38e986c8e0271 Mon Sep 17 00:00:00 2001 From: root Date: Tue, 8 Sep 2026 12:30:13 +0000 Subject: [PATCH 4/4] =?UTF-8?q?revert:=20drop=20co-gated=20prose=20edits?= =?UTF-8?q?=20(gpu-fleet,=20infrastructure-control)=20=E2=80=94=20script-o?= =?UTF-8?q?nly=20scope=20per=20captain=20decision?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The no-mistakes document step had auto-updated the .8 systemd unit name (llama-server -> llama-chat-api.service) in gpu-fleet.prose.md and infrastructure-control.prose.md. Those contracts are co-gated with another agent and cannot be amended inside a script-scope task (firstmate decision 2026-09-08, key prose-doc-step-scope); the unit-name doc sync is separate follow-up work. This restores both files to their origin/master content. The litellm-self-heal.prose.md agent-health-check v3/env.sh description update (same document commit) is intentional and kept. --- gpu-fleet.prose.md | 3 +-- infrastructure-control.prose.md | 2 +- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index 847c589..f3b4aab 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -220,8 +220,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. | `/root/scripts/gpu_benchmark.py` | pi (.24) | GPU inference benchmark module (tok/s tracking) | | `/root/scripts/gpu-saturation-watchdog.py` | pi (.24) | Auto-restart stuck llama-server | | `/root/dashboard/gpu-fleet.html` | pi (.24) | Live HTML dashboard | -| `/etc/systemd/system/llama-chat-api.service` | .8 | RTX 3090 GPU daemon — active unit since 2026-09-06; the old `llama-server.service` file on .8 is stale/inactive | -| `/etc/systemd/system/llama-server.service` | .110 | llama-server daemon (RTX 5070) | +| `/etc/systemd/system/llama-server.service` | .8, .110 | llama-server daemons (Nvidia GPUs) | | `/etc/systemd/system/strix-server.service` | .15 (amdpve) | llama-server daemon (Vulkan, Strix Halo) running unsloth/Qwen3.6-35B-A3B-MTP-GGUF. Note: `llama-server.service` and `llama-server@.service` are **masked** on .15 to prevent port 8080 collisions. | ## Prometheus & Grafana diff --git a/infrastructure-control.prose.md b/infrastructure-control.prose.md index 65b87fe..fe3e350 100644 --- a/infrastructure-control.prose.md +++ b/infrastructure-control.prose.md @@ -593,7 +593,7 @@ curl -s -X POST http://192.168.68.116:9000/admin/roster/reload \ -H "Authorization: Bearer sk-admin-ee09fffd04978b61a1569ac670c68814" # Restart stuck GPU (saturation watchdog alternative) -ssh root@192.168.68.8 "systemctl restart llama-chat-api.service" # .8 active unit since 2026-09-06; llama-server.service file on .8 is stale/inactive +ssh root@192.168.68.8 "systemctl restart llama-server" ssh root@192.168.68.110 "systemctl restart llama-server" ``` -- 2.54.0