|
|
|
@@ -19,6 +19,13 @@ Changelog:
|
|
|
|
|
name format ({NAME}_LITELLM_API_KEY not LITELLM_API_KEY_{NAME}).
|
|
|
|
|
Fleet roster: tanko (.122), mumuni (.24, inside abiba CT100), koby (.129), koonimo (.114),
|
|
|
|
|
abiba (.24).
|
|
|
|
|
v3 (2026-09-08): GPU unit repoint verified live (.8 llama-chat-api.service,
|
|
|
|
|
.110 llama-server.service, .15 strix-server.service) — .8 was probing a stale
|
|
|
|
|
llama-server unit that reads inactive, producing false UNREACHABLE legs.
|
|
|
|
|
systemctl is-active no longer swallows non-zero exit as SSH failure.
|
|
|
|
|
Fixed UnboundLocalError on the abiba/koonimo gateway leg (pid unbound in the
|
|
|
|
|
summary f-string). Abiba's LiteLLM key now comes from /root/.pi/agent/env.sh
|
|
|
|
|
(#735 agent separation; creds moved out of shared /root/.bashrc).
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
import subprocess, json, sys, os, time
|
|
|
|
@@ -40,15 +47,25 @@ PVE_NODES = {
|
|
|
|
|
# Agent definitions: ct, host, user, pve_node, vault_key_name
|
|
|
|
|
AGENTS = {
|
|
|
|
|
"tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"},
|
|
|
|
|
"abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", "vault_key": None}, # Pi agent + Mumuni Zulip, no vault key
|
|
|
|
|
# abiba = pi agent (.24) — no vault key; its LiteLLM key is read from its
|
|
|
|
|
# local env file (key_env below), not from the shared vault or .bashrc.
|
|
|
|
|
"abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve",
|
|
|
|
|
"vault_key": None,
|
|
|
|
|
"key_env": {"file": "/root/.pi/agent/env.sh", "var": "LITELLM_API_KEY"}},
|
|
|
|
|
"koby": {"ct": 111, "host": "192.168.68.129", "user": "root", "pve": "amdpve", "vault_key": "KOBY_LITELLM_API_KEY"},
|
|
|
|
|
"koonimo": {"ct": 113, "host": "192.168.68.114", "user": "root", "pve": "amdpve", "vault_key": "KOONIMO_LITELLM_API_KEY"},
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
# Systemd units verified live 2026-09-08 (systemctl list-units on each host):
|
|
|
|
|
# .8 rtx3090 (gpu-dense) -> llama-chat-api.service (active; the old
|
|
|
|
|
# llama-server.service unit file is stale/inactive — probing it read as
|
|
|
|
|
# UNREACHABLE for a healthy process)
|
|
|
|
|
# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active)
|
|
|
|
|
# .15 strixhalo (amdpve) -> strix-server.service (active)
|
|
|
|
|
GPU_HOSTS = {
|
|
|
|
|
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-server"},
|
|
|
|
|
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server"},
|
|
|
|
|
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server"},
|
|
|
|
|
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"},
|
|
|
|
|
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"},
|
|
|
|
|
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"},
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
FAIL = []
|
|
|
|
@@ -161,10 +178,41 @@ def _get_agent_key(agent_name, vault_key_name):
|
|
|
|
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
# Inject keys from vault for each agent
|
|
|
|
|
|
|
|
|
|
def _read_env_export(path, var):
|
|
|
|
|
"""Parse `export VAR=value` (or `VAR=value`) out of a local env file.
|
|
|
|
|
|
|
|
|
|
#735 agent separation (2026-09-06): agent creds moved out of the shared
|
|
|
|
|
/root/.bashrc into per-agent env files under /root/.pi/agent/ (bashrc's
|
|
|
|
|
source line keeps abiba shells resolving them, but the file of record is
|
|
|
|
|
env.sh). Do NOT fall back to /root/.bashrc here: desktop (.200) SSH
|
|
|
|
|
sessions override LITELLM_API_KEY with mumuni's key, so sourcing bashrc
|
|
|
|
|
would validate the wrong identity.
|
|
|
|
|
"""
|
|
|
|
|
try:
|
|
|
|
|
with open(os.path.expanduser(path)) as _f:
|
|
|
|
|
for line in _f:
|
|
|
|
|
line = line.strip()
|
|
|
|
|
if not (line.startswith("export " + var + "=") or line.startswith(var + "=")):
|
|
|
|
|
continue
|
|
|
|
|
value = line.split("=", 1)[1].strip().strip('"').strip("'")
|
|
|
|
|
if value:
|
|
|
|
|
return value
|
|
|
|
|
except (OSError, UnicodeDecodeError):
|
|
|
|
|
pass
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# Inject keys for each agent:
|
|
|
|
|
# - vault-backed agents (tanko/koby/koonimo): {NAME}_LITELLM_API_KEY from
|
|
|
|
|
# Infisical (project 322fceab-39da-4854-a55a-568e76c0f13f, env prod).
|
|
|
|
|
# - abiba (pi agent, no vault key): LITELLM_API_KEY from its local env file
|
|
|
|
|
# /root/.pi/agent/env.sh (moved there from /root/.bashrc in #735).
|
|
|
|
|
for agent_name in AGENTS:
|
|
|
|
|
info = AGENTS[agent_name]
|
|
|
|
|
key = _get_agent_key(agent_name, info.get("vault_key"))
|
|
|
|
|
if not key and info.get("key_env"):
|
|
|
|
|
key = _read_env_export(info["key_env"]["file"], info["key_env"]["var"])
|
|
|
|
|
AGENTS[agent_name]["key"] = key
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@@ -176,7 +224,7 @@ def check_keys():
|
|
|
|
|
for name, agent in AGENTS.items():
|
|
|
|
|
key = agent.get("key")
|
|
|
|
|
if not key:
|
|
|
|
|
print(f" ❌ {name}: NO KEY FOUND (vault empty or unreachable)")
|
|
|
|
|
print(f" ❌ {name}: NO KEY FOUND (vault/env empty or unreachable)")
|
|
|
|
|
FAIL.append(f"key:{name}:no-key")
|
|
|
|
|
continue
|
|
|
|
|
data = http_json(f"{LITELLM}/v1/models",
|
|
|
|
@@ -190,7 +238,7 @@ def check_keys():
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
|
# CHECK 2: GPU Port Conflict Detection (unchanged)
|
|
|
|
|
# CHECK 2: GPU Port Conflict Detection (unit names verified live 2026-09-08)
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
|
|
|
|
|
|
def check_gpu_ports():
|
|
|
|
@@ -199,7 +247,11 @@ def check_gpu_ports():
|
|
|
|
|
port = gpu["port"]
|
|
|
|
|
svc = gpu["service"]
|
|
|
|
|
|
|
|
|
|
svc_status = ssh(host, f"systemctl is-active {svc}")
|
|
|
|
|
# `systemctl is-active` exits non-zero when the unit is inactive or
|
|
|
|
|
# missing, which the ssh() helper would swallow as an SSH failure and
|
|
|
|
|
# report as UNREACHABLE. `|| true` keeps the real state word so we can
|
|
|
|
|
# tell "unit inactive" from "host unreachable".
|
|
|
|
|
svc_status = ssh(host, f"systemctl is-active {svc} || true")
|
|
|
|
|
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1")
|
|
|
|
|
|
|
|
|
|
if not svc_status:
|
|
|
|
@@ -254,14 +306,22 @@ def check_agents():
|
|
|
|
|
print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check")
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
# Resolve the Hermes gateway PID once, before the report-only branch:
|
|
|
|
|
# the summary line below renders `pid`, and it used to be bound only in
|
|
|
|
|
# the report-only path — leaving it unbound on the abiba/koonimo path
|
|
|
|
|
# raised UnboundLocalError and crashed the whole check. Agents without
|
|
|
|
|
# a gateway get pid=?.
|
|
|
|
|
pid = ssh(host, "pgrep -f '[h]ermes_cli.main gateway run' | grep -v infisical | head -1", user=user)
|
|
|
|
|
if not pid:
|
|
|
|
|
pid = ssh(host, "pgrep -f '[h]ermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user)
|
|
|
|
|
if not pid:
|
|
|
|
|
pid = "?"
|
|
|
|
|
|
|
|
|
|
# ⛔ KOBY IS NEVER REPAIRED — diagnostic only
|
|
|
|
|
if report_only:
|
|
|
|
|
print(f" 🔍 {name}: REPORT-ONLY mode (diagnostic only, no repairs on .129)")
|
|
|
|
|
# Still check gateway status for reporting purposes
|
|
|
|
|
pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user)
|
|
|
|
|
if not pid:
|
|
|
|
|
pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user)
|
|
|
|
|
if not pid:
|
|
|
|
|
if pid == "?":
|
|
|
|
|
print(f" ⚠️ {name}: GATEWAY NOT RUNNING (reported only)")
|
|
|
|
|
FAIL.append(f"gateway-down:{name}")
|
|
|
|
|
continue
|
|
|
|
|