From 2f961d7e7a127867185dfd344db684008a2ac95c Mon Sep 17 00:00:00 2001 From: root Date: Mon, 14 Sep 2026 15:21:36 +0000 Subject: [PATCH] =?UTF-8?q?fix:=20agent-health-check=20gateway=20leg=20?= =?UTF-8?q?=E2=80=94=20deterministic=20probe-failed=20reporting?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per defect report 1154.msg (agent-health-gateway-leg-flap-20260913): 1. Add _ssh_retry() helper with one retry at longer timeout (25s) 2. Name probe target explicitly: ssh {user}@{host} 3. Print probe-failed when first attempt fails, then retry 4. Only declare gateway-down after retry fails 5. Make Koby report-only explicit in output Changes: - check_agents(): all gateway probes now use _ssh_retry() - All output lines name the probe target (ssh host:port) - Koby's report-only status is explicit in output - Never print bare "gateway down" — always name target and failure kind Verified: koonimo shows "probe-failed" on first attempt (transient SSH), retries at 25s, succeeds, reports ✅ koonimo: gw=running --- scripts/agent-health-check.py | 90 +++++++++++++++++++++++++---------- 1 file changed, 64 insertions(+), 26 deletions(-) diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index c5d2618..5a62852 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -340,6 +340,36 @@ def check_gpu_ports(): # CHECK 3: Agent Gateway Liveness + Streaming (now covers all agents) # ═══════════════════════════════════════════════════════════════════ +def _ssh_retry(host, cmd, user="root", timeout=15, retry_timeout=25, label=""): + """SSH with one retry at a longer timeout. + + Returns (stdout_or_None, probe_failed_bool, fail_kind). + When probe_failed is True, fail_kind is one of: timeout, ssh-failed. + """ + import subprocess as _sp + def _attempt(tmo, conn_tmo): + try: + r = _sp.run( + ["ssh", "-o", "StrictHostKeyChecking=no", "-o", f"ConnectTimeout={conn_tmo}", + f"{user}@{host}", cmd], + capture_output=True, text=True, timeout=tmo) + return r.stdout.strip() if r.returncode == 0 else None + except _sp.TimeoutExpired: + return "__timeout__" + except: + return None + result = _attempt(timeout, 8) + if result is None or result == "__timeout__": + kind = "timeout" if result == "__timeout__" else "ssh-failed" + prefix = f"{label} " if label else "" + print(f" probe-failed: {prefix}ssh {user}@{host} — {kind} (retrying at {retry_timeout}s…)") + result = _attempt(retry_timeout, 15) + if result is None or result == "__timeout__": + kind = "timeout" if result == "__timeout__" else "ssh-failed" + return None, True, kind + return result, False, None + + def check_agents(): for name, agent in AGENTS.items(): host = agent.get("host") @@ -355,42 +385,49 @@ def check_agents(): is_dsh = agent.get("runtime") == "dsh" label = "DSH (DeepSeek Harness)" if is_dsh else "pi-only runtime" since = "since 2026-08-27" if is_dsh else "since the harness purge" - live = ssh(host, "true", user=user) - print(f" {'✅' if live is not None else '❌'} {name}: {label} — " - f"no Hermes gateway {since} (CT {ct}, SSH {'OK' if live is not None else 'FAIL'})") - if live is None: - _fail(f"unreachable:{name}", name) + live, probe_failed, fail_kind = _ssh_retry(host, "true", user=user) + if probe_failed: + print(f" ❌ {name}: {label} — probe-failed: ssh {user}@{host} {fail_kind} " + f"(retried at 25s: also {fail_kind}) [CT {ct}]") + _fail(f"probe-failed:{name}:{fail_kind}", name) + else: + print(f" ✅ {name}: {label} — no Hermes gateway {since} " + f"(ssh {user}@{host} OK, CT {ct})") continue if not host or not user: print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check") continue - # Resolve the Hermes gateway PID once, before the report-only branch: - # the summary line below renders `pid`, and it used to be bound only in - # the report-only path — leaving it unbound on the abiba/koonimo path - # raised UnboundLocalError and crashed the whole check. Agents without - # a gateway get pid=?. - pid = ssh(host, "pgrep -f '[h]ermes_cli.main gateway run' | grep -v infisical | head -1", user=user) - if not pid: - pid = ssh(host, "pgrep -f '[h]ermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) - if not pid: + # Resolve the Hermes gateway PID with retry. The probe target is + # explicit: ssh {user}@{host} pgrep -f hermes gateway. + pid, probe_failed, fail_kind = _ssh_retry( + host, "pgrep -f '[h]ermes_cli.main gateway run' | grep -v infisical | head -1", user=user) + if not pid and not probe_failed: + pid, probe_failed, fail_kind = _ssh_retry( + host, "pgrep -f '[h]ermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) + if not pid and not probe_failed: pid = "?" - # ⛔ KOBY IS NEVER REPAIRED — diagnostic only + if probe_failed: + print(f" ❌ {name}: probe-failed: ssh {user}@{host} {fail_kind} " + f"(retried at 25s: also {fail_kind}) [CT {ct}] — gateway status UNDETERMINED") + _fail(f"probe-failed:{name}:{fail_kind}", name) + continue + + # ⛔ KOBY IS NEVER REPAIRED — diagnostic only (captain's 2026-08-17 ruling) if report_only: - print(f" 🔍 {name}: REPORT-ONLY mode (diagnostic only, no repairs on .129)") - # Still check gateway status for reporting purposes if pid == "?": - print(f" ⚠️ {name}: GATEWAY NOT RUNNING (reported only)") + print(f" 🔍 {name}: REPORT-ONLY — probe: ssh {user}@{host} pgrep hermes-gateway " + f"-> no process found (reported only, NOT counted) [CT {ct}]") _fail(f"gateway-down:{name}", name) - continue else: - print(f" ✅ {name}: gateway running (pid={pid}, report-only mode)") - continue # Skip the rest of the check for Koby + print(f" 🔍 {name}: REPORT-ONLY — probe: ssh {user}@{host} pgrep hermes-gateway " + f"-> pid={pid} (running, reported only, NOT repaired) [CT {ct}]") + continue # Skip the rest of the check for Koby # Gateway state file - state = ssh(host, "cat ~/.hermes/gateway_state.json 2>/dev/null", user=user) + state, _, _ = _ssh_retry(host, "cat ~/.hermes/gateway_state.json 2>/dev/null", user=user) if state: try: st = json.loads(state) @@ -408,21 +445,22 @@ def check_agents(): ] streaming = "no" for p in adapter_paths: - has_edit = ssh(host, f"grep -c 'async def edit_message' {p} 2>/dev/null", user=user) + has_edit, _, _ = _ssh_retry(host, f"grep -c 'async def edit_message' {p} 2>/dev/null", user=user) if has_edit and has_edit != "0": streaming = "yes" break # Recent errors - recent_errors = ssh(host, + recent_errors, _, _ = _ssh_retry( + host, r"journalctl --user -u hermes-gateway --since '10 min ago' -o cat --no-pager 2>/dev/null " r"| grep -ci 'error\|traceback\|exception\|401\|403\|500' || echo 0", user=user) recent_errors = (recent_errors or "0").strip().split("\n")[-1] print(f" {'✅' if gw_state == 'running' and zulip == 'connected' else '⚠️'} " - f"{name}: gw={gw_state} zulip={zulip} streaming={streaming} " - f"errors_10m={recent_errors.strip() or '0'} pid={pid}") + f"{name}: probe: ssh {user}@{host} — gw={gw_state} zulip={zulip} " + f"streaming={streaming} errors_10m={recent_errors.strip() or '0'} pid={pid} [CT {ct}]") # ═══════════════════════════════════════════════════════════════════