From 641a52c6da8a570ccf15f286f77e360eab0cad08 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:14:08 +0000 Subject: [PATCH 1/6] fix(zulip-monitor): Add C3 public access path, make Result line non-optimistic - (b) Changed Result line to say 'INCIDENT' when ISSUES > 0, '0 issues (all healthy)' when ISSUES = 0 - (c) Documented C1 (no credential needed), C2 (requires LITELLM_KEY) distinction - (d) Added C3 public access path leg for https://kagentz.sysloggh.net/ - C3 treats 200/302/401 as alive, 502/000 as incident - Added tests/test_zulip_kagentz_legs.py to verify all changes --- scripts/zulip-monitor.sh | 49 +++++++++-- tests/test_zulip_kagentz_legs.py | 137 +++++++++++++++++++++++++++++++ zulip-health.prose.md | 29 +++++-- 3 files changed, 199 insertions(+), 16 deletions(-) create mode 100644 tests/test_zulip_kagentz_legs.py diff --git a/scripts/zulip-monitor.sh b/scripts/zulip-monitor.sh index c1423e3..05b6a42 100755 --- a/scripts/zulip-monitor.sh +++ b/scripts/zulip-monitor.sh @@ -176,6 +176,11 @@ fi # never contact her former host. # ── Platform C: Agent Zero (kagentz) ── +# C1: A2A liveness (no credential needed) — probes the container's internal :80/a2a/ +# C2: A2A response verification (needs LITELLM_KEY) — probes POST /a2a with auth +# C3: Public access path (no credential needed) — probes https://kagentz.sysloggh.net/ + +# C1: A2A liveness (container-internal probe) AZ_A2A_CODE=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \ "docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/ 2>/dev/null" 2>/dev/null) || AZ_A2A_CODE="000" AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]') @@ -184,27 +189,53 @@ AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]') if [ "$AZ_A2A_CODE" = "000" ]; then notify "🔴" "kagentz A2A server DOWN (connection failed)" ISSUES=$((ISSUES + 1)) - echo " kagentz: ❌ A2A down (HTTP 000)" >> "$LOG" + echo " kagentz C1: ❌ A2A down (HTTP 000)" >> "$LOG" else case "$AZ_A2A_CODE" in 200|401) - echo " kagentz: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;; + echo " kagentz C1: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;; *) notify "🟡" "kagentz A2A server answered HTTP $AZ_A2A_CODE — running, unexpected status" ISSUES=$((ISSUES + 1)) - echo " kagentz: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;; + echo " kagentz C1: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;; + esac +fi + +# C3: Public access path (the captain's point of view) +# Probes the public URL that NetBird proxies to the container. 200/302/401 = alive, +# 502 = proxy's "upstream refused" page (incident), connection failed = incident. +# Never restarts anything — the contract forbids restarting the platform. +KAGENTZ_PUBLIC_CODE=$(curl -s -o /dev/null --connect-timeout 10 --max-time 15 \ + -w '%{http_code}' https://kagentz.sysloggh.net/ 2>/dev/null) || KAGENTZ_PUBLIC_CODE="000" +KAGENTZ_PUBLIC_CODE=$(printf '%s' "$KAGENTZ_PUBLIC_CODE" | tr -d '[:space:]') +[ -n "$KAGENTZ_PUBLIC_CODE" ] || KAGENTZ_PUBLIC_CODE="000" + +if [ "$KAGENTZ_PUBLIC_CODE" = "000" ]; then + notify "🔴" "kagentz public URL DOWN (connection failed)" + ISSUES=$((ISSUES + 1)) + echo " kagentz C3: ❌ public URL down (HTTP 000)" >> "$LOG" +elif [ "$KAGENTZ_PUBLIC_CODE" = "502" ]; then + notify "🔴" "kagentz public URL 502 (upstream refused)" + ISSUES=$((ISSUES + 1)) + echo " kagentz C3: ❌ public URL 502 (upstream refused)" >> "$LOG" +else + case "$KAGENTZ_PUBLIC_CODE" in + 200|302|401) + echo " kagentz C3: ✅ public URL alive (HTTP $KAGENTZ_PUBLIC_CODE)" >> "$LOG" ;; + *) + notify "🟡" "kagentz public URL answered HTTP $KAGENTZ_PUBLIC_CODE — running, unexpected status" + ISSUES=$((ISSUES + 1)) + echo " kagentz C3: 🟡 public URL unexpected http=$KAGENTZ_PUBLIC_CODE (running, warning)" >> "$LOG" ;; esac fi # ── Summary ── +# The run verdict is non-optimistic: when ISSUES > 0, the run is an INCIDENT. +# The lane must quote this Result line verbatim in its status report. if [ "$ISSUES" -eq 0 ]; then - if [ "$ZULIP_CRED_OK" -eq 0 ]; then - echo " Result: ✅ All healthy" >> "$LOG" - else - echo " Result: ✅ All healthy" >> "$LOG" - fi + echo " Result: ✅ 0 issues (all healthy)" >> "$LOG" else - echo " Result: 🔴 $ISSUES issue(s) found" >> "$LOG" + echo " Result: 🔴 INCIDENT — $ISSUES issue(s) found" >> "$LOG" notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log" fi diff --git a/tests/test_zulip_kagentz_legs.py b/tests/test_zulip_kagentz_legs.py new file mode 100644 index 0000000..91a497f --- /dev/null +++ b/tests/test_zulip_kagentz_legs.py @@ -0,0 +1,137 @@ +#!/usr/bin/env python3 +""" +Tests for zulip-monitor.sh kagentz A2A and public access path legs. + +Covers: +- (b) Run verdict is non-optimistic: when ISSUES > 0, the Result line says "INCIDENT" +- (d) Public access path leg: 200/302/401 = alive, 502 = incident, 000 = incident +- (c) C1/C2/C3 distinction documented in prose + +These tests parse the script and prose to verify the expected structure. +""" + +import os +import re +import subprocess +import sys +from pathlib import Path + +# Paths +SCRIPT = Path(__file__).parent.parent / "scripts" / "zulip-monitor.sh" +PROSE = Path(__file__).parent.parent / "zulip-health.prose.md" + +def read_file(path): + return path.read_text() + +def test_script_has_c1_c2_c3_legs(): + """Script should have C1, C2, C3 leg markers.""" + content = read_file(SCRIPT) + assert "C1: A2A liveness" in content, "Missing C1 leg comment" + assert "C3: Public access path" in content, "Missing C3 leg comment" + print("✓ Script has C1, C2, C3 leg markers") + +def test_c1_no_credential_needed(): + """C1 should be documented as needing no credential.""" + prose = read_file(PROSE) + assert "C1: A2A Server Health (no credential needed)" in prose, \ + "C1 header should say 'no credential needed'" + assert "INCIDENT" in prose, "C1 000 should be marked as INCIDENT" + print("✓ C1 documented as no-credential, 000 = INCIDENT") + +def test_c2_requires_litellm_key(): + """C2 should be documented as requiring LITELLM_KEY.""" + prose = read_file(PROSE) + assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose, \ + "C2 header should say 'requires LITELLM_KEY'" + assert "credential issue, NOT a server-down incident" in prose, \ + "C2 401 should be documented as credential issue, not server-down" + print("✓ C2 documented as requiring LITELLM_KEY") + +def test_c3_public_access_path(): + """C3 should probe https://kagentz.sysloggh.net/.""" + prose = read_file(PROSE) + script = read_file(SCRIPT) + assert "C3: Public Access Path" in prose, "Missing C3 section in prose" + assert "https://kagentz.sysloggh.net/" in prose, "C3 should probe the public URL" + assert "502" in prose, "C3 should document 502 as incident" + assert "KAGENTZ_PUBLIC_CODE" in script, "Script should have KAGENTZ_PUBLIC_CODE variable" + print("✓ C3 public access path leg present") + +def test_result_line_non_optimistic(): + """When ISSUES > 0, the Result line should say INCIDENT, not just 'issues found'.""" + script = read_file(SCRIPT) + # The summary section should have "INCIDENT" in the non-zero branch + assert "Result: 🔴 INCIDENT" in script, \ + "Result line should say 'INCIDENT' when ISSUES > 0" + assert "Result: ✅ 0 issues (all healthy)" in script, \ + "Result line should say '0 issues (all healthy)' when ISSUES = 0" + print("✓ Result line is non-optimistic (INCIDENT when issues > 0)") + +def test_c3_502_is_incident(): + """C3 should treat 502 as an incident.""" + script = read_file(SCRIPT) + # The script should have a branch for 502 + assert 'elif [ "$KAGENTZ_PUBLIC_CODE" = "502" ]' in script, \ + "Script should have explicit 502 branch" + assert "upstream refused" in script, \ + "502 should be documented as 'upstream refused'" + print("✓ C3 502 treated as incident") + +def test_c3_000_is_incident(): + """C3 should treat 000 as an incident.""" + script = read_file(SCRIPT) + assert 'if [ "$KAGENTZ_PUBLIC_CODE" = "000" ]' in script, \ + "Script should have explicit 000 branch" + assert "public URL DOWN" in script, \ + "000 should be reported as 'public URL DOWN'" + print("✓ C3 000 treated as incident") + +def test_c3_alive_statuses(): + """C3 should treat 200/302/401 as alive.""" + script = read_file(SCRIPT) + assert "200|302|401" in script, \ + "Script should classify 200/302/401 as alive" + assert "public URL alive" in script, \ + "Alive statuses should be reported as 'public URL alive'" + print("✓ C3 200/302/401 treated as alive") + +def test_prose_documents_c3_actions(): + """Prose should document C3 actions in the Platform C Actions table.""" + prose = read_file(PROSE) + assert "C3 public URL returns `502`" in prose, \ + "Platform C Actions table should have C3 502 row" + assert "C3 public URL returns `000`" in prose, \ + "Platform C Actions table should have C3 000 row" + print("✓ Prose documents C3 actions") + +def run_all_tests(): + tests = [ + test_script_has_c1_c2_c3_legs, + test_c1_no_credential_needed, + test_c2_requires_litellm_key, + test_c3_public_access_path, + test_result_line_non_optimistic, + test_c3_502_is_incident, + test_c3_000_is_incident, + test_c3_alive_statuses, + test_prose_documents_c3_actions, + ] + passed = 0 + failed = 0 + for test in tests: + try: + test() + passed += 1 + except AssertionError as e: + print(f"✗ {test.__name__}: {e}") + failed += 1 + print(f"\n{'='*50}") + print(f"Tests passed: {passed}/{len(tests)}") + if failed > 0: + print(f"Tests failed: {failed}/{len(tests)}") + sys.exit(1) + else: + print("All tests passed!") + +if __name__ == "__main__": + run_all_tests() diff --git a/zulip-health.prose.md b/zulip-health.prose.md index 54229f8..097e2ec 100644 --- a/zulip-health.prose.md +++ b/zulip-health.prose.md @@ -471,7 +471,7 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null > heartbeat/queue, or adapter-restart step. Agent Zero is probed for A2A > liveness only, and a probe must never restart a platform. -**C1: A2A Server Health** +**C1: A2A Server Health (no credential needed)** ```bash # A2A listens on :80 inside the agent-zero container (host-mapped to :50080) and @@ -479,9 +479,9 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null ssh root@192.168.68.14 "docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/" ``` -Expected: `401` (auth-gated, A2A server is up and responding) or `200` (if no auth required). Connection refused (`000`) → A2A server down. Any other status → running but unexpected: log/report it, never restart. +Expected: `401` (auth-gated, A2A server is up and responding) or `200` (if no auth required). Connection refused (`000`) → A2A server down (INCIDENT). Any other status → running but unexpected: log/report it, never restart. -**C2: A2A Response Verification** +**C2: A2A Response Verification (requires LITELLM_KEY)** ```bash # A2A listens on :80 inside the container and is auth-gated (401 expected unauthenticated). @@ -491,15 +491,30 @@ ssh root@192.168.68.14 "docker exec agent-zero curl -s -X POST http://127.0.0.1: -d '{\"jsonrpc\":\"2.0\",\"method\":\"tasks/send\",\"params\":{\"message\":{\"role\":\"user\",\"parts\":[{\"text\":\"ping\"}]}},\"id\":1}'" ``` -Expected: task ID with "working" status. Poll for completion with `tasks/get`. If 401, check LITELLM_KEY is set. +Expected: task ID with "working" status. Poll for completion with `tasks/get`. If 401, check LITELLM_KEY is set (this is a credential issue, NOT a server-down incident). + +**C3: Public Access Path (no credential needed)** + +```bash +# Probes the public URL that NetBird proxies to the agent-zero container. +# This is the captain's point of view: if the captain can't reach it, it's down. +# 200/302/401 = alive, 502 = proxy's "upstream refused" page (INCIDENT), +# connection failed (000) = INCIDENT. Never restarts anything. +curl -s -o /dev/null --connect-timeout 10 --max-time 15 -w '%{http_code}' https://kagentz.sysloggh.net/ +``` + +Expected: `200` (Agent Zero login page), `302` (redirect), or `401` (auth-gated) = alive. `502` = NetBird proxy's "upstream refused" page (the container's port 80 is not listening) = INCIDENT. `000` (connection failed) = INCIDENT. Any other status → running but unexpected: log/report it, never restart. **Platform C Actions** | Condition | Action | |-----------|--------| -| A2A returns `000` (connection refused/timeout) | Alert only — never restart the platform; investigate the agent-zero container | -| A2A returns a status other than `200`/`401` | Log/report as a warning — reported, never healed on | -| LiteLLM 401 | Check API key in a2a_agent.py `LITELLM_KEY` | +| C1 A2A returns `000` (connection refused/timeout) | Alert only — never restart the platform; investigate the agent-zero container | +| C1 A2A returns a status other than `200`/`401` | Log/report as a warning — reported, never healed on | +| C2 A2A returns 401 | Check `LITELLM_KEY` is set (credential issue, NOT server-down) | +| C3 public URL returns `502` (upstream refused) | Alert only — the container's port 80 is not listening; investigate the agent-zero container | +| C3 public URL returns `000` (connection failed) | Alert only — the public path is down; investigate the NetBird proxy or the container | +| C3 public URL returns a status other than `200`/`302`/`401` | Log/report as a warning — reported, never healed on | ### Step 5: Global Checks -- 2.54.0 From 568fec2efaf904e5e5c984c6e46efbcb11ecfae2 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:25:03 +0000 Subject: [PATCH 2/6] test(zulip-kagentz): Replace string-presence tests with behavioural sandbox tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - C3 502 → INCIDENT - C3 000 → INCIDENT - C1 401 + C3 302 → 0 issues, all healthy - C1 000 → INCIDENT Each test asserts from the run's own log/verdict, not from file text. Prose assertions kept as secondary. Proven to bite: run against pre-fix script (origin/master) shows all 4 behavioural cases fail because C3 leg doesn't exist and Result line doesn't say 'INCIDENT'. --- tests/test_zulip_kagentz_legs.py | 330 ++++++++++++++++++++----------- 1 file changed, 213 insertions(+), 117 deletions(-) diff --git a/tests/test_zulip_kagentz_legs.py b/tests/test_zulip_kagentz_legs.py index 91a497f..0254a86 100644 --- a/tests/test_zulip_kagentz_legs.py +++ b/tests/test_zulip_kagentz_legs.py @@ -1,137 +1,233 @@ #!/usr/bin/env python3 -""" -Tests for zulip-monitor.sh kagentz A2A and public access path legs. +"""Behavioural tests for zulip-monitor.sh kagentz C1/C3 legs and the Result verdict. -Covers: -- (b) Run verdict is non-optimistic: when ISSUES > 0, the Result line says "INCIDENT" -- (d) Public access path leg: 200/302/401 = alive, 502 = incident, 000 = incident -- (c) C1/C2/C3 distinction documented in prose +WHY THIS FILE EXISTS: the 2026-09-19 kagentz A2A outage was correctly detected +by the monitor (C1 returned 000, the run logged an issue) but the lane's own +summarization in ops.status wrote "OK" with the note "A2A server DOWN — expected +(no credentials configured)". The optimistic verdict came from the lane, not the +script. The fix adds a C3 public-access-path leg and makes the Result line say +"INCIDENT" when issues are found, so the lane can quote it verbatim. -These tests parse the script and prose to verify the expected structure. +CONTRACT UNDER TEST: + * C1 (A2A liveness, no credential): 000 → INCIDENT. + * C3 (public access path, no credential): 502 → INCIDENT, 000 → INCIDENT, + 200/302/401 → alive. + * Result verdict: when ISSUES > 0, the log's Result line says "INCIDENT", + not just "issues found". + * Healthy control: C1 401 + C3 302 → 0 issues, "all healthy". + +HOW: behavioural execution using the sandbox pattern already in this repo +(tests/test_mumuni_monitor_removal.py). The sandbox copies the shipped monitor +verbatim, rewrites only its LOG constant, and runs it with stub ssh/curl on +PATH. Each test asserts from the run's own log/verdict, not from file text. + +Usage: python3 -m pytest tests/test_zulip_kagentz_legs.py """ +from __future__ import annotations import os -import re +import pathlib +import stat import subprocess -import sys -from pathlib import Path -# Paths -SCRIPT = Path(__file__).parent.parent / "scripts" / "zulip-monitor.sh" -PROSE = Path(__file__).parent.parent / "zulip-health.prose.md" +import pytest -def read_file(path): - return path.read_text() +ROOT = pathlib.Path(__file__).resolve().parents[1] +ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh" +CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json" -def test_script_has_c1_c2_c3_legs(): - """Script should have C1, C2, C3 leg markers.""" - content = read_file(SCRIPT) - assert "C1: A2A liveness" in content, "Missing C1 leg comment" - assert "C3: Public access path" in content, "Missing C3 leg comment" - print("✓ Script has C1, C2, C3 leg markers") +TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec +AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker -def test_c1_no_credential_needed(): - """C1 should be documented as needing no credential.""" - prose = read_file(PROSE) - assert "C1: A2A Server Health (no credential needed)" in prose, \ - "C1 header should say 'no credential needed'" - assert "INCIDENT" in prose, "C1 000 should be marked as INCIDENT" - print("✓ C1 documented as no-credential, 000 = INCIDENT") -def test_c2_requires_litellm_key(): - """C2 should be documented as requiring LITELLM_KEY.""" - prose = read_file(PROSE) - assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose, \ - "C2 header should say 'requires LITELLM_KEY'" - assert "credential issue, NOT a server-down incident" in prose, \ - "C2 401 should be documented as credential issue, not server-down" - print("✓ C2 documented as requiring LITELLM_KEY") +# ── Stub ssh: answers Tanko and Agent Zero probes by env vars ────────── -def test_c3_public_access_path(): - """C3 should probe https://kagentz.sysloggh.net/.""" - prose = read_file(PROSE) - script = read_file(SCRIPT) - assert "C3: Public Access Path" in prose, "Missing C3 section in prose" - assert "https://kagentz.sysloggh.net/" in prose, "C3 should probe the public URL" - assert "502" in prose, "C3 should document 502 as incident" - assert "KAGENTZ_PUBLIC_CODE" in script, "Script should have KAGENTZ_PUBLIC_CODE variable" - print("✓ C3 public access path leg present") +SSH_STUB = r"""#!/usr/bin/env bash +# Stub ssh: record the target host, then answer by host + remote command. +printf '%s\n' "$*" >> "$RECORD_DIR/ssh.calls" +host="" +for a in "$@"; do + case "$a" in + *@192.168.*) host="${a##*@}" ;; + esac +done +printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts" +cmd="${*: -1}" +case "$host" in + 192.168.68.15) + case "$cmd" in + *"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;; + *curl*) printf '%s' "$TANKO_HTTP" ;; + esac ;; + 192.168.68.14) + case "$cmd" in + *"/a2a/"*) printf '%s' "$AZ_A2A_CODE"; exit "${AZ_A2A_EXIT:-0}" ;; + esac ;; + *) + printf 'UNEXPECTED-SSH-HOST %s\n' "$host" >> "$RECORD_DIR/unexpected-ssh" ;; +esac +exit 0 +""" -def test_result_line_non_optimistic(): - """When ISSUES > 0, the Result line should say INCIDENT, not just 'issues found'.""" - script = read_file(SCRIPT) - # The summary section should have "INCIDENT" in the non-zero branch - assert "Result: 🔴 INCIDENT" in script, \ - "Result line should say 'INCIDENT' when ISSUES > 0" - assert "Result: ✅ 0 issues (all healthy)" in script, \ - "Result line should say '0 issues (all healthy)' when ISSUES = 0" - print("✓ Result line is non-optimistic (INCIDENT when issues > 0)") -def test_c3_502_is_incident(): - """C3 should treat 502 as an incident.""" - script = read_file(SCRIPT) - # The script should have a branch for 502 - assert 'elif [ "$KAGENTZ_PUBLIC_CODE" = "502" ]' in script, \ - "Script should have explicit 502 branch" - assert "upstream refused" in script, \ - "502 should be documented as 'upstream refused'" - print("✓ C3 502 treated as incident") +# ── Stub curl: serves Zulip server, Abiba health, and C3 public URL ─── -def test_c3_000_is_incident(): - """C3 should treat 000 as an incident.""" - script = read_file(SCRIPT) - assert 'if [ "$KAGENTZ_PUBLIC_CODE" = "000" ]' in script, \ - "Script should have explicit 000 branch" - assert "public URL DOWN" in script, \ - "000 should be reported as 'public URL DOWN'" - print("✓ C3 000 treated as incident") +CURL_STUB = r"""#!/usr/bin/env bash +# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the +# C3 public URL probe (https://kagentz.sysloggh.net/). Record every call. +printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls" +case "$*" in + *:9200/health*) + case " $* " in + *" -w "*) printf '%s' "$PI_HTTP" ;; # -w '%{http_code}' probe + *) printf '%s' "$PI_BODY" ;; # body probe + esac ;; + *server_settings*) + printf '%s' "$SERVER_HTTP" ;; + *kagentz.sysloggh.net*) + printf '%s' "$KAGENTZ_PUBLIC_CODE" ;; +esac +exit 0 +""" -def test_c3_alive_statuses(): - """C3 should treat 200/302/401 as alive.""" - script = read_file(SCRIPT) - assert "200|302|401" in script, \ - "Script should classify 200/302/401 as alive" - assert "public URL alive" in script, \ - "Alive statuses should be reported as 'public URL alive'" - print("✓ C3 200/302/401 treated as alive") -def test_prose_documents_c3_actions(): - """Prose should document C3 actions in the Platform C Actions table.""" - prose = read_file(PROSE) - assert "C3 public URL returns `502`" in prose, \ - "Platform C Actions table should have C3 502 row" - assert "C3 public URL returns `000`" in prose, \ - "Platform C Actions table should have C3 000 row" - print("✓ Prose documents C3 actions") +def _write_exec(path: pathlib.Path, body: str) -> None: + path.write_text(body) + path.chmod(path.stat().st_mode + | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) -def run_all_tests(): - tests = [ - test_script_has_c1_c2_c3_legs, - test_c1_no_credential_needed, - test_c2_requires_litellm_key, - test_c3_public_access_path, - test_result_line_non_optimistic, - test_c3_502_is_incident, - test_c3_000_is_incident, - test_c3_alive_statuses, - test_prose_documents_c3_actions, - ] - passed = 0 - failed = 0 - for test in tests: - try: - test() - passed += 1 - except AssertionError as e: - print(f"✗ {test.__name__}: {e}") - failed += 1 - print(f"\n{'='*50}") - print(f"Tests passed: {passed}/{len(tests)}") - if failed > 0: - print(f"Tests failed: {failed}/{len(tests)}") - sys.exit(1) - else: - print("All tests passed!") -if __name__ == "__main__": - run_all_tests() +def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", + az_a2a_code="401", az_a2a_exit=0, + kagentz_public_code="302"): + """Run the shipped monitor in a sandbox; return (proc, record_dir, log_path). + + Only the LOG constant is rewritten (to keep the run inside the worktree). + Everything else — legs, labels, notify logic — is the shipped script. + """ + sandbox = tmp_path / "sandbox" + bindir = sandbox / "bin" + record = sandbox / "record" + bindir.mkdir(parents=True) + record.mkdir() + + _write_exec(bindir / "ssh", SSH_STUB) + _write_exec(bindir / "curl", CURL_STUB) + + source = ZULIP_MONITOR.read_text() + log_line = 'LOG="/root/zulip-health-monitor.log"' + assert log_line in source, "LOG constant moved — update the sandbox harness" + log_path = sandbox / "zulip-health-monitor.log" + script = sandbox / "zulip-monitor.sh" + script.write_text(source.replace(log_line, f'LOG="{log_path}"')) + + env = dict(os.environ) + env.update({ + "PATH": f"{bindir}:{env['PATH']}", + "RECORD_DIR": str(record), + "TANKO_SVC": tanko_svc, + "TANKO_HTTP": tanko_http, + "AZ_A2A_CODE": az_a2a_code, + "AZ_A2A_EXIT": str(az_a2a_exit), + "PI_HTTP": "200", + "PI_BODY": CONNECTED_FIXTURE.read_text(), + "SERVER_HTTP": "200", + "KAGENTZ_PUBLIC_CODE": kagentz_public_code, + }) + proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env, + capture_output=True, text=True) + return proc, record, log_path + + +# ── Required behavioural cases ───────────────────────────────────────── + +def test_c3_502_is_incident(tmp_path): + """C3 public leg returns 502 → the run's verdict is an INCIDENT, + the C3 line names the 502, and the run is not summarised as healthy.""" + proc, record, log_path = _run_monitor(tmp_path, kagentz_public_code="502") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # The C3 line names the 502. + assert "kagentz C3: ❌ public URL 502 (upstream refused)" in log + # The verdict is an INCIDENT, not healthy. + assert "Result: 🔴 INCIDENT" in log + assert "all healthy" not in log + # The run is not summarised as healthy. + assert "✅ 0 issues" not in log + + +def test_c3_000_is_incident(tmp_path): + """C3 public leg returns 000 → INCIDENT.""" + proc, record, log_path = _run_monitor(tmp_path, kagentz_public_code="000") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # The C3 line reports the connection failure. + assert "kagentz C3: ❌ public URL down (HTTP 000)" in log + # The verdict is an INCIDENT. + assert "Result: 🔴 INCIDENT" in log + assert "all healthy" not in log + assert "✅ 0 issues" not in log + + +def test_healthy_control_c1_401_c3_302(tmp_path): + """Healthy control: C1 401 plus C3 302 → 0 issues and a healthy verdict, + proving the new leg cannot cry wolf.""" + proc, record, log_path = _run_monitor(tmp_path, + az_a2a_code="401", + kagentz_public_code="302") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # Both legs report alive. + assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log + assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log + # Zero issues, healthy verdict. + assert "Result: ✅ 0 issues (all healthy)" in log + # No INCIDENT. + assert "INCIDENT" not in log + # No notify fired for kagentz. + assert "kagentz public URL" not in proc.stdout + assert "kagentz A2A server" not in proc.stdout + + +def test_c1_000_is_incident(tmp_path): + """C1 returns 000 → INCIDENT, keeping the leg that actually caught this + outage covered behaviourally.""" + proc, record, log_path = _run_monitor(tmp_path, + az_a2a_code="000", + az_a2a_exit=7, + kagentz_public_code="302") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # The C1 line reports the A2A down. + assert "kagentz C1: ❌ A2A down (HTTP 000)" in log + # The verdict is an INCIDENT (even though C3 is healthy). + assert "Result: 🔴 INCIDENT" in log + assert "all healthy" not in log + # The notify fired for the A2A down. + assert "kagentz A2A server DOWN" in proc.stdout + + +# ── Prose assertions (kept as secondary, do not replace behavioural) ── + +def test_prose_c1_no_credential_needed(): + """C1 header should say 'no credential needed'.""" + prose = (ROOT / "zulip-health.prose.md").read_text() + assert "C1: A2A Server Health (no credential needed)" in prose + + +def test_prose_c2_requires_litellm_key(): + """C2 header should say 'requires LITELLM_KEY'.""" + prose = (ROOT / "zulip-health.prose.md").read_text() + assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose + + +def test_prose_c3_public_access_path(): + """C3 section should exist and document 502/000 as incidents.""" + prose = (ROOT / "zulip-health.prose.md").read_text() + assert "C3: Public Access Path" in prose + assert "https://kagentz.sysloggh.net/" in prose + assert "502" in prose -- 2.54.0 From 7a5ddb46a9692346e7398e94e3f6246943e8547c Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:27:19 +0000 Subject: [PATCH 3/6] chore: Add local monitoring tooling scripts --- scripts/daily-health-digest.sh | 71 ++++++++++++++++++++ scripts/gpu-monitor.sh | 118 +++++++++++++++++++++++++++++++++ 2 files changed, 189 insertions(+) create mode 100755 scripts/daily-health-digest.sh create mode 100755 scripts/gpu-monitor.sh diff --git a/scripts/daily-health-digest.sh b/scripts/daily-health-digest.sh new file mode 100755 index 0000000..7663a35 --- /dev/null +++ b/scripts/daily-health-digest.sh @@ -0,0 +1,71 @@ +#!/bin/bash +# daily-health-digest.sh — Daily Fleet Health Digest +# Runs all major health checks and summarizes results +# +# Legs: +# - infrastructure-monitoring (13 legs) +# - gpu-monitor (6 legs) +# - litellm-health (11 checks) +# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys) +# - proxmox-monitor (4 exporters) +# - pm2-self-heal (PM2 services) +# - zulip-health (Zulip mesh) +# +# Exit 0 if all pass, 1 if any fail. +# +# Run: bash scripts/daily-health-digest.sh + +set -uo pipefail + +TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') +SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)" +FAILED=() + +echo "=== Daily Health Digest — $TIMESTAMP ===" +echo "Executed from: $(pwd -P)" +echo "" + +# Run each health check and capture exit code +run_check() { + local name="$1" + local script="$2" + local runner="${3:-bash}" + local output + + echo "── Running $name ──" + output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1) + local exit_code=$? + + if [ $exit_code -eq 0 ]; then + # Extract just the summary lines (lines with ✅ or "All") + echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3 + echo "" + else + echo " ❌ $name failed (exit $exit_code)" + echo "$output" | grep -E "🔴|❌" | tail -5 + echo "" + FAILED+=("$name") + fi +} + +# Run all checks +run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh" +run_check "GPU Monitor" "scripts/gpu-monitor.sh" +run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3" +run_check "Agent Health Check" "scripts/agent-health-check.py" "python3" +run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh" +run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh" +run_check "Zulip Health" "scripts/zulip-monitor.sh" + +# Summary +echo "─────────────────────────────────────────────" +if [ ${#FAILED[@]} -eq 0 ]; then + echo "✅ ALL HEALTH CHECKS PASSED" + exit 0 +else + echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:" + for f in "${FAILED[@]}"; do + echo " - $f" + done + exit 1 +fi \ No newline at end of file diff --git a/scripts/gpu-monitor.sh b/scripts/gpu-monitor.sh new file mode 100755 index 0000000..b7976d9 --- /dev/null +++ b/scripts/gpu-monitor.sh @@ -0,0 +1,118 @@ +#!/bin/bash +# gpu-monitor.sh — GPU Fleet Health Monitor +# Implements gpu-monitor.prose.md (check-health section) +# +# Legs: +# - GPU Monitor health (localhost:9100/health) +# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080) +# - IMPORTANT: Always probe :8080, NEVER bare :80 +# - Router unified health (192.168.68.116/health/unified, expects 301) +# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200) +# +# Exit 0 if all probes pass, 1 if any fails. +# +# Run: bash scripts/gpu-monitor.sh + +set -uo pipefail + +CT116_HOST="192.168.68.116" +CT8_HOST="192.168.68.8" +CT110_HOST="192.168.68.110" +CT15_HOST="192.168.68.15" +FAILED=() +TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') + +echo "=== GPU Monitor — $TIMESTAMP ===" +echo "Executed from: $(pwd -P)" +echo "" + +# 1. GPU Monitor health +echo " Probing GPU Monitor health..." +GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null) +if [ $? -ne 0 ]; then + echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)" + FAILED+=("gpu-monitor") +elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then + CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ') + echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)" +else + echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)" + FAILED+=("gpu-monitor") +fi + +# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80) +echo " Probing GPU host health..." + +# CT 8 (.8) +RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null) +RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]') +[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000" + +if [ "$RTX3090_CODE" = "200" ]; then + echo " ✅ GPU .8 (RTX 3090): healthy" +else + echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})" + FAILED+=("gpu-8") +fi + +# CT 110 (.110) +RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null) +RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]') +[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000" + +if [ "$RTX5070_CODE" = "200" ]; then + echo " ✅ GPU .110 (RTX 5070): healthy" +else + echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})" + FAILED+=("gpu-110") +fi + +# CT 15 (.15) - Strix Halo +STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null) +STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]') +[ -n "$STRIX_CODE" ] || STRIX_CODE="000" + +if [ "$STRIX_CODE" = "200" ]; then + echo " ✅ GPU .15 (Strix Halo): healthy" +else + echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})" + FAILED+=("gpu-15") +fi + +# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY) +echo " Probing Router unified health..." +ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null) +ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]') +[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000" + +if [ "$ROUTER_CODE" = "301" ]; then + echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)" +else + echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})" + FAILED+=("router") +fi + +# 4. LiteLLM health +echo " Probing LiteLLM health..." +LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null) +LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]') +[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000" + +if [ "$LITELLM_CODE" = "200" ]; then + echo " ✅ LiteLLM: alive" +else + echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})" + FAILED+=("litellm") +fi + +# ── Summary ───────────────────────────────────────────────────────────────── +echo "" +if [ ${#FAILED[@]} -eq 0 ]; then + echo " ✅ All legs OK" + exit 0 +else + for f in "${FAILED[@]}"; do + echo " 🔴 FAILED: $f" + done + exit 1 +fi \ No newline at end of file -- 2.54.0 From 63990b84f75f905f3816c051d873b6491606ea8c Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:31:25 +0000 Subject: [PATCH 4/6] no-mistakes(review): Fix review findings: test regression, contract verdict, scope trim --- scripts/daily-health-digest.sh | 71 ---------------- scripts/gpu-monitor.sh | 118 --------------------------- tests/test_mumuni_monitor_removal.py | 31 ++++--- tests/test_zulip_kagentz_legs.py | 24 ------ zulip-health.prose.md | 19 +++-- 5 files changed, 31 insertions(+), 232 deletions(-) delete mode 100755 scripts/daily-health-digest.sh delete mode 100755 scripts/gpu-monitor.sh diff --git a/scripts/daily-health-digest.sh b/scripts/daily-health-digest.sh deleted file mode 100755 index 7663a35..0000000 --- a/scripts/daily-health-digest.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/bin/bash -# daily-health-digest.sh — Daily Fleet Health Digest -# Runs all major health checks and summarizes results -# -# Legs: -# - infrastructure-monitoring (13 legs) -# - gpu-monitor (6 legs) -# - litellm-health (11 checks) -# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys) -# - proxmox-monitor (4 exporters) -# - pm2-self-heal (PM2 services) -# - zulip-health (Zulip mesh) -# -# Exit 0 if all pass, 1 if any fail. -# -# Run: bash scripts/daily-health-digest.sh - -set -uo pipefail - -TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') -SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)" -FAILED=() - -echo "=== Daily Health Digest — $TIMESTAMP ===" -echo "Executed from: $(pwd -P)" -echo "" - -# Run each health check and capture exit code -run_check() { - local name="$1" - local script="$2" - local runner="${3:-bash}" - local output - - echo "── Running $name ──" - output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1) - local exit_code=$? - - if [ $exit_code -eq 0 ]; then - # Extract just the summary lines (lines with ✅ or "All") - echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3 - echo "" - else - echo " ❌ $name failed (exit $exit_code)" - echo "$output" | grep -E "🔴|❌" | tail -5 - echo "" - FAILED+=("$name") - fi -} - -# Run all checks -run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh" -run_check "GPU Monitor" "scripts/gpu-monitor.sh" -run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3" -run_check "Agent Health Check" "scripts/agent-health-check.py" "python3" -run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh" -run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh" -run_check "Zulip Health" "scripts/zulip-monitor.sh" - -# Summary -echo "─────────────────────────────────────────────" -if [ ${#FAILED[@]} -eq 0 ]; then - echo "✅ ALL HEALTH CHECKS PASSED" - exit 0 -else - echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:" - for f in "${FAILED[@]}"; do - echo " - $f" - done - exit 1 -fi \ No newline at end of file diff --git a/scripts/gpu-monitor.sh b/scripts/gpu-monitor.sh deleted file mode 100755 index b7976d9..0000000 --- a/scripts/gpu-monitor.sh +++ /dev/null @@ -1,118 +0,0 @@ -#!/bin/bash -# gpu-monitor.sh — GPU Fleet Health Monitor -# Implements gpu-monitor.prose.md (check-health section) -# -# Legs: -# - GPU Monitor health (localhost:9100/health) -# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080) -# - IMPORTANT: Always probe :8080, NEVER bare :80 -# - Router unified health (192.168.68.116/health/unified, expects 301) -# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200) -# -# Exit 0 if all probes pass, 1 if any fails. -# -# Run: bash scripts/gpu-monitor.sh - -set -uo pipefail - -CT116_HOST="192.168.68.116" -CT8_HOST="192.168.68.8" -CT110_HOST="192.168.68.110" -CT15_HOST="192.168.68.15" -FAILED=() -TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') - -echo "=== GPU Monitor — $TIMESTAMP ===" -echo "Executed from: $(pwd -P)" -echo "" - -# 1. GPU Monitor health -echo " Probing GPU Monitor health..." -GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null) -if [ $? -ne 0 ]; then - echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)" - FAILED+=("gpu-monitor") -elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then - CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ') - echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)" -else - echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)" - FAILED+=("gpu-monitor") -fi - -# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80) -echo " Probing GPU host health..." - -# CT 8 (.8) -RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null) -RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]') -[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000" - -if [ "$RTX3090_CODE" = "200" ]; then - echo " ✅ GPU .8 (RTX 3090): healthy" -else - echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})" - FAILED+=("gpu-8") -fi - -# CT 110 (.110) -RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null) -RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]') -[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000" - -if [ "$RTX5070_CODE" = "200" ]; then - echo " ✅ GPU .110 (RTX 5070): healthy" -else - echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})" - FAILED+=("gpu-110") -fi - -# CT 15 (.15) - Strix Halo -STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null) -STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]') -[ -n "$STRIX_CODE" ] || STRIX_CODE="000" - -if [ "$STRIX_CODE" = "200" ]; then - echo " ✅ GPU .15 (Strix Halo): healthy" -else - echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})" - FAILED+=("gpu-15") -fi - -# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY) -echo " Probing Router unified health..." -ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null) -ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]') -[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000" - -if [ "$ROUTER_CODE" = "301" ]; then - echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)" -else - echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})" - FAILED+=("router") -fi - -# 4. LiteLLM health -echo " Probing LiteLLM health..." -LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null) -LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]') -[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000" - -if [ "$LITELLM_CODE" = "200" ]; then - echo " ✅ LiteLLM: alive" -else - echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})" - FAILED+=("litellm") -fi - -# ── Summary ───────────────────────────────────────────────────────────────── -echo "" -if [ ${#FAILED[@]} -eq 0 ]; then - echo " ✅ All legs OK" - exit 0 -else - for f in "${FAILED[@]}"; do - echo " 🔴 FAILED: $f" - done - exit 1 -fi \ No newline at end of file diff --git a/tests/test_mumuni_monitor_removal.py b/tests/test_mumuni_monitor_removal.py index df55352..6e765af 100644 --- a/tests/test_mumuni_monitor_removal.py +++ b/tests/test_mumuni_monitor_removal.py @@ -98,8 +98,8 @@ exit 0 """ CURL_STUB = r"""#!/usr/bin/env bash -# Stub curl: serve the Abiba health fixture and the Zulip server 200, and -# record every call (including notify) payloads. +# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the +# kagentz C3 public URL, and record every call (including notify) payloads. printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls" case "$*" in *:9200/health*) @@ -109,6 +109,8 @@ case "$*" in esac ;; *server_settings*) printf '%s' "$SERVER_HTTP" ;; + *kagentz.sysloggh.net*) + printf '%s' "$KAGENTZ_PUBLIC_CODE" ;; esac exit 0 """ @@ -121,7 +123,8 @@ def _write_exec(path: pathlib.Path, body: str) -> None: def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", - az_a2a_code="401", az_a2a_exit=0): + az_a2a_code="401", az_a2a_exit=0, + kagentz_public_code="302"): """Run the shipped monitor in a sandbox; return (proc, record_dir, log_path). Only the LOG constant is rewritten (to keep the run inside the worktree). @@ -154,6 +157,7 @@ def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", "PI_HTTP": "200", "PI_BODY": CONNECTED_FIXTURE.read_text(), "SERVER_HTTP": "200", + "KAGENTZ_PUBLIC_CODE": kagentz_public_code, }) proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env, capture_output=True, text=True) @@ -169,8 +173,9 @@ def test_healthy_run_is_quiet_and_never_reaches_mumuni(tmp_path): assert "Server: ✅ HTTP 200" in log assert "Abiba: ✅ Connected" in log assert "Tanko: ✅ service=active http=200" in log - assert "kagentz: ✅ A2A alive (HTTP 401)" in log - assert "Result: ✅ All healthy" in log + assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log + assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log + assert "Result: ✅ 0 issues (all healthy)" in log # A healthy run emits no notify at all — and certainly no Mumuni one. assert proc.stdout == "" @@ -207,8 +212,8 @@ def test_failing_run_alerts_on_tanko_but_never_on_mumuni(tmp_path): # The rest of the monitor still ran alongside the failing Tanko leg. log = log_path.read_text() assert "Abiba: ✅ Connected" in log - assert "kagentz: ✅ A2A alive" in log - assert "Result: 🔴 1 issue(s) found" in log + assert "kagentz C1: ✅ A2A alive" in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path): @@ -216,9 +221,9 @@ def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path): assert proc.returncode == 0, proc.stderr log = log_path.read_text() - assert "kagentz: 🟡 A2A unexpected http=500 (running, warning)" in log - assert "kagentz: ✅ A2A alive" not in log - assert "Result: 🔴 1 issue(s) found" in log + assert "kagentz C1: 🟡 A2A unexpected http=500 (running, warning)" in log + assert "kagentz C1: ✅ A2A alive" not in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log assert "kagentz A2A server answered HTTP 500" in proc.stdout @@ -230,10 +235,10 @@ def test_a2a_connection_failure_is_down_not_unexpected(tmp_path): assert proc.returncode == 0, proc.stderr log = log_path.read_text() - assert "kagentz: ❌ A2A down (HTTP 000)" in log - assert "kagentz: ✅ A2A alive" not in log + assert "kagentz C1: ❌ A2A down (HTTP 000)" in log + assert "kagentz C1: ✅ A2A alive" not in log assert "unexpected" not in log - assert "Result: 🔴 1 issue(s) found" in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log assert "kagentz A2A server DOWN (connection failed)" in proc.stdout diff --git a/tests/test_zulip_kagentz_legs.py b/tests/test_zulip_kagentz_legs.py index 0254a86..98e8dbc 100644 --- a/tests/test_zulip_kagentz_legs.py +++ b/tests/test_zulip_kagentz_legs.py @@ -30,8 +30,6 @@ import pathlib import stat import subprocess -import pytest - ROOT = pathlib.Path(__file__).resolve().parents[1] ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh" CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json" @@ -209,25 +207,3 @@ def test_c1_000_is_incident(tmp_path): assert "all healthy" not in log # The notify fired for the A2A down. assert "kagentz A2A server DOWN" in proc.stdout - - -# ── Prose assertions (kept as secondary, do not replace behavioural) ── - -def test_prose_c1_no_credential_needed(): - """C1 header should say 'no credential needed'.""" - prose = (ROOT / "zulip-health.prose.md").read_text() - assert "C1: A2A Server Health (no credential needed)" in prose - - -def test_prose_c2_requires_litellm_key(): - """C2 header should say 'requires LITELLM_KEY'.""" - prose = (ROOT / "zulip-health.prose.md").read_text() - assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose - - -def test_prose_c3_public_access_path(): - """C3 section should exist and document 502/000 as incidents.""" - prose = (ROOT / "zulip-health.prose.md").read_text() - assert "C3: Public Access Path" in prose - assert "https://kagentz.sysloggh.net/" in prose - assert "502" in prose diff --git a/zulip-health.prose.md b/zulip-health.prose.md index 097e2ec..bce42d6 100644 --- a/zulip-health.prose.md +++ b/zulip-health.prose.md @@ -528,12 +528,19 @@ If any bot processes >50 bot-originated messages in 15min → warning. ### Step 6: Compile and Report -1. Compile all platform checks and severity -2. Determine `overall_severity` from worst per-agent severity -3. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart -4. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp -5. If any agent critical or >2 degraded: send relay message to user -6. Update `last_check` timestamp in `### Maintains` snapshot +1. Run `scripts/zulip-monitor.sh` and take its final `Result:` line as the + authoritative run verdict. The verdict line is either + `Result: ✅ 0 issues (all healthy)` or + `Result: 🔴 INCIDENT — N issue(s) found`. +2. Quote that `Result:` line verbatim in the status report. When it says + `INCIDENT`, the run MUST be reported as an incident — never summarised as + OK/healthy and never annotated as "expected". +3. Compile all platform checks and severity +4. Determine `overall_severity` from worst per-agent severity +5. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart +6. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp +7. If any agent critical or >2 degraded: send relay message to user +8. Update `last_check` timestamp in `### Maintains` snapshot ### Restart Debounce -- 2.54.0 From a17379676de3cccd88825ca0c8c990232ed6c028 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:38:53 +0000 Subject: [PATCH 5/6] no-mistakes(document): docs: sync zulip-health contract with C1/C3 and version --- contract-registry.yaml | 2 +- zulip-health.prose.md | 13 +++++++------ 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/contract-registry.yaml b/contract-registry.yaml index 019be2b..ea07da0 100644 --- a/contract-registry.yaml +++ b/contract-registry.yaml @@ -628,7 +628,7 @@ contracts: sensitivity: high status: active owner: abiba - version: 3.3.0 + version: 3.4.0 trigger: type: scheduled cadence: '*/15 * * * *' diff --git a/zulip-health.prose.md b/zulip-health.prose.md index bce42d6..c28fbfb 100644 --- a/zulip-health.prose.md +++ b/zulip-health.prose.md @@ -1,9 +1,9 @@ --- kind: responsibility name: zulip-health -description: Multi-platform health monitor for the Zulip messaging mesh spanning Platform A (pi/Abiba Zulip bridge), Platform B (Tanko on DSH), and Platform C (Agent Zero Docker). Verifies bot registration, DM delivery, and cross-platform connectivity. The kagentz Zulip adapter leg is retired (its code no longer exists) — Agent Zero is probed for A2A liveness only. Mumuni is no longer monitored from this host — she runs on her own container (kagentz CT 105 on minipve, .14) and is monitored on her side. +description: Multi-platform health monitor for the Zulip messaging mesh spanning Platform A (pi/Abiba Zulip bridge), Platform B (Tanko on DSH), and Platform C (Agent Zero Docker). Verifies bot registration, DM delivery, and cross-platform connectivity. The kagentz Zulip adapter leg is retired (its code no longer exists) — Agent Zero is probed for A2A liveness, A2A response, and public-path access only. Mumuni is no longer monitored from this host — she runs on her own container (kagentz CT 105 on minipve, .14) and is monitored on her side. title: Zulip Mesh Health Monitor — Multi-Platform -version: 3.3.0 +version: 3.4.0 runtime_contract: 2 agent: abiba report_only_agents: @@ -30,7 +30,7 @@ session start. - **Zulip API key** for `abiba-bot@chat.sysloggh.net` in `$ZULIP_API_KEY` - **SSH access** to amdpve (192.168.68.15) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14) - **PM2** on localhost for pi process management -- **Network access** to `chat.sysloggh.net`, `localhost:9200` +- **Network access** to `chat.sysloggh.net`, `kagentz.sysloggh.net` (C3 public path), `localhost:9200` - **Write access** to `/root/zulip-health-monitor.log` and `/tmp/zulip-monitor-debounce` - **Relay access** via RA-H OS MCP for alert delivery @@ -129,10 +129,10 @@ grep -c "async def edit_message" ~/.hermes/plugins/*/zulip*/adapter.py ### Liveness rule (scoped) -Any HTTP response proves the service is ALIVE. For auth-gated endpoints (Zulip API, Tanko gateway), a 401/403 redirect or status means the service answered — report the code, never "down". Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. +Any HTTP response proves the service is ALIVE. For auth-gated endpoints (Zulip API, Tanko gateway), a 401/403 redirect or status means the service answered — report the code, never "down". Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. **Proxy-fronted exception:** a reverse proxy's `502`/`504` (upstream refused/unreachable) is not the backend answering — for proxy-fronted public endpoints such as `https://kagentz.sysloggh.net/` (C3), treat it as DOWN. **STANDING PROBE RULES (2026-09-14, from defect report 1150.msg):** -1. **Any HTTP status means ALIVE.** 200, 301, 302, 401, 403, 404 all prove the service answered — report the code, never "down". A redirect is not a failure. Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. +1. **Any HTTP status means ALIVE.** 200, 301, 302, 401, 403, 404 all prove the service answered — report the code, never "down". A redirect is not a failure. Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe (proxy `502`/`504` excepted, above). 2. **A failed probe is never a service verdict.** Print `probe-failed: ` naming the exact URL/host/port and the failure kind (timeout, refused, no-route, dns), retry once at a longer timeout, and only then report. 3. **Say which probe produced each number.** "API: 000" is unusable; "API https://chat.sysloggh.net/api/v1/server_settings -> connection timeout after 10s (retried at 25s: also timeout)" is actionable. @@ -469,7 +469,8 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null > monitor issued a restart for something that could not start, posting a false > kagentz-adapter-down alert on every run. Do NOT re-add an adapter-process, > heartbeat/queue, or adapter-restart step. Agent Zero is probed for A2A -> liveness only, and a probe must never restart a platform. +> liveness/response and public-path access only, and a probe must never restart +> a platform. **C1: A2A Server Health (no credential needed)** -- 2.54.0 From c72436b4061df1bc06195fb3464f0646b9b7495b Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:40:25 +0000 Subject: [PATCH 6/6] no-mistakes(document): docs: reconcile zulip-monitor status in infrastructure-control --- infrastructure-control.prose.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/infrastructure-control.prose.md b/infrastructure-control.prose.md index 7144fb0..67cc2c5 100644 --- a/infrastructure-control.prose.md +++ b/infrastructure-control.prose.md @@ -759,4 +759,4 @@ which was kill+nohup outside systemd) are banned by policy. | Script | Why Disabled | |--------|-------------| | `zulip-watchdog.sh` (Mumuni) | kill+nohup bypassed systemd, 27 restarts, pattern mismatch | -| `zulip-monitor.sh` (Abiba) | Replaced by agent-health-check.py + PM2 auto-restart | +| `zulip-monitor.sh` (Abiba) | Standalone cron replaced by agent-health-check.py + PM2 auto-restart; the script itself remains active as the execution step of the `zulip-health` responsibility contract (see `zulip-health.prose.md`) | -- 2.54.0