diff --git a/contract-registry.yaml b/contract-registry.yaml index 019be2b..ea07da0 100644 --- a/contract-registry.yaml +++ b/contract-registry.yaml @@ -628,7 +628,7 @@ contracts: sensitivity: high status: active owner: abiba - version: 3.3.0 + version: 3.4.0 trigger: type: scheduled cadence: '*/15 * * * *' diff --git a/infrastructure-control.prose.md b/infrastructure-control.prose.md index 7144fb0..67cc2c5 100644 --- a/infrastructure-control.prose.md +++ b/infrastructure-control.prose.md @@ -759,4 +759,4 @@ which was kill+nohup outside systemd) are banned by policy. | Script | Why Disabled | |--------|-------------| | `zulip-watchdog.sh` (Mumuni) | kill+nohup bypassed systemd, 27 restarts, pattern mismatch | -| `zulip-monitor.sh` (Abiba) | Replaced by agent-health-check.py + PM2 auto-restart | +| `zulip-monitor.sh` (Abiba) | Standalone cron replaced by agent-health-check.py + PM2 auto-restart; the script itself remains active as the execution step of the `zulip-health` responsibility contract (see `zulip-health.prose.md`) | diff --git a/scripts/zulip-monitor.sh b/scripts/zulip-monitor.sh index c1423e3..05b6a42 100755 --- a/scripts/zulip-monitor.sh +++ b/scripts/zulip-monitor.sh @@ -176,6 +176,11 @@ fi # never contact her former host. # ── Platform C: Agent Zero (kagentz) ── +# C1: A2A liveness (no credential needed) — probes the container's internal :80/a2a/ +# C2: A2A response verification (needs LITELLM_KEY) — probes POST /a2a with auth +# C3: Public access path (no credential needed) — probes https://kagentz.sysloggh.net/ + +# C1: A2A liveness (container-internal probe) AZ_A2A_CODE=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \ "docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/ 2>/dev/null" 2>/dev/null) || AZ_A2A_CODE="000" AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]') @@ -184,27 +189,53 @@ AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]') if [ "$AZ_A2A_CODE" = "000" ]; then notify "🔴" "kagentz A2A server DOWN (connection failed)" ISSUES=$((ISSUES + 1)) - echo " kagentz: ❌ A2A down (HTTP 000)" >> "$LOG" + echo " kagentz C1: ❌ A2A down (HTTP 000)" >> "$LOG" else case "$AZ_A2A_CODE" in 200|401) - echo " kagentz: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;; + echo " kagentz C1: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;; *) notify "🟡" "kagentz A2A server answered HTTP $AZ_A2A_CODE — running, unexpected status" ISSUES=$((ISSUES + 1)) - echo " kagentz: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;; + echo " kagentz C1: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;; + esac +fi + +# C3: Public access path (the captain's point of view) +# Probes the public URL that NetBird proxies to the container. 200/302/401 = alive, +# 502 = proxy's "upstream refused" page (incident), connection failed = incident. +# Never restarts anything — the contract forbids restarting the platform. +KAGENTZ_PUBLIC_CODE=$(curl -s -o /dev/null --connect-timeout 10 --max-time 15 \ + -w '%{http_code}' https://kagentz.sysloggh.net/ 2>/dev/null) || KAGENTZ_PUBLIC_CODE="000" +KAGENTZ_PUBLIC_CODE=$(printf '%s' "$KAGENTZ_PUBLIC_CODE" | tr -d '[:space:]') +[ -n "$KAGENTZ_PUBLIC_CODE" ] || KAGENTZ_PUBLIC_CODE="000" + +if [ "$KAGENTZ_PUBLIC_CODE" = "000" ]; then + notify "🔴" "kagentz public URL DOWN (connection failed)" + ISSUES=$((ISSUES + 1)) + echo " kagentz C3: ❌ public URL down (HTTP 000)" >> "$LOG" +elif [ "$KAGENTZ_PUBLIC_CODE" = "502" ]; then + notify "🔴" "kagentz public URL 502 (upstream refused)" + ISSUES=$((ISSUES + 1)) + echo " kagentz C3: ❌ public URL 502 (upstream refused)" >> "$LOG" +else + case "$KAGENTZ_PUBLIC_CODE" in + 200|302|401) + echo " kagentz C3: ✅ public URL alive (HTTP $KAGENTZ_PUBLIC_CODE)" >> "$LOG" ;; + *) + notify "🟡" "kagentz public URL answered HTTP $KAGENTZ_PUBLIC_CODE — running, unexpected status" + ISSUES=$((ISSUES + 1)) + echo " kagentz C3: 🟡 public URL unexpected http=$KAGENTZ_PUBLIC_CODE (running, warning)" >> "$LOG" ;; esac fi # ── Summary ── +# The run verdict is non-optimistic: when ISSUES > 0, the run is an INCIDENT. +# The lane must quote this Result line verbatim in its status report. if [ "$ISSUES" -eq 0 ]; then - if [ "$ZULIP_CRED_OK" -eq 0 ]; then - echo " Result: ✅ All healthy" >> "$LOG" - else - echo " Result: ✅ All healthy" >> "$LOG" - fi + echo " Result: ✅ 0 issues (all healthy)" >> "$LOG" else - echo " Result: 🔴 $ISSUES issue(s) found" >> "$LOG" + echo " Result: 🔴 INCIDENT — $ISSUES issue(s) found" >> "$LOG" notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log" fi diff --git a/tests/test_mumuni_monitor_removal.py b/tests/test_mumuni_monitor_removal.py index df55352..6e765af 100644 --- a/tests/test_mumuni_monitor_removal.py +++ b/tests/test_mumuni_monitor_removal.py @@ -98,8 +98,8 @@ exit 0 """ CURL_STUB = r"""#!/usr/bin/env bash -# Stub curl: serve the Abiba health fixture and the Zulip server 200, and -# record every call (including notify) payloads. +# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the +# kagentz C3 public URL, and record every call (including notify) payloads. printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls" case "$*" in *:9200/health*) @@ -109,6 +109,8 @@ case "$*" in esac ;; *server_settings*) printf '%s' "$SERVER_HTTP" ;; + *kagentz.sysloggh.net*) + printf '%s' "$KAGENTZ_PUBLIC_CODE" ;; esac exit 0 """ @@ -121,7 +123,8 @@ def _write_exec(path: pathlib.Path, body: str) -> None: def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", - az_a2a_code="401", az_a2a_exit=0): + az_a2a_code="401", az_a2a_exit=0, + kagentz_public_code="302"): """Run the shipped monitor in a sandbox; return (proc, record_dir, log_path). Only the LOG constant is rewritten (to keep the run inside the worktree). @@ -154,6 +157,7 @@ def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", "PI_HTTP": "200", "PI_BODY": CONNECTED_FIXTURE.read_text(), "SERVER_HTTP": "200", + "KAGENTZ_PUBLIC_CODE": kagentz_public_code, }) proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env, capture_output=True, text=True) @@ -169,8 +173,9 @@ def test_healthy_run_is_quiet_and_never_reaches_mumuni(tmp_path): assert "Server: ✅ HTTP 200" in log assert "Abiba: ✅ Connected" in log assert "Tanko: ✅ service=active http=200" in log - assert "kagentz: ✅ A2A alive (HTTP 401)" in log - assert "Result: ✅ All healthy" in log + assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log + assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log + assert "Result: ✅ 0 issues (all healthy)" in log # A healthy run emits no notify at all — and certainly no Mumuni one. assert proc.stdout == "" @@ -207,8 +212,8 @@ def test_failing_run_alerts_on_tanko_but_never_on_mumuni(tmp_path): # The rest of the monitor still ran alongside the failing Tanko leg. log = log_path.read_text() assert "Abiba: ✅ Connected" in log - assert "kagentz: ✅ A2A alive" in log - assert "Result: 🔴 1 issue(s) found" in log + assert "kagentz C1: ✅ A2A alive" in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path): @@ -216,9 +221,9 @@ def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path): assert proc.returncode == 0, proc.stderr log = log_path.read_text() - assert "kagentz: 🟡 A2A unexpected http=500 (running, warning)" in log - assert "kagentz: ✅ A2A alive" not in log - assert "Result: 🔴 1 issue(s) found" in log + assert "kagentz C1: 🟡 A2A unexpected http=500 (running, warning)" in log + assert "kagentz C1: ✅ A2A alive" not in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log assert "kagentz A2A server answered HTTP 500" in proc.stdout @@ -230,10 +235,10 @@ def test_a2a_connection_failure_is_down_not_unexpected(tmp_path): assert proc.returncode == 0, proc.stderr log = log_path.read_text() - assert "kagentz: ❌ A2A down (HTTP 000)" in log - assert "kagentz: ✅ A2A alive" not in log + assert "kagentz C1: ❌ A2A down (HTTP 000)" in log + assert "kagentz C1: ✅ A2A alive" not in log assert "unexpected" not in log - assert "Result: 🔴 1 issue(s) found" in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log assert "kagentz A2A server DOWN (connection failed)" in proc.stdout diff --git a/tests/test_zulip_kagentz_legs.py b/tests/test_zulip_kagentz_legs.py new file mode 100644 index 0000000..98e8dbc --- /dev/null +++ b/tests/test_zulip_kagentz_legs.py @@ -0,0 +1,209 @@ +#!/usr/bin/env python3 +"""Behavioural tests for zulip-monitor.sh kagentz C1/C3 legs and the Result verdict. + +WHY THIS FILE EXISTS: the 2026-09-19 kagentz A2A outage was correctly detected +by the monitor (C1 returned 000, the run logged an issue) but the lane's own +summarization in ops.status wrote "OK" with the note "A2A server DOWN — expected +(no credentials configured)". The optimistic verdict came from the lane, not the +script. The fix adds a C3 public-access-path leg and makes the Result line say +"INCIDENT" when issues are found, so the lane can quote it verbatim. + +CONTRACT UNDER TEST: + * C1 (A2A liveness, no credential): 000 → INCIDENT. + * C3 (public access path, no credential): 502 → INCIDENT, 000 → INCIDENT, + 200/302/401 → alive. + * Result verdict: when ISSUES > 0, the log's Result line says "INCIDENT", + not just "issues found". + * Healthy control: C1 401 + C3 302 → 0 issues, "all healthy". + +HOW: behavioural execution using the sandbox pattern already in this repo +(tests/test_mumuni_monitor_removal.py). The sandbox copies the shipped monitor +verbatim, rewrites only its LOG constant, and runs it with stub ssh/curl on +PATH. Each test asserts from the run's own log/verdict, not from file text. + +Usage: python3 -m pytest tests/test_zulip_kagentz_legs.py +""" +from __future__ import annotations + +import os +import pathlib +import stat +import subprocess + +ROOT = pathlib.Path(__file__).resolve().parents[1] +ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh" +CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json" + +TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec +AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker + + +# ── Stub ssh: answers Tanko and Agent Zero probes by env vars ────────── + +SSH_STUB = r"""#!/usr/bin/env bash +# Stub ssh: record the target host, then answer by host + remote command. +printf '%s\n' "$*" >> "$RECORD_DIR/ssh.calls" +host="" +for a in "$@"; do + case "$a" in + *@192.168.*) host="${a##*@}" ;; + esac +done +printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts" +cmd="${*: -1}" +case "$host" in + 192.168.68.15) + case "$cmd" in + *"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;; + *curl*) printf '%s' "$TANKO_HTTP" ;; + esac ;; + 192.168.68.14) + case "$cmd" in + *"/a2a/"*) printf '%s' "$AZ_A2A_CODE"; exit "${AZ_A2A_EXIT:-0}" ;; + esac ;; + *) + printf 'UNEXPECTED-SSH-HOST %s\n' "$host" >> "$RECORD_DIR/unexpected-ssh" ;; +esac +exit 0 +""" + + +# ── Stub curl: serves Zulip server, Abiba health, and C3 public URL ─── + +CURL_STUB = r"""#!/usr/bin/env bash +# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the +# C3 public URL probe (https://kagentz.sysloggh.net/). Record every call. +printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls" +case "$*" in + *:9200/health*) + case " $* " in + *" -w "*) printf '%s' "$PI_HTTP" ;; # -w '%{http_code}' probe + *) printf '%s' "$PI_BODY" ;; # body probe + esac ;; + *server_settings*) + printf '%s' "$SERVER_HTTP" ;; + *kagentz.sysloggh.net*) + printf '%s' "$KAGENTZ_PUBLIC_CODE" ;; +esac +exit 0 +""" + + +def _write_exec(path: pathlib.Path, body: str) -> None: + path.write_text(body) + path.chmod(path.stat().st_mode + | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + + +def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", + az_a2a_code="401", az_a2a_exit=0, + kagentz_public_code="302"): + """Run the shipped monitor in a sandbox; return (proc, record_dir, log_path). + + Only the LOG constant is rewritten (to keep the run inside the worktree). + Everything else — legs, labels, notify logic — is the shipped script. + """ + sandbox = tmp_path / "sandbox" + bindir = sandbox / "bin" + record = sandbox / "record" + bindir.mkdir(parents=True) + record.mkdir() + + _write_exec(bindir / "ssh", SSH_STUB) + _write_exec(bindir / "curl", CURL_STUB) + + source = ZULIP_MONITOR.read_text() + log_line = 'LOG="/root/zulip-health-monitor.log"' + assert log_line in source, "LOG constant moved — update the sandbox harness" + log_path = sandbox / "zulip-health-monitor.log" + script = sandbox / "zulip-monitor.sh" + script.write_text(source.replace(log_line, f'LOG="{log_path}"')) + + env = dict(os.environ) + env.update({ + "PATH": f"{bindir}:{env['PATH']}", + "RECORD_DIR": str(record), + "TANKO_SVC": tanko_svc, + "TANKO_HTTP": tanko_http, + "AZ_A2A_CODE": az_a2a_code, + "AZ_A2A_EXIT": str(az_a2a_exit), + "PI_HTTP": "200", + "PI_BODY": CONNECTED_FIXTURE.read_text(), + "SERVER_HTTP": "200", + "KAGENTZ_PUBLIC_CODE": kagentz_public_code, + }) + proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env, + capture_output=True, text=True) + return proc, record, log_path + + +# ── Required behavioural cases ───────────────────────────────────────── + +def test_c3_502_is_incident(tmp_path): + """C3 public leg returns 502 → the run's verdict is an INCIDENT, + the C3 line names the 502, and the run is not summarised as healthy.""" + proc, record, log_path = _run_monitor(tmp_path, kagentz_public_code="502") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # The C3 line names the 502. + assert "kagentz C3: ❌ public URL 502 (upstream refused)" in log + # The verdict is an INCIDENT, not healthy. + assert "Result: 🔴 INCIDENT" in log + assert "all healthy" not in log + # The run is not summarised as healthy. + assert "✅ 0 issues" not in log + + +def test_c3_000_is_incident(tmp_path): + """C3 public leg returns 000 → INCIDENT.""" + proc, record, log_path = _run_monitor(tmp_path, kagentz_public_code="000") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # The C3 line reports the connection failure. + assert "kagentz C3: ❌ public URL down (HTTP 000)" in log + # The verdict is an INCIDENT. + assert "Result: 🔴 INCIDENT" in log + assert "all healthy" not in log + assert "✅ 0 issues" not in log + + +def test_healthy_control_c1_401_c3_302(tmp_path): + """Healthy control: C1 401 plus C3 302 → 0 issues and a healthy verdict, + proving the new leg cannot cry wolf.""" + proc, record, log_path = _run_monitor(tmp_path, + az_a2a_code="401", + kagentz_public_code="302") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # Both legs report alive. + assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log + assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log + # Zero issues, healthy verdict. + assert "Result: ✅ 0 issues (all healthy)" in log + # No INCIDENT. + assert "INCIDENT" not in log + # No notify fired for kagentz. + assert "kagentz public URL" not in proc.stdout + assert "kagentz A2A server" not in proc.stdout + + +def test_c1_000_is_incident(tmp_path): + """C1 returns 000 → INCIDENT, keeping the leg that actually caught this + outage covered behaviourally.""" + proc, record, log_path = _run_monitor(tmp_path, + az_a2a_code="000", + az_a2a_exit=7, + kagentz_public_code="302") + assert proc.returncode == 0, proc.stderr + log = log_path.read_text() + + # The C1 line reports the A2A down. + assert "kagentz C1: ❌ A2A down (HTTP 000)" in log + # The verdict is an INCIDENT (even though C3 is healthy). + assert "Result: 🔴 INCIDENT" in log + assert "all healthy" not in log + # The notify fired for the A2A down. + assert "kagentz A2A server DOWN" in proc.stdout diff --git a/zulip-health.prose.md b/zulip-health.prose.md index 54229f8..c28fbfb 100644 --- a/zulip-health.prose.md +++ b/zulip-health.prose.md @@ -1,9 +1,9 @@ --- kind: responsibility name: zulip-health -description: Multi-platform health monitor for the Zulip messaging mesh spanning Platform A (pi/Abiba Zulip bridge), Platform B (Tanko on DSH), and Platform C (Agent Zero Docker). Verifies bot registration, DM delivery, and cross-platform connectivity. The kagentz Zulip adapter leg is retired (its code no longer exists) — Agent Zero is probed for A2A liveness only. Mumuni is no longer monitored from this host — she runs on her own container (kagentz CT 105 on minipve, .14) and is monitored on her side. +description: Multi-platform health monitor for the Zulip messaging mesh spanning Platform A (pi/Abiba Zulip bridge), Platform B (Tanko on DSH), and Platform C (Agent Zero Docker). Verifies bot registration, DM delivery, and cross-platform connectivity. The kagentz Zulip adapter leg is retired (its code no longer exists) — Agent Zero is probed for A2A liveness, A2A response, and public-path access only. Mumuni is no longer monitored from this host — she runs on her own container (kagentz CT 105 on minipve, .14) and is monitored on her side. title: Zulip Mesh Health Monitor — Multi-Platform -version: 3.3.0 +version: 3.4.0 runtime_contract: 2 agent: abiba report_only_agents: @@ -30,7 +30,7 @@ session start. - **Zulip API key** for `abiba-bot@chat.sysloggh.net` in `$ZULIP_API_KEY` - **SSH access** to amdpve (192.168.68.15) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14) - **PM2** on localhost for pi process management -- **Network access** to `chat.sysloggh.net`, `localhost:9200` +- **Network access** to `chat.sysloggh.net`, `kagentz.sysloggh.net` (C3 public path), `localhost:9200` - **Write access** to `/root/zulip-health-monitor.log` and `/tmp/zulip-monitor-debounce` - **Relay access** via RA-H OS MCP for alert delivery @@ -129,10 +129,10 @@ grep -c "async def edit_message" ~/.hermes/plugins/*/zulip*/adapter.py ### Liveness rule (scoped) -Any HTTP response proves the service is ALIVE. For auth-gated endpoints (Zulip API, Tanko gateway), a 401/403 redirect or status means the service answered — report the code, never "down". Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. +Any HTTP response proves the service is ALIVE. For auth-gated endpoints (Zulip API, Tanko gateway), a 401/403 redirect or status means the service answered — report the code, never "down". Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. **Proxy-fronted exception:** a reverse proxy's `502`/`504` (upstream refused/unreachable) is not the backend answering — for proxy-fronted public endpoints such as `https://kagentz.sysloggh.net/` (C3), treat it as DOWN. **STANDING PROBE RULES (2026-09-14, from defect report 1150.msg):** -1. **Any HTTP status means ALIVE.** 200, 301, 302, 401, 403, 404 all prove the service answered — report the code, never "down". A redirect is not a failure. Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. +1. **Any HTTP status means ALIVE.** 200, 301, 302, 401, 403, 404 all prove the service answered — report the code, never "down". A redirect is not a failure. Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe (proxy `502`/`504` excepted, above). 2. **A failed probe is never a service verdict.** Print `probe-failed: ` naming the exact URL/host/port and the failure kind (timeout, refused, no-route, dns), retry once at a longer timeout, and only then report. 3. **Say which probe produced each number.** "API: 000" is unusable; "API https://chat.sysloggh.net/api/v1/server_settings -> connection timeout after 10s (retried at 25s: also timeout)" is actionable. @@ -469,9 +469,10 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null > monitor issued a restart for something that could not start, posting a false > kagentz-adapter-down alert on every run. Do NOT re-add an adapter-process, > heartbeat/queue, or adapter-restart step. Agent Zero is probed for A2A -> liveness only, and a probe must never restart a platform. +> liveness/response and public-path access only, and a probe must never restart +> a platform. -**C1: A2A Server Health** +**C1: A2A Server Health (no credential needed)** ```bash # A2A listens on :80 inside the agent-zero container (host-mapped to :50080) and @@ -479,9 +480,9 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null ssh root@192.168.68.14 "docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/" ``` -Expected: `401` (auth-gated, A2A server is up and responding) or `200` (if no auth required). Connection refused (`000`) → A2A server down. Any other status → running but unexpected: log/report it, never restart. +Expected: `401` (auth-gated, A2A server is up and responding) or `200` (if no auth required). Connection refused (`000`) → A2A server down (INCIDENT). Any other status → running but unexpected: log/report it, never restart. -**C2: A2A Response Verification** +**C2: A2A Response Verification (requires LITELLM_KEY)** ```bash # A2A listens on :80 inside the container and is auth-gated (401 expected unauthenticated). @@ -491,15 +492,30 @@ ssh root@192.168.68.14 "docker exec agent-zero curl -s -X POST http://127.0.0.1: -d '{\"jsonrpc\":\"2.0\",\"method\":\"tasks/send\",\"params\":{\"message\":{\"role\":\"user\",\"parts\":[{\"text\":\"ping\"}]}},\"id\":1}'" ``` -Expected: task ID with "working" status. Poll for completion with `tasks/get`. If 401, check LITELLM_KEY is set. +Expected: task ID with "working" status. Poll for completion with `tasks/get`. If 401, check LITELLM_KEY is set (this is a credential issue, NOT a server-down incident). + +**C3: Public Access Path (no credential needed)** + +```bash +# Probes the public URL that NetBird proxies to the agent-zero container. +# This is the captain's point of view: if the captain can't reach it, it's down. +# 200/302/401 = alive, 502 = proxy's "upstream refused" page (INCIDENT), +# connection failed (000) = INCIDENT. Never restarts anything. +curl -s -o /dev/null --connect-timeout 10 --max-time 15 -w '%{http_code}' https://kagentz.sysloggh.net/ +``` + +Expected: `200` (Agent Zero login page), `302` (redirect), or `401` (auth-gated) = alive. `502` = NetBird proxy's "upstream refused" page (the container's port 80 is not listening) = INCIDENT. `000` (connection failed) = INCIDENT. Any other status → running but unexpected: log/report it, never restart. **Platform C Actions** | Condition | Action | |-----------|--------| -| A2A returns `000` (connection refused/timeout) | Alert only — never restart the platform; investigate the agent-zero container | -| A2A returns a status other than `200`/`401` | Log/report as a warning — reported, never healed on | -| LiteLLM 401 | Check API key in a2a_agent.py `LITELLM_KEY` | +| C1 A2A returns `000` (connection refused/timeout) | Alert only — never restart the platform; investigate the agent-zero container | +| C1 A2A returns a status other than `200`/`401` | Log/report as a warning — reported, never healed on | +| C2 A2A returns 401 | Check `LITELLM_KEY` is set (credential issue, NOT server-down) | +| C3 public URL returns `502` (upstream refused) | Alert only — the container's port 80 is not listening; investigate the agent-zero container | +| C3 public URL returns `000` (connection failed) | Alert only — the public path is down; investigate the NetBird proxy or the container | +| C3 public URL returns a status other than `200`/`302`/`401` | Log/report as a warning — reported, never healed on | ### Step 5: Global Checks @@ -513,12 +529,19 @@ If any bot processes >50 bot-originated messages in 15min → warning. ### Step 6: Compile and Report -1. Compile all platform checks and severity -2. Determine `overall_severity` from worst per-agent severity -3. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart -4. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp -5. If any agent critical or >2 degraded: send relay message to user -6. Update `last_check` timestamp in `### Maintains` snapshot +1. Run `scripts/zulip-monitor.sh` and take its final `Result:` line as the + authoritative run verdict. The verdict line is either + `Result: ✅ 0 issues (all healthy)` or + `Result: 🔴 INCIDENT — N issue(s) found`. +2. Quote that `Result:` line verbatim in the status report. When it says + `INCIDENT`, the run MUST be reported as an incident — never summarised as + OK/healthy and never annotated as "expected". +3. Compile all platform checks and severity +4. Determine `overall_severity` from worst per-agent severity +5. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart +6. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp +7. If any agent critical or >2 degraded: send relay message to user +8. Update `last_check` timestamp in `### Maintains` snapshot ### Restart Debounce