fix(zulip-monitor): add C3 public access path, make run verdict non-optimistic, separate C1/C2/C3 #124
@@ -628,7 +628,7 @@ contracts:
|
|||||||
sensitivity: high
|
sensitivity: high
|
||||||
status: active
|
status: active
|
||||||
owner: abiba
|
owner: abiba
|
||||||
version: 3.3.0
|
version: 3.4.0
|
||||||
trigger:
|
trigger:
|
||||||
type: scheduled
|
type: scheduled
|
||||||
cadence: '*/15 * * * *'
|
cadence: '*/15 * * * *'
|
||||||
|
|||||||
@@ -759,4 +759,4 @@ which was kill+nohup outside systemd) are banned by policy.
|
|||||||
| Script | Why Disabled |
|
| Script | Why Disabled |
|
||||||
|--------|-------------|
|
|--------|-------------|
|
||||||
| `zulip-watchdog.sh` (Mumuni) | kill+nohup bypassed systemd, 27 restarts, pattern mismatch |
|
| `zulip-watchdog.sh` (Mumuni) | kill+nohup bypassed systemd, 27 restarts, pattern mismatch |
|
||||||
| `zulip-monitor.sh` (Abiba) | Replaced by agent-health-check.py + PM2 auto-restart |
|
| `zulip-monitor.sh` (Abiba) | Standalone cron replaced by agent-health-check.py + PM2 auto-restart; the script itself remains active as the execution step of the `zulip-health` responsibility contract (see `zulip-health.prose.md`) |
|
||||||
|
|||||||
@@ -176,6 +176,11 @@ fi
|
|||||||
# never contact her former host.
|
# never contact her former host.
|
||||||
|
|
||||||
# ── Platform C: Agent Zero (kagentz) ──
|
# ── Platform C: Agent Zero (kagentz) ──
|
||||||
|
# C1: A2A liveness (no credential needed) — probes the container's internal :80/a2a/
|
||||||
|
# C2: A2A response verification (needs LITELLM_KEY) — probes POST /a2a with auth
|
||||||
|
# C3: Public access path (no credential needed) — probes https://kagentz.sysloggh.net/
|
||||||
|
|
||||||
|
# C1: A2A liveness (container-internal probe)
|
||||||
AZ_A2A_CODE=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \
|
AZ_A2A_CODE=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \
|
||||||
"docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/ 2>/dev/null" 2>/dev/null) || AZ_A2A_CODE="000"
|
"docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/ 2>/dev/null" 2>/dev/null) || AZ_A2A_CODE="000"
|
||||||
AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]')
|
AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]')
|
||||||
@@ -184,27 +189,53 @@ AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]')
|
|||||||
if [ "$AZ_A2A_CODE" = "000" ]; then
|
if [ "$AZ_A2A_CODE" = "000" ]; then
|
||||||
notify "🔴" "kagentz A2A server DOWN (connection failed)"
|
notify "🔴" "kagentz A2A server DOWN (connection failed)"
|
||||||
ISSUES=$((ISSUES + 1))
|
ISSUES=$((ISSUES + 1))
|
||||||
echo " kagentz: ❌ A2A down (HTTP 000)" >> "$LOG"
|
echo " kagentz C1: ❌ A2A down (HTTP 000)" >> "$LOG"
|
||||||
else
|
else
|
||||||
case "$AZ_A2A_CODE" in
|
case "$AZ_A2A_CODE" in
|
||||||
200|401)
|
200|401)
|
||||||
echo " kagentz: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;;
|
echo " kagentz C1: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;;
|
||||||
*)
|
*)
|
||||||
notify "🟡" "kagentz A2A server answered HTTP $AZ_A2A_CODE — running, unexpected status"
|
notify "🟡" "kagentz A2A server answered HTTP $AZ_A2A_CODE — running, unexpected status"
|
||||||
ISSUES=$((ISSUES + 1))
|
ISSUES=$((ISSUES + 1))
|
||||||
echo " kagentz: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;;
|
echo " kagentz C1: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;;
|
||||||
|
esac
|
||||||
|
fi
|
||||||
|
|
||||||
|
# C3: Public access path (the captain's point of view)
|
||||||
|
# Probes the public URL that NetBird proxies to the container. 200/302/401 = alive,
|
||||||
|
# 502 = proxy's "upstream refused" page (incident), connection failed = incident.
|
||||||
|
# Never restarts anything — the contract forbids restarting the platform.
|
||||||
|
KAGENTZ_PUBLIC_CODE=$(curl -s -o /dev/null --connect-timeout 10 --max-time 15 \
|
||||||
|
-w '%{http_code}' https://kagentz.sysloggh.net/ 2>/dev/null) || KAGENTZ_PUBLIC_CODE="000"
|
||||||
|
KAGENTZ_PUBLIC_CODE=$(printf '%s' "$KAGENTZ_PUBLIC_CODE" | tr -d '[:space:]')
|
||||||
|
[ -n "$KAGENTZ_PUBLIC_CODE" ] || KAGENTZ_PUBLIC_CODE="000"
|
||||||
|
|
||||||
|
if [ "$KAGENTZ_PUBLIC_CODE" = "000" ]; then
|
||||||
|
notify "🔴" "kagentz public URL DOWN (connection failed)"
|
||||||
|
ISSUES=$((ISSUES + 1))
|
||||||
|
echo " kagentz C3: ❌ public URL down (HTTP 000)" >> "$LOG"
|
||||||
|
elif [ "$KAGENTZ_PUBLIC_CODE" = "502" ]; then
|
||||||
|
notify "🔴" "kagentz public URL 502 (upstream refused)"
|
||||||
|
ISSUES=$((ISSUES + 1))
|
||||||
|
echo " kagentz C3: ❌ public URL 502 (upstream refused)" >> "$LOG"
|
||||||
|
else
|
||||||
|
case "$KAGENTZ_PUBLIC_CODE" in
|
||||||
|
200|302|401)
|
||||||
|
echo " kagentz C3: ✅ public URL alive (HTTP $KAGENTZ_PUBLIC_CODE)" >> "$LOG" ;;
|
||||||
|
*)
|
||||||
|
notify "🟡" "kagentz public URL answered HTTP $KAGENTZ_PUBLIC_CODE — running, unexpected status"
|
||||||
|
ISSUES=$((ISSUES + 1))
|
||||||
|
echo " kagentz C3: 🟡 public URL unexpected http=$KAGENTZ_PUBLIC_CODE (running, warning)" >> "$LOG" ;;
|
||||||
esac
|
esac
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# ── Summary ──
|
# ── Summary ──
|
||||||
|
# The run verdict is non-optimistic: when ISSUES > 0, the run is an INCIDENT.
|
||||||
|
# The lane must quote this Result line verbatim in its status report.
|
||||||
if [ "$ISSUES" -eq 0 ]; then
|
if [ "$ISSUES" -eq 0 ]; then
|
||||||
if [ "$ZULIP_CRED_OK" -eq 0 ]; then
|
echo " Result: ✅ 0 issues (all healthy)" >> "$LOG"
|
||||||
echo " Result: ✅ All healthy" >> "$LOG"
|
|
||||||
else
|
|
||||||
echo " Result: ✅ All healthy" >> "$LOG"
|
|
||||||
fi
|
|
||||||
else
|
else
|
||||||
echo " Result: 🔴 $ISSUES issue(s) found" >> "$LOG"
|
echo " Result: 🔴 INCIDENT — $ISSUES issue(s) found" >> "$LOG"
|
||||||
notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log"
|
notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|||||||
@@ -98,8 +98,8 @@ exit 0
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
CURL_STUB = r"""#!/usr/bin/env bash
|
CURL_STUB = r"""#!/usr/bin/env bash
|
||||||
# Stub curl: serve the Abiba health fixture and the Zulip server 200, and
|
# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the
|
||||||
# record every call (including notify) payloads.
|
# kagentz C3 public URL, and record every call (including notify) payloads.
|
||||||
printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls"
|
printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls"
|
||||||
case "$*" in
|
case "$*" in
|
||||||
*:9200/health*)
|
*:9200/health*)
|
||||||
@@ -109,6 +109,8 @@ case "$*" in
|
|||||||
esac ;;
|
esac ;;
|
||||||
*server_settings*)
|
*server_settings*)
|
||||||
printf '%s' "$SERVER_HTTP" ;;
|
printf '%s' "$SERVER_HTTP" ;;
|
||||||
|
*kagentz.sysloggh.net*)
|
||||||
|
printf '%s' "$KAGENTZ_PUBLIC_CODE" ;;
|
||||||
esac
|
esac
|
||||||
exit 0
|
exit 0
|
||||||
"""
|
"""
|
||||||
@@ -121,7 +123,8 @@ def _write_exec(path: pathlib.Path, body: str) -> None:
|
|||||||
|
|
||||||
|
|
||||||
def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
|
def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
|
||||||
az_a2a_code="401", az_a2a_exit=0):
|
az_a2a_code="401", az_a2a_exit=0,
|
||||||
|
kagentz_public_code="302"):
|
||||||
"""Run the shipped monitor in a sandbox; return (proc, record_dir, log_path).
|
"""Run the shipped monitor in a sandbox; return (proc, record_dir, log_path).
|
||||||
|
|
||||||
Only the LOG constant is rewritten (to keep the run inside the worktree).
|
Only the LOG constant is rewritten (to keep the run inside the worktree).
|
||||||
@@ -154,6 +157,7 @@ def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
|
|||||||
"PI_HTTP": "200",
|
"PI_HTTP": "200",
|
||||||
"PI_BODY": CONNECTED_FIXTURE.read_text(),
|
"PI_BODY": CONNECTED_FIXTURE.read_text(),
|
||||||
"SERVER_HTTP": "200",
|
"SERVER_HTTP": "200",
|
||||||
|
"KAGENTZ_PUBLIC_CODE": kagentz_public_code,
|
||||||
})
|
})
|
||||||
proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env,
|
proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env,
|
||||||
capture_output=True, text=True)
|
capture_output=True, text=True)
|
||||||
@@ -169,8 +173,9 @@ def test_healthy_run_is_quiet_and_never_reaches_mumuni(tmp_path):
|
|||||||
assert "Server: ✅ HTTP 200" in log
|
assert "Server: ✅ HTTP 200" in log
|
||||||
assert "Abiba: ✅ Connected" in log
|
assert "Abiba: ✅ Connected" in log
|
||||||
assert "Tanko: ✅ service=active http=200" in log
|
assert "Tanko: ✅ service=active http=200" in log
|
||||||
assert "kagentz: ✅ A2A alive (HTTP 401)" in log
|
assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log
|
||||||
assert "Result: ✅ All healthy" in log
|
assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log
|
||||||
|
assert "Result: ✅ 0 issues (all healthy)" in log
|
||||||
|
|
||||||
# A healthy run emits no notify at all — and certainly no Mumuni one.
|
# A healthy run emits no notify at all — and certainly no Mumuni one.
|
||||||
assert proc.stdout == ""
|
assert proc.stdout == ""
|
||||||
@@ -207,8 +212,8 @@ def test_failing_run_alerts_on_tanko_but_never_on_mumuni(tmp_path):
|
|||||||
# The rest of the monitor still ran alongside the failing Tanko leg.
|
# The rest of the monitor still ran alongside the failing Tanko leg.
|
||||||
log = log_path.read_text()
|
log = log_path.read_text()
|
||||||
assert "Abiba: ✅ Connected" in log
|
assert "Abiba: ✅ Connected" in log
|
||||||
assert "kagentz: ✅ A2A alive" in log
|
assert "kagentz C1: ✅ A2A alive" in log
|
||||||
assert "Result: 🔴 1 issue(s) found" in log
|
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
|
||||||
|
|
||||||
|
|
||||||
def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
|
def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
|
||||||
@@ -216,9 +221,9 @@ def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
|
|||||||
assert proc.returncode == 0, proc.stderr
|
assert proc.returncode == 0, proc.stderr
|
||||||
log = log_path.read_text()
|
log = log_path.read_text()
|
||||||
|
|
||||||
assert "kagentz: 🟡 A2A unexpected http=500 (running, warning)" in log
|
assert "kagentz C1: 🟡 A2A unexpected http=500 (running, warning)" in log
|
||||||
assert "kagentz: ✅ A2A alive" not in log
|
assert "kagentz C1: ✅ A2A alive" not in log
|
||||||
assert "Result: 🔴 1 issue(s) found" in log
|
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
|
||||||
assert "kagentz A2A server answered HTTP 500" in proc.stdout
|
assert "kagentz A2A server answered HTTP 500" in proc.stdout
|
||||||
|
|
||||||
|
|
||||||
@@ -230,10 +235,10 @@ def test_a2a_connection_failure_is_down_not_unexpected(tmp_path):
|
|||||||
assert proc.returncode == 0, proc.stderr
|
assert proc.returncode == 0, proc.stderr
|
||||||
log = log_path.read_text()
|
log = log_path.read_text()
|
||||||
|
|
||||||
assert "kagentz: ❌ A2A down (HTTP 000)" in log
|
assert "kagentz C1: ❌ A2A down (HTTP 000)" in log
|
||||||
assert "kagentz: ✅ A2A alive" not in log
|
assert "kagentz C1: ✅ A2A alive" not in log
|
||||||
assert "unexpected" not in log
|
assert "unexpected" not in log
|
||||||
assert "Result: 🔴 1 issue(s) found" in log
|
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
|
||||||
assert "kagentz A2A server DOWN (connection failed)" in proc.stdout
|
assert "kagentz A2A server DOWN (connection failed)" in proc.stdout
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,209 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Behavioural tests for zulip-monitor.sh kagentz C1/C3 legs and the Result verdict.
|
||||||
|
|
||||||
|
WHY THIS FILE EXISTS: the 2026-09-19 kagentz A2A outage was correctly detected
|
||||||
|
by the monitor (C1 returned 000, the run logged an issue) but the lane's own
|
||||||
|
summarization in ops.status wrote "OK" with the note "A2A server DOWN — expected
|
||||||
|
(no credentials configured)". The optimistic verdict came from the lane, not the
|
||||||
|
script. The fix adds a C3 public-access-path leg and makes the Result line say
|
||||||
|
"INCIDENT" when issues are found, so the lane can quote it verbatim.
|
||||||
|
|
||||||
|
CONTRACT UNDER TEST:
|
||||||
|
* C1 (A2A liveness, no credential): 000 → INCIDENT.
|
||||||
|
* C3 (public access path, no credential): 502 → INCIDENT, 000 → INCIDENT,
|
||||||
|
200/302/401 → alive.
|
||||||
|
* Result verdict: when ISSUES > 0, the log's Result line says "INCIDENT",
|
||||||
|
not just "issues found".
|
||||||
|
* Healthy control: C1 401 + C3 302 → 0 issues, "all healthy".
|
||||||
|
|
||||||
|
HOW: behavioural execution using the sandbox pattern already in this repo
|
||||||
|
(tests/test_mumuni_monitor_removal.py). The sandbox copies the shipped monitor
|
||||||
|
verbatim, rewrites only its LOG constant, and runs it with stub ssh/curl on
|
||||||
|
PATH. Each test asserts from the run's own log/verdict, not from file text.
|
||||||
|
|
||||||
|
Usage: python3 -m pytest tests/test_zulip_kagentz_legs.py
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import pathlib
|
||||||
|
import stat
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||||
|
ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh"
|
||||||
|
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
|
||||||
|
|
||||||
|
TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec
|
||||||
|
AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker
|
||||||
|
|
||||||
|
|
||||||
|
# ── Stub ssh: answers Tanko and Agent Zero probes by env vars ──────────
|
||||||
|
|
||||||
|
SSH_STUB = r"""#!/usr/bin/env bash
|
||||||
|
# Stub ssh: record the target host, then answer by host + remote command.
|
||||||
|
printf '%s\n' "$*" >> "$RECORD_DIR/ssh.calls"
|
||||||
|
host=""
|
||||||
|
for a in "$@"; do
|
||||||
|
case "$a" in
|
||||||
|
*@192.168.*) host="${a##*@}" ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts"
|
||||||
|
cmd="${*: -1}"
|
||||||
|
case "$host" in
|
||||||
|
192.168.68.15)
|
||||||
|
case "$cmd" in
|
||||||
|
*"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;;
|
||||||
|
*curl*) printf '%s' "$TANKO_HTTP" ;;
|
||||||
|
esac ;;
|
||||||
|
192.168.68.14)
|
||||||
|
case "$cmd" in
|
||||||
|
*"/a2a/"*) printf '%s' "$AZ_A2A_CODE"; exit "${AZ_A2A_EXIT:-0}" ;;
|
||||||
|
esac ;;
|
||||||
|
*)
|
||||||
|
printf 'UNEXPECTED-SSH-HOST %s\n' "$host" >> "$RECORD_DIR/unexpected-ssh" ;;
|
||||||
|
esac
|
||||||
|
exit 0
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
# ── Stub curl: serves Zulip server, Abiba health, and C3 public URL ───
|
||||||
|
|
||||||
|
CURL_STUB = r"""#!/usr/bin/env bash
|
||||||
|
# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the
|
||||||
|
# C3 public URL probe (https://kagentz.sysloggh.net/). Record every call.
|
||||||
|
printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls"
|
||||||
|
case "$*" in
|
||||||
|
*:9200/health*)
|
||||||
|
case " $* " in
|
||||||
|
*" -w "*) printf '%s' "$PI_HTTP" ;; # -w '%{http_code}' probe
|
||||||
|
*) printf '%s' "$PI_BODY" ;; # body probe
|
||||||
|
esac ;;
|
||||||
|
*server_settings*)
|
||||||
|
printf '%s' "$SERVER_HTTP" ;;
|
||||||
|
*kagentz.sysloggh.net*)
|
||||||
|
printf '%s' "$KAGENTZ_PUBLIC_CODE" ;;
|
||||||
|
esac
|
||||||
|
exit 0
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def _write_exec(path: pathlib.Path, body: str) -> None:
|
||||||
|
path.write_text(body)
|
||||||
|
path.chmod(path.stat().st_mode
|
||||||
|
| stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH)
|
||||||
|
|
||||||
|
|
||||||
|
def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
|
||||||
|
az_a2a_code="401", az_a2a_exit=0,
|
||||||
|
kagentz_public_code="302"):
|
||||||
|
"""Run the shipped monitor in a sandbox; return (proc, record_dir, log_path).
|
||||||
|
|
||||||
|
Only the LOG constant is rewritten (to keep the run inside the worktree).
|
||||||
|
Everything else — legs, labels, notify logic — is the shipped script.
|
||||||
|
"""
|
||||||
|
sandbox = tmp_path / "sandbox"
|
||||||
|
bindir = sandbox / "bin"
|
||||||
|
record = sandbox / "record"
|
||||||
|
bindir.mkdir(parents=True)
|
||||||
|
record.mkdir()
|
||||||
|
|
||||||
|
_write_exec(bindir / "ssh", SSH_STUB)
|
||||||
|
_write_exec(bindir / "curl", CURL_STUB)
|
||||||
|
|
||||||
|
source = ZULIP_MONITOR.read_text()
|
||||||
|
log_line = 'LOG="/root/zulip-health-monitor.log"'
|
||||||
|
assert log_line in source, "LOG constant moved — update the sandbox harness"
|
||||||
|
log_path = sandbox / "zulip-health-monitor.log"
|
||||||
|
script = sandbox / "zulip-monitor.sh"
|
||||||
|
script.write_text(source.replace(log_line, f'LOG="{log_path}"'))
|
||||||
|
|
||||||
|
env = dict(os.environ)
|
||||||
|
env.update({
|
||||||
|
"PATH": f"{bindir}:{env['PATH']}",
|
||||||
|
"RECORD_DIR": str(record),
|
||||||
|
"TANKO_SVC": tanko_svc,
|
||||||
|
"TANKO_HTTP": tanko_http,
|
||||||
|
"AZ_A2A_CODE": az_a2a_code,
|
||||||
|
"AZ_A2A_EXIT": str(az_a2a_exit),
|
||||||
|
"PI_HTTP": "200",
|
||||||
|
"PI_BODY": CONNECTED_FIXTURE.read_text(),
|
||||||
|
"SERVER_HTTP": "200",
|
||||||
|
"KAGENTZ_PUBLIC_CODE": kagentz_public_code,
|
||||||
|
})
|
||||||
|
proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env,
|
||||||
|
capture_output=True, text=True)
|
||||||
|
return proc, record, log_path
|
||||||
|
|
||||||
|
|
||||||
|
# ── Required behavioural cases ─────────────────────────────────────────
|
||||||
|
|
||||||
|
def test_c3_502_is_incident(tmp_path):
|
||||||
|
"""C3 public leg returns 502 → the run's verdict is an INCIDENT,
|
||||||
|
the C3 line names the 502, and the run is not summarised as healthy."""
|
||||||
|
proc, record, log_path = _run_monitor(tmp_path, kagentz_public_code="502")
|
||||||
|
assert proc.returncode == 0, proc.stderr
|
||||||
|
log = log_path.read_text()
|
||||||
|
|
||||||
|
# The C3 line names the 502.
|
||||||
|
assert "kagentz C3: ❌ public URL 502 (upstream refused)" in log
|
||||||
|
# The verdict is an INCIDENT, not healthy.
|
||||||
|
assert "Result: 🔴 INCIDENT" in log
|
||||||
|
assert "all healthy" not in log
|
||||||
|
# The run is not summarised as healthy.
|
||||||
|
assert "✅ 0 issues" not in log
|
||||||
|
|
||||||
|
|
||||||
|
def test_c3_000_is_incident(tmp_path):
|
||||||
|
"""C3 public leg returns 000 → INCIDENT."""
|
||||||
|
proc, record, log_path = _run_monitor(tmp_path, kagentz_public_code="000")
|
||||||
|
assert proc.returncode == 0, proc.stderr
|
||||||
|
log = log_path.read_text()
|
||||||
|
|
||||||
|
# The C3 line reports the connection failure.
|
||||||
|
assert "kagentz C3: ❌ public URL down (HTTP 000)" in log
|
||||||
|
# The verdict is an INCIDENT.
|
||||||
|
assert "Result: 🔴 INCIDENT" in log
|
||||||
|
assert "all healthy" not in log
|
||||||
|
assert "✅ 0 issues" not in log
|
||||||
|
|
||||||
|
|
||||||
|
def test_healthy_control_c1_401_c3_302(tmp_path):
|
||||||
|
"""Healthy control: C1 401 plus C3 302 → 0 issues and a healthy verdict,
|
||||||
|
proving the new leg cannot cry wolf."""
|
||||||
|
proc, record, log_path = _run_monitor(tmp_path,
|
||||||
|
az_a2a_code="401",
|
||||||
|
kagentz_public_code="302")
|
||||||
|
assert proc.returncode == 0, proc.stderr
|
||||||
|
log = log_path.read_text()
|
||||||
|
|
||||||
|
# Both legs report alive.
|
||||||
|
assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log
|
||||||
|
assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log
|
||||||
|
# Zero issues, healthy verdict.
|
||||||
|
assert "Result: ✅ 0 issues (all healthy)" in log
|
||||||
|
# No INCIDENT.
|
||||||
|
assert "INCIDENT" not in log
|
||||||
|
# No notify fired for kagentz.
|
||||||
|
assert "kagentz public URL" not in proc.stdout
|
||||||
|
assert "kagentz A2A server" not in proc.stdout
|
||||||
|
|
||||||
|
|
||||||
|
def test_c1_000_is_incident(tmp_path):
|
||||||
|
"""C1 returns 000 → INCIDENT, keeping the leg that actually caught this
|
||||||
|
outage covered behaviourally."""
|
||||||
|
proc, record, log_path = _run_monitor(tmp_path,
|
||||||
|
az_a2a_code="000",
|
||||||
|
az_a2a_exit=7,
|
||||||
|
kagentz_public_code="302")
|
||||||
|
assert proc.returncode == 0, proc.stderr
|
||||||
|
log = log_path.read_text()
|
||||||
|
|
||||||
|
# The C1 line reports the A2A down.
|
||||||
|
assert "kagentz C1: ❌ A2A down (HTTP 000)" in log
|
||||||
|
# The verdict is an INCIDENT (even though C3 is healthy).
|
||||||
|
assert "Result: 🔴 INCIDENT" in log
|
||||||
|
assert "all healthy" not in log
|
||||||
|
# The notify fired for the A2A down.
|
||||||
|
assert "kagentz A2A server DOWN" in proc.stdout
|
||||||
+42
-19
@@ -1,9 +1,9 @@
|
|||||||
---
|
---
|
||||||
kind: responsibility
|
kind: responsibility
|
||||||
name: zulip-health
|
name: zulip-health
|
||||||
description: Multi-platform health monitor for the Zulip messaging mesh spanning Platform A (pi/Abiba Zulip bridge), Platform B (Tanko on DSH), and Platform C (Agent Zero Docker). Verifies bot registration, DM delivery, and cross-platform connectivity. The kagentz Zulip adapter leg is retired (its code no longer exists) — Agent Zero is probed for A2A liveness only. Mumuni is no longer monitored from this host — she runs on her own container (kagentz CT 105 on minipve, .14) and is monitored on her side.
|
description: Multi-platform health monitor for the Zulip messaging mesh spanning Platform A (pi/Abiba Zulip bridge), Platform B (Tanko on DSH), and Platform C (Agent Zero Docker). Verifies bot registration, DM delivery, and cross-platform connectivity. The kagentz Zulip adapter leg is retired (its code no longer exists) — Agent Zero is probed for A2A liveness, A2A response, and public-path access only. Mumuni is no longer monitored from this host — she runs on her own container (kagentz CT 105 on minipve, .14) and is monitored on her side.
|
||||||
title: Zulip Mesh Health Monitor — Multi-Platform
|
title: Zulip Mesh Health Monitor — Multi-Platform
|
||||||
version: 3.3.0
|
version: 3.4.0
|
||||||
runtime_contract: 2
|
runtime_contract: 2
|
||||||
agent: abiba
|
agent: abiba
|
||||||
report_only_agents:
|
report_only_agents:
|
||||||
@@ -30,7 +30,7 @@ session start.
|
|||||||
- **Zulip API key** for `abiba-bot@chat.sysloggh.net` in `$ZULIP_API_KEY`
|
- **Zulip API key** for `abiba-bot@chat.sysloggh.net` in `$ZULIP_API_KEY`
|
||||||
- **SSH access** to amdpve (192.168.68.15) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14)
|
- **SSH access** to amdpve (192.168.68.15) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14)
|
||||||
- **PM2** on localhost for pi process management
|
- **PM2** on localhost for pi process management
|
||||||
- **Network access** to `chat.sysloggh.net`, `localhost:9200`
|
- **Network access** to `chat.sysloggh.net`, `kagentz.sysloggh.net` (C3 public path), `localhost:9200`
|
||||||
- **Write access** to `/root/zulip-health-monitor.log` and `/tmp/zulip-monitor-debounce`
|
- **Write access** to `/root/zulip-health-monitor.log` and `/tmp/zulip-monitor-debounce`
|
||||||
- **Relay access** via RA-H OS MCP for alert delivery
|
- **Relay access** via RA-H OS MCP for alert delivery
|
||||||
|
|
||||||
@@ -129,10 +129,10 @@ grep -c "async def edit_message" ~/.hermes/plugins/*/zulip*/adapter.py
|
|||||||
|
|
||||||
### Liveness rule (scoped)
|
### Liveness rule (scoped)
|
||||||
|
|
||||||
Any HTTP response proves the service is ALIVE. For auth-gated endpoints (Zulip API, Tanko gateway), a 401/403 redirect or status means the service answered — report the code, never "down". Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe.
|
Any HTTP response proves the service is ALIVE. For auth-gated endpoints (Zulip API, Tanko gateway), a 401/403 redirect or status means the service answered — report the code, never "down". Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe. **Proxy-fronted exception:** a reverse proxy's `502`/`504` (upstream refused/unreachable) is not the backend answering — for proxy-fronted public endpoints such as `https://kagentz.sysloggh.net/` (C3), treat it as DOWN.
|
||||||
|
|
||||||
**STANDING PROBE RULES (2026-09-14, from defect report 1150.msg):**
|
**STANDING PROBE RULES (2026-09-14, from defect report 1150.msg):**
|
||||||
1. **Any HTTP status means ALIVE.** 200, 301, 302, 401, 403, 404 all prove the service answered — report the code, never "down". A redirect is not a failure. Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe.
|
1. **Any HTTP status means ALIVE.** 200, 301, 302, 401, 403, 404 all prove the service answered — report the code, never "down". A redirect is not a failure. Only a failed CONNECTION (curl status 000, timeout, refused) is a failed probe (proxy `502`/`504` excepted, above).
|
||||||
2. **A failed probe is never a service verdict.** Print `probe-failed: <target> <kind>` naming the exact URL/host/port and the failure kind (timeout, refused, no-route, dns), retry once at a longer timeout, and only then report.
|
2. **A failed probe is never a service verdict.** Print `probe-failed: <target> <kind>` naming the exact URL/host/port and the failure kind (timeout, refused, no-route, dns), retry once at a longer timeout, and only then report.
|
||||||
3. **Say which probe produced each number.** "API: 000" is unusable; "API https://chat.sysloggh.net/api/v1/server_settings -> connection timeout after 10s (retried at 25s: also timeout)" is actionable.
|
3. **Say which probe produced each number.** "API: 000" is unusable; "API https://chat.sysloggh.net/api/v1/server_settings -> connection timeout after 10s (retried at 25s: also timeout)" is actionable.
|
||||||
|
|
||||||
@@ -469,9 +469,10 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null
|
|||||||
> monitor issued a restart for something that could not start, posting a false
|
> monitor issued a restart for something that could not start, posting a false
|
||||||
> kagentz-adapter-down alert on every run. Do NOT re-add an adapter-process,
|
> kagentz-adapter-down alert on every run. Do NOT re-add an adapter-process,
|
||||||
> heartbeat/queue, or adapter-restart step. Agent Zero is probed for A2A
|
> heartbeat/queue, or adapter-restart step. Agent Zero is probed for A2A
|
||||||
> liveness only, and a probe must never restart a platform.
|
> liveness/response and public-path access only, and a probe must never restart
|
||||||
|
> a platform.
|
||||||
|
|
||||||
**C1: A2A Server Health**
|
**C1: A2A Server Health (no credential needed)**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# A2A listens on :80 inside the agent-zero container (host-mapped to :50080) and
|
# A2A listens on :80 inside the agent-zero container (host-mapped to :50080) and
|
||||||
@@ -479,9 +480,9 @@ ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null
|
|||||||
ssh root@192.168.68.14 "docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/"
|
ssh root@192.168.68.14 "docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/"
|
||||||
```
|
```
|
||||||
|
|
||||||
Expected: `401` (auth-gated, A2A server is up and responding) or `200` (if no auth required). Connection refused (`000`) → A2A server down. Any other status → running but unexpected: log/report it, never restart.
|
Expected: `401` (auth-gated, A2A server is up and responding) or `200` (if no auth required). Connection refused (`000`) → A2A server down (INCIDENT). Any other status → running but unexpected: log/report it, never restart.
|
||||||
|
|
||||||
**C2: A2A Response Verification**
|
**C2: A2A Response Verification (requires LITELLM_KEY)**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# A2A listens on :80 inside the container and is auth-gated (401 expected unauthenticated).
|
# A2A listens on :80 inside the container and is auth-gated (401 expected unauthenticated).
|
||||||
@@ -491,15 +492,30 @@ ssh root@192.168.68.14 "docker exec agent-zero curl -s -X POST http://127.0.0.1:
|
|||||||
-d '{\"jsonrpc\":\"2.0\",\"method\":\"tasks/send\",\"params\":{\"message\":{\"role\":\"user\",\"parts\":[{\"text\":\"ping\"}]}},\"id\":1}'"
|
-d '{\"jsonrpc\":\"2.0\",\"method\":\"tasks/send\",\"params\":{\"message\":{\"role\":\"user\",\"parts\":[{\"text\":\"ping\"}]}},\"id\":1}'"
|
||||||
```
|
```
|
||||||
|
|
||||||
Expected: task ID with "working" status. Poll for completion with `tasks/get`. If 401, check LITELLM_KEY is set.
|
Expected: task ID with "working" status. Poll for completion with `tasks/get`. If 401, check LITELLM_KEY is set (this is a credential issue, NOT a server-down incident).
|
||||||
|
|
||||||
|
**C3: Public Access Path (no credential needed)**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Probes the public URL that NetBird proxies to the agent-zero container.
|
||||||
|
# This is the captain's point of view: if the captain can't reach it, it's down.
|
||||||
|
# 200/302/401 = alive, 502 = proxy's "upstream refused" page (INCIDENT),
|
||||||
|
# connection failed (000) = INCIDENT. Never restarts anything.
|
||||||
|
curl -s -o /dev/null --connect-timeout 10 --max-time 15 -w '%{http_code}' https://kagentz.sysloggh.net/
|
||||||
|
```
|
||||||
|
|
||||||
|
Expected: `200` (Agent Zero login page), `302` (redirect), or `401` (auth-gated) = alive. `502` = NetBird proxy's "upstream refused" page (the container's port 80 is not listening) = INCIDENT. `000` (connection failed) = INCIDENT. Any other status → running but unexpected: log/report it, never restart.
|
||||||
|
|
||||||
**Platform C Actions**
|
**Platform C Actions**
|
||||||
|
|
||||||
| Condition | Action |
|
| Condition | Action |
|
||||||
|-----------|--------|
|
|-----------|--------|
|
||||||
| A2A returns `000` (connection refused/timeout) | Alert only — never restart the platform; investigate the agent-zero container |
|
| C1 A2A returns `000` (connection refused/timeout) | Alert only — never restart the platform; investigate the agent-zero container |
|
||||||
| A2A returns a status other than `200`/`401` | Log/report as a warning — reported, never healed on |
|
| C1 A2A returns a status other than `200`/`401` | Log/report as a warning — reported, never healed on |
|
||||||
| LiteLLM 401 | Check API key in a2a_agent.py `LITELLM_KEY` |
|
| C2 A2A returns 401 | Check `LITELLM_KEY` is set (credential issue, NOT server-down) |
|
||||||
|
| C3 public URL returns `502` (upstream refused) | Alert only — the container's port 80 is not listening; investigate the agent-zero container |
|
||||||
|
| C3 public URL returns `000` (connection failed) | Alert only — the public path is down; investigate the NetBird proxy or the container |
|
||||||
|
| C3 public URL returns a status other than `200`/`302`/`401` | Log/report as a warning — reported, never healed on |
|
||||||
|
|
||||||
### Step 5: Global Checks
|
### Step 5: Global Checks
|
||||||
|
|
||||||
@@ -513,12 +529,19 @@ If any bot processes >50 bot-originated messages in 15min → warning.
|
|||||||
|
|
||||||
### Step 6: Compile and Report
|
### Step 6: Compile and Report
|
||||||
|
|
||||||
1. Compile all platform checks and severity
|
1. Run `scripts/zulip-monitor.sh` and take its final `Result:` line as the
|
||||||
2. Determine `overall_severity` from worst per-agent severity
|
authoritative run verdict. The verdict line is either
|
||||||
3. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart
|
`Result: ✅ 0 issues (all healthy)` or
|
||||||
4. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp
|
`Result: 🔴 INCIDENT — N issue(s) found`.
|
||||||
5. If any agent critical or >2 degraded: send relay message to user
|
2. Quote that `Result:` line verbatim in the status report. When it says
|
||||||
6. Update `last_check` timestamp in `### Maintains` snapshot
|
`INCIDENT`, the run MUST be reported as an incident — never summarised as
|
||||||
|
OK/healthy and never annotated as "expected".
|
||||||
|
3. Compile all platform checks and severity
|
||||||
|
4. Determine `overall_severity` from worst per-agent severity
|
||||||
|
5. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart
|
||||||
|
6. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp
|
||||||
|
7. If any agent critical or >2 degraded: send relay message to user
|
||||||
|
8. Update `last_check` timestamp in `### Maintains` snapshot
|
||||||
|
|
||||||
### Restart Debounce
|
### Restart Debounce
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user