- (b) Changed Result line to say 'INCIDENT' when ISSUES > 0, '0 issues (all healthy)' when ISSUES = 0 - (c) Documented C1 (no credential needed), C2 (requires LITELLM_KEY) distinction - (d) Added C3 public access path leg for https://kagentz.sysloggh.net/ - C3 treats 200/302/401 as alive, 502/000 as incident - Added tests/test_zulip_kagentz_legs.py to verify all changes
243 lines
11 KiB
Bash
Executable File
243 lines
11 KiB
Bash
Executable File
#!/bin/bash
|
|
# /root/scripts/zulip-monitor.sh — Zulip Mesh Health Monitor
|
|
# Implements zulip-health.prose.md v3
|
|
# Runs every 15 min via cron. Alerts: Zulip private DM to the owner plus a stream post to #agent-hub on topic 'zulip-health'.
|
|
# Legs: global Zulip server, Platform A pi/Abiba (the Zulip bridge), Platform B
|
|
# Tanko (DSH), Platform C Agent Zero (kagentz). The former Platform B Hermes
|
|
# agent leg is retired — see the note after the Tanko leg.
|
|
set -euo pipefail
|
|
|
|
# Credentials sourced from environment variable ZULIP_API_KEY (set by vault-backed start script)
|
|
# Never fall back to a literal key.
|
|
# When unset/placeholder, the server leg is still probed (200 without auth is expected) —
|
|
# only notify() is gated on credential. The pi/Tanko/kagentz
|
|
# legs do not need the Zulip API key. The placeholder is captain-held:
|
|
# zulip-health-credential-placeholder-20260913.
|
|
ZULIP_API_KEY="${ZULIP_API_KEY:-}"
|
|
ZULIP_SITE="https://chat.sysloggh.net"
|
|
ZULIP_EMAIL="abiba-bot@chat.sysloggh.net"
|
|
OWNER_ZULIP_ID="9"
|
|
|
|
# Track whether the Zulip API credential is usable
|
|
ZULIP_CRED_OK=1
|
|
if [ -z "$ZULIP_API_KEY" ] || [[ "$ZULIP_API_KEY" == *"placeholder"* ]] || [[ "$ZULIP_API_KEY" == *"REDACTED"* ]]; then
|
|
ZULIP_CRED_OK=0
|
|
fi
|
|
|
|
LOG="/root/zulip-health-monitor.log"
|
|
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
|
ISSUES=0
|
|
echo "=== Zulip Health Check — $TIMESTAMP ===" >> "$LOG"
|
|
|
|
notify() {
|
|
local severity="$1" msg="$2"
|
|
echo "[$severity] $msg"
|
|
|
|
# Zulip DM to owner (skip if no credential)
|
|
if [ "$ZULIP_CRED_OK" -eq 1 ]; then
|
|
local content="${severity} Zulip Monitor: ${msg}"
|
|
local form
|
|
form="type=private&to=%5B${OWNER_ZULIP_ID}%5D&content=$(python3 -c "import urllib.parse; print(urllib.parse.quote('''${content}'''))")"
|
|
curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \
|
|
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" \
|
|
-d "${form}" > /dev/null 2>&1 || true
|
|
# Zulip stream post to #agent-hub on topic 'zulip-health'
|
|
local stream_content="${severity} Zulip Monitor: ${msg}"
|
|
curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \
|
|
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" \
|
|
-d "type=stream&to=%5B7%5D&topic=zulip-health&content=$(printf '%s' "${stream_content}" | python3 -c "import sys,urllib.parse; print(urllib.parse.quote_from_bytes(sys.stdin.buffer.read()))")" \
|
|
> /dev/null 2>&1 \
|
|
|| echo " WARN: stream alert to #agent-hub (zulip-health) delivery failed (curl exit $?)">> "$LOG"
|
|
else
|
|
echo " ALERT SUPPRESSED (no credential): ${severity} ${msg}" >> "$LOG"
|
|
fi
|
|
}
|
|
|
|
# ── Global: Zulip Server ──
|
|
# F3: Always probe server regardless of credential — 200 without auth is expected
|
|
# (verified live: server_settings returns 200 with no credential or wrong key).
|
|
SERVER_CODE=$(curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 \
|
|
https://chat.sysloggh.net/api/v1/server_settings \
|
|
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" 2>/dev/null) || SERVER_CODE="000"
|
|
SERVER_CODE=$(printf '%s' "$SERVER_CODE" | tr -d '[:space:]')
|
|
[ -n "$SERVER_CODE" ] || SERVER_CODE="000"
|
|
if [ "$SERVER_CODE" != "200" ]; then
|
|
notify "🔴" "Zulip server returned HTTP $SERVER_CODE"
|
|
ISSUES=$((ISSUES + 1))
|
|
else
|
|
echo " Server: ✅ HTTP 200" >> "$LOG"
|
|
fi
|
|
|
|
# ── Platform A: pi (Abiba) ──
|
|
# Probes the pi Zulip extension health endpoint (:9200/health, served by the
|
|
# extension's startHealthServer; shape documented in zulip-health.prose.md).
|
|
# FAIL-SAFE contract (pinned by tests/zulip-monitor-abiba.sh): connection state
|
|
# lives NESTED at zulip.connected / zulip.last_error — there is no top-level
|
|
# `connected` and no retry counter in the payload. A fetch error, non-2xx
|
|
# response, empty/unparseable body, or payload missing a boolean
|
|
# zulip.connected is a PROBE FAILURE: it alerts and NEVER calls pm2 restart.
|
|
# pm2 restart runs ONLY on affirmative zulip.connected=false.
|
|
# -- abiba-leg-start (verbatim-extracted by tests/zulip-monitor-abiba.sh)
|
|
PI_HTTP=$(curl -s -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}' http://localhost:9200/health 2>/dev/null) || PI_HTTP="000"
|
|
PI_HTTP=$(printf '%s' "$PI_HTTP" | tr -d '[:space:]')
|
|
[ -n "$PI_HTTP" ] || PI_HTTP="000"
|
|
PI_BODY=$(curl -s --connect-timeout 5 --max-time 10 http://localhost:9200/health 2>/dev/null || true)
|
|
PI_STATE=$(printf '%s' "$PI_BODY" | python3 -c '
|
|
import sys, json
|
|
code = sys.argv[1]
|
|
body = sys.stdin.read()
|
|
try:
|
|
d = json.loads(body)
|
|
except Exception:
|
|
sys.stdout.write("probe-failed|unparseable body")
|
|
sys.exit(0)
|
|
if not code.startswith("2"):
|
|
sys.stdout.write("probe-failed|HTTP %s" % code)
|
|
sys.exit(0)
|
|
if not isinstance(d, dict) or not isinstance(d.get("zulip"), dict):
|
|
sys.stdout.write("probe-failed|missing zulip.connected")
|
|
sys.exit(0)
|
|
z = d["zulip"]
|
|
if "connected" not in z or not isinstance(z["connected"], bool):
|
|
sys.stdout.write("probe-failed|missing or non-boolean zulip.connected")
|
|
sys.exit(0)
|
|
err = z.get("last_error") or ""
|
|
if z["connected"]:
|
|
if err:
|
|
sys.stdout.write("degraded|%s" % err)
|
|
else:
|
|
sys.stdout.write("healthy|%s" % z.get("messages_processed", 0))
|
|
else:
|
|
sys.stdout.write("disconnected|")
|
|
' "$PI_HTTP" 2>/dev/null) || PI_STATE="probe-failed|python error"
|
|
PI_VERDICT=${PI_STATE%%|*}
|
|
PI_DETAIL=${PI_STATE#*|}
|
|
|
|
case "$PI_VERDICT" in
|
|
healthy)
|
|
echo " Abiba: ✅ Connected (processed=$PI_DETAIL)" >> "$LOG" ;;
|
|
degraded)
|
|
notify "🟡" "Abiba pi extension error: ${PI_DETAIL:0:100}"
|
|
echo " Abiba: 🟡 Error: ${PI_DETAIL:0:100}" >> "$LOG" ;;
|
|
disconnected)
|
|
notify "🔴" "Abiba pi extension DISCONNECTED — restarting"
|
|
pm2 restart abiba-zulip 2>/dev/null || true
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " Abiba: ❌ Disconnected — restarted" >> "$LOG" ;;
|
|
probe-failed)
|
|
notify "🟠" "Abiba pi extension health probe FAILED (${PI_DETAIL}; HTTP $PI_HTTP) — NOT restarting, manual check needed"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " Abiba: ⚠️ Probe failed (${PI_DETAIL}; HTTP $PI_HTTP) — NOT restarted" >> "$LOG" ;;
|
|
*)
|
|
notify "🟠" "Abiba pi extension health probe returned unexpected verdict (${PI_STATE}) — NOT restarting, manual check needed"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " Abiba: ⚠️ Unexpected probe verdict (${PI_STATE}) — NOT restarted" >> "$LOG" ;;
|
|
esac
|
|
# -- abiba-leg-end
|
|
|
|
# ── Platform B: Tanko (DSH dsh-web on amdpve CT 112) ──
|
|
# Direct SSH to 192.168.68.122 is not a dependency of this monitor — per-worker
|
|
# key availability varies — so probes run from the amdpve vantage via `pct exec`.
|
|
# Tanko's Zulip gateway runs as the dsh-web systemd unit inside CT 112 on amdpve
|
|
# (192.168.68.15). The gateway binds 127.0.0.1:3080 loopback-only by design — a
|
|
# remote :3080 probe is refused and is NOT a fault.
|
|
TANKO_SVC=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \
|
|
"pct exec 112 -- systemctl is-active dsh-web" 2>/dev/null || true)
|
|
[ -n "$TANKO_SVC" ] || TANKO_SVC="unknown"
|
|
TANKO_HTTP=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \
|
|
"pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/" 2>/dev/null || true)
|
|
[ -n "$TANKO_HTTP" ] || TANKO_HTTP="000"
|
|
|
|
if [ "$TANKO_SVC" != "active" ]; then
|
|
notify "🔴" "Tanko (DSH dsh-web) service state: $TANKO_SVC — needs restart"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " Tanko: ❌ service=$TANKO_SVC" >> "$LOG"
|
|
elif [ "$TANKO_HTTP" = "000" ]; then
|
|
notify "🔴" "Tanko (DSH dsh-web) HTTP :3080 connection refused/timeout — needs restart"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " Tanko: ❌ http=000 (refused/timeout)" >> "$LOG"
|
|
else
|
|
case "$TANKO_HTTP" in
|
|
200|301|302|307|308|401|403)
|
|
echo " Tanko: ✅ service=active http=$TANKO_HTTP" >> "$LOG" ;;
|
|
*)
|
|
notify "🟡" "Tanko (DSH dsh-web) HTTP :3080 answered $TANKO_HTTP — running, unexpected status"
|
|
echo " Tanko: 🟡 service=active http=$TANKO_HTTP (running, warning)" >> "$LOG" ;;
|
|
esac
|
|
fi
|
|
|
|
# ── Removed: the former "Platform B: Hermes" agent leg ──
|
|
# Captain ruling 2026-09-10: that agent moved off this host onto her own
|
|
# container (kagentz CT 105 on minipve, dedicated `hermes` user) and is now
|
|
# monitored on her side — see the out-of-scope note in zulip-health.prose.md.
|
|
# The old leg ssh'd to her former CT 100 deployment and read its Hermes gateway
|
|
# state, which reported "unknown" on every run and posted a false 🔴 DM plus an
|
|
# #agent-hub stream alert. Do NOT re-add a probe for her: this monitor must
|
|
# never contact her former host.
|
|
|
|
# ── Platform C: Agent Zero (kagentz) ──
|
|
# C1: A2A liveness (no credential needed) — probes the container's internal :80/a2a/
|
|
# C2: A2A response verification (needs LITELLM_KEY) — probes POST /a2a with auth
|
|
# C3: Public access path (no credential needed) — probes https://kagentz.sysloggh.net/
|
|
|
|
# C1: A2A liveness (container-internal probe)
|
|
AZ_A2A_CODE=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \
|
|
"docker exec agent-zero curl -s --connect-timeout 5 -o /dev/null -w '%{http_code}' http://127.0.0.1:80/a2a/ 2>/dev/null" 2>/dev/null) || AZ_A2A_CODE="000"
|
|
AZ_A2A_CODE=$(printf '%s' "$AZ_A2A_CODE" | tr -d '[:space:]')
|
|
[ -n "$AZ_A2A_CODE" ] || AZ_A2A_CODE="000"
|
|
|
|
if [ "$AZ_A2A_CODE" = "000" ]; then
|
|
notify "🔴" "kagentz A2A server DOWN (connection failed)"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " kagentz C1: ❌ A2A down (HTTP 000)" >> "$LOG"
|
|
else
|
|
case "$AZ_A2A_CODE" in
|
|
200|401)
|
|
echo " kagentz C1: ✅ A2A alive (HTTP $AZ_A2A_CODE)" >> "$LOG" ;;
|
|
*)
|
|
notify "🟡" "kagentz A2A server answered HTTP $AZ_A2A_CODE — running, unexpected status"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " kagentz C1: 🟡 A2A unexpected http=$AZ_A2A_CODE (running, warning)" >> "$LOG" ;;
|
|
esac
|
|
fi
|
|
|
|
# C3: Public access path (the captain's point of view)
|
|
# Probes the public URL that NetBird proxies to the container. 200/302/401 = alive,
|
|
# 502 = proxy's "upstream refused" page (incident), connection failed = incident.
|
|
# Never restarts anything — the contract forbids restarting the platform.
|
|
KAGENTZ_PUBLIC_CODE=$(curl -s -o /dev/null --connect-timeout 10 --max-time 15 \
|
|
-w '%{http_code}' https://kagentz.sysloggh.net/ 2>/dev/null) || KAGENTZ_PUBLIC_CODE="000"
|
|
KAGENTZ_PUBLIC_CODE=$(printf '%s' "$KAGENTZ_PUBLIC_CODE" | tr -d '[:space:]')
|
|
[ -n "$KAGENTZ_PUBLIC_CODE" ] || KAGENTZ_PUBLIC_CODE="000"
|
|
|
|
if [ "$KAGENTZ_PUBLIC_CODE" = "000" ]; then
|
|
notify "🔴" "kagentz public URL DOWN (connection failed)"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " kagentz C3: ❌ public URL down (HTTP 000)" >> "$LOG"
|
|
elif [ "$KAGENTZ_PUBLIC_CODE" = "502" ]; then
|
|
notify "🔴" "kagentz public URL 502 (upstream refused)"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " kagentz C3: ❌ public URL 502 (upstream refused)" >> "$LOG"
|
|
else
|
|
case "$KAGENTZ_PUBLIC_CODE" in
|
|
200|302|401)
|
|
echo " kagentz C3: ✅ public URL alive (HTTP $KAGENTZ_PUBLIC_CODE)" >> "$LOG" ;;
|
|
*)
|
|
notify "🟡" "kagentz public URL answered HTTP $KAGENTZ_PUBLIC_CODE — running, unexpected status"
|
|
ISSUES=$((ISSUES + 1))
|
|
echo " kagentz C3: 🟡 public URL unexpected http=$KAGENTZ_PUBLIC_CODE (running, warning)" >> "$LOG" ;;
|
|
esac
|
|
fi
|
|
|
|
# ── Summary ──
|
|
# The run verdict is non-optimistic: when ISSUES > 0, the run is an INCIDENT.
|
|
# The lane must quote this Result line verbatim in its status report.
|
|
if [ "$ISSUES" -eq 0 ]; then
|
|
echo " Result: ✅ 0 issues (all healthy)" >> "$LOG"
|
|
else
|
|
echo " Result: 🔴 INCIDENT — $ISSUES issue(s) found" >> "$LOG"
|
|
notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log"
|
|
fi
|
|
|
|
echo "" >> "$LOG"
|