#!/bin/bash # /root/scripts/zulip-monitor.sh — Zulip Mesh Health Monitor # Implements zulip-health.prose.md v2 # Runs every 15 min via cron. Alerts via Telegram. set -euo pipefail ZULIP_SITE="https://chat.sysloggh.net" ZULIP_EMAIL="abiba-bot@chat.sysloggh.net" ZULIP_KEY="cKTDMZAPW08dk3zl05sStzO7HRztzyn8" OWNER_ZULIP_ID="9" LOG="/root/zulip-health-monitor.log" TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') ISSUES=0 echo "=== Zulip Health Check — $TIMESTAMP ===" >> "$LOG" notify() { local severity="$1" msg="$2" echo "[$severity] $msg" # Zulip DM to owner local content="${severity} Zulip Monitor: ${msg}" local form="type=private&to=%5B${OWNER_ZULIP_ID}%5D&content=$(python3 -c "import urllib.parse; print(urllib.parse.quote('''${content}'''))")" curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \ -u "${ZULIP_EMAIL}:${ZULIP_KEY}" \ -d "${form}" > /dev/null 2>&1 || true # Zulip stream post to #agent-hub on topic 'zulip-health' local stream_content="${severity} Zulip Monitor: ${msg}" curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \ -u "${ZULIP_EMAIL}:${ZULIP_KEY}" \ -d "type=stream\&to=%5B7%5D\&topic=zulip-health\&content=$(python3 -c "import urllib.parse; print(urllib.parse.quote(str()))")" \ > /dev/null 2>&1 || true } # ── Global: Zulip Server ── SERVER_CODE=$(curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 \ https://chat.sysloggh.net/api/v1/server_settings \ -u 'abiba-bot@chat.sysloggh.net:cKTDMZAPW08dk3zl05sStzO7HRztzyn8' 2>/dev/null || echo "000") if [ "$SERVER_CODE" != "200" ]; then notify "🔴" "Zulip server returned HTTP $SERVER_CODE" ISSUES=$((ISSUES + 1)) else echo " Server: ✅ HTTP 200" >> "$LOG" fi # ── Platform A: pi (Abiba) ── PI_HEALTH=$(curl -sf --connect-timeout 5 http://localhost:9200/health 2>/dev/null || echo "{}") PI_CONNECTED=$(echo "$PI_HEALTH" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('connected',False))" 2>/dev/null) PI_ERROR=$(echo "$PI_HEALTH" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('last_error') or '')" 2>/dev/null) PI_RETRIES=$(echo "$PI_HEALTH" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('retry_count',0))" 2>/dev/null) if [ "$PI_CONNECTED" != "True" ]; then notify "🔴" "Abiba pi extension DISCONNECTED — restarting" pm2 restart abiba-zulip 2>/dev/null || true ISSUES=$((ISSUES + 1)) echo " Abiba: ❌ Disconnected — restarted" >> "$LOG" elif [ -n "$PI_ERROR" ]; then notify "🟡" "Abiba pi extension error: ${PI_ERROR:0:100}" echo " Abiba: 🟡 Error: ${PI_ERROR:0:100}" >> "$LOG" elif [ "$PI_RETRIES" -ge 3 ]; then notify "🟡" "Abiba pi extension: $PI_RETRIES retries — restarting" pm2 restart abiba-zulip 2>/dev/null || true echo " Abiba: 🟡 $PI_RETRIES retries — restarted" >> "$LOG" else echo " Abiba: ✅ Connected (processed=$(echo "$PI_HEALTH" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('messages_processed',0))" 2>/dev/null))" >> "$LOG" fi # ── Platform B: Tanko (DSH dsh-web on amdpve CT 112) ── # Direct SSH to 192.168.68.122 is not a dependency of this monitor — per-worker # key availability varies — so probes run from the amdpve vantage via `pct exec`. # Tanko's Zulip gateway runs as the dsh-web systemd unit inside CT 112 on amdpve # (192.168.68.15). The gateway binds 127.0.0.1:3080 loopback-only by design — a # remote :3080 probe is refused and is NOT a fault. TANKO_SVC=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \ "pct exec 112 -- systemctl is-active dsh-web" 2>/dev/null || true) [ -n "$TANKO_SVC" ] || TANKO_SVC="unknown" TANKO_HTTP=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \ "pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/" 2>/dev/null || true) [ -n "$TANKO_HTTP" ] || TANKO_HTTP="000" if [ "$TANKO_SVC" != "active" ]; then notify "🔴" "Tanko (DSH dsh-web) service state: $TANKO_SVC — needs restart" ISSUES=$((ISSUES + 1)) echo " Tanko: ❌ service=$TANKO_SVC" >> "$LOG" elif [ "$TANKO_HTTP" = "000" ]; then notify "🔴" "Tanko (DSH dsh-web) HTTP :3080 connection refused/timeout — needs restart" ISSUES=$((ISSUES + 1)) echo " Tanko: ❌ http=000 (refused/timeout)" >> "$LOG" else case "$TANKO_HTTP" in 200|301|302|307|308|401|403) echo " Tanko: ✅ service=active http=$TANKO_HTTP" >> "$LOG" ;; *) notify "🟡" "Tanko (DSH dsh-web) HTTP :3080 answered $TANKO_HTTP — running, unexpected status" echo " Tanko: 🟡 service=active http=$TANKO_HTTP (running, warning)" >> "$LOG" ;; esac fi # ── Platform B: Hermes (Mumuni) ── MUMUNI_STATE=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.24 \ "cat ~/.hermes/gateway_state.json 2>/dev/null" 2>/dev/null || echo "{}") MUMUNI_ZULIP=$(echo "$MUMUNI_STATE" | python3 -c " import sys,json d=json.load(sys.stdin) p=d.get('platforms',{}).get('zulip',{}) print(p.get('state','unknown')) " 2>/dev/null) if [ "$MUMUNI_ZULIP" != "connected" ]; then notify "🔴" "Mumuni (Hermes) Zulip state: $MUMUNI_ZULIP" ISSUES=$((ISSUES + 1)) echo " Mumuni: ❌ state=$MUMUNI_ZULIP" >> "$LOG" else echo " Mumuni: ✅ Zulip connected" >> "$LOG" fi # ── Platform C: Agent Zero (kagentz) ── AZ_A2A=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \ "docker exec agent-zero curl -s --connect-timeout 5 http://127.0.0.1:8001/.well-known/agent.json 2>/dev/null" 2>/dev/null || echo "") AZ_ALIVE=$(echo "$AZ_A2A" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('name',''))" 2>/dev/null) if [ "$AZ_ALIVE" != "kagentz" ]; then notify "🔴" "kagentz A2A server DOWN — restarting" ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \ "docker exec agent-zero bash -c 'pkill -9 -f a2a_agent; sleep 1; cd /a0 && /opt/venv-a0/bin/python3 -u /a0/usr/a2a_agent.py > /tmp/a2a.log 2>&1 &'" 2>/dev/null || true ISSUES=$((ISSUES + 1)) echo " kagentz: ❌ A2A down — restarted" >> "$LOG" else echo " kagentz: ✅ A2A alive" >> "$LOG" # Check adapter process AZ_ADAPTER=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \ "docker exec agent-zero ps aux 2>/dev/null | grep adapter | grep -v grep | wc -l" 2>/dev/null || echo "0") if [ "$AZ_ADAPTER" -lt 1 ]; then notify "🔴" "kagentz Zulip adapter DOWN — restarting" ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.14 \ "docker exec agent-zero bash -c 'cd /a0/usr/kagentz-zulip && ZULIP_SITE=https://chat.sysloggh.net ZULIP_EMAIL=kagentz-bot@chat.sysloggh.net ZULIP_API_KEY=E9q9PXJTxftPYBkb5pBDWupDO7KK21ty ZULIP_AGENT_NAME=kagentz A2A_URL=http://localhost:8001/a2a A2A_TOKEN=8zNgdOEXzYxjQvTl /opt/venv-a0/bin/python3 -u adapter.py > /tmp/zulip-adapter.log 2>&1 &'" 2>/dev/null || true ISSUES=$((ISSUES + 1)) echo " kagentz: ❌ Adapter down — restarted" >> "$LOG" else echo " kagentz: ✅ Adapter running" >> "$LOG" fi fi # ── Summary ── if [ "$ISSUES" -eq 0 ]; then echo " Result: ✅ All healthy" >> "$LOG" else echo " Result: 🔴 $ISSUES issue(s) found" >> "$LOG" notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log" fi echo "" >> "$LOG"