diff --git a/scripts/daily-health-digest.sh b/scripts/daily-health-digest.sh deleted file mode 100755 index 7663a35..0000000 --- a/scripts/daily-health-digest.sh +++ /dev/null @@ -1,71 +0,0 @@ -#!/bin/bash -# daily-health-digest.sh — Daily Fleet Health Digest -# Runs all major health checks and summarizes results -# -# Legs: -# - infrastructure-monitoring (13 legs) -# - gpu-monitor (6 legs) -# - litellm-health (11 checks) -# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys) -# - proxmox-monitor (4 exporters) -# - pm2-self-heal (PM2 services) -# - zulip-health (Zulip mesh) -# -# Exit 0 if all pass, 1 if any fail. -# -# Run: bash scripts/daily-health-digest.sh - -set -uo pipefail - -TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') -SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)" -FAILED=() - -echo "=== Daily Health Digest — $TIMESTAMP ===" -echo "Executed from: $(pwd -P)" -echo "" - -# Run each health check and capture exit code -run_check() { - local name="$1" - local script="$2" - local runner="${3:-bash}" - local output - - echo "── Running $name ──" - output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1) - local exit_code=$? - - if [ $exit_code -eq 0 ]; then - # Extract just the summary lines (lines with ✅ or "All") - echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3 - echo "" - else - echo " ❌ $name failed (exit $exit_code)" - echo "$output" | grep -E "🔴|❌" | tail -5 - echo "" - FAILED+=("$name") - fi -} - -# Run all checks -run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh" -run_check "GPU Monitor" "scripts/gpu-monitor.sh" -run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3" -run_check "Agent Health Check" "scripts/agent-health-check.py" "python3" -run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh" -run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh" -run_check "Zulip Health" "scripts/zulip-monitor.sh" - -# Summary -echo "─────────────────────────────────────────────" -if [ ${#FAILED[@]} -eq 0 ]; then - echo "✅ ALL HEALTH CHECKS PASSED" - exit 0 -else - echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:" - for f in "${FAILED[@]}"; do - echo " - $f" - done - exit 1 -fi \ No newline at end of file diff --git a/scripts/gpu-monitor.sh b/scripts/gpu-monitor.sh deleted file mode 100755 index b7976d9..0000000 --- a/scripts/gpu-monitor.sh +++ /dev/null @@ -1,118 +0,0 @@ -#!/bin/bash -# gpu-monitor.sh — GPU Fleet Health Monitor -# Implements gpu-monitor.prose.md (check-health section) -# -# Legs: -# - GPU Monitor health (localhost:9100/health) -# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080) -# - IMPORTANT: Always probe :8080, NEVER bare :80 -# - Router unified health (192.168.68.116/health/unified, expects 301) -# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200) -# -# Exit 0 if all probes pass, 1 if any fails. -# -# Run: bash scripts/gpu-monitor.sh - -set -uo pipefail - -CT116_HOST="192.168.68.116" -CT8_HOST="192.168.68.8" -CT110_HOST="192.168.68.110" -CT15_HOST="192.168.68.15" -FAILED=() -TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') - -echo "=== GPU Monitor — $TIMESTAMP ===" -echo "Executed from: $(pwd -P)" -echo "" - -# 1. GPU Monitor health -echo " Probing GPU Monitor health..." -GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null) -if [ $? -ne 0 ]; then - echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)" - FAILED+=("gpu-monitor") -elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then - CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ') - echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)" -else - echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)" - FAILED+=("gpu-monitor") -fi - -# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80) -echo " Probing GPU host health..." - -# CT 8 (.8) -RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null) -RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]') -[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000" - -if [ "$RTX3090_CODE" = "200" ]; then - echo " ✅ GPU .8 (RTX 3090): healthy" -else - echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})" - FAILED+=("gpu-8") -fi - -# CT 110 (.110) -RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null) -RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]') -[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000" - -if [ "$RTX5070_CODE" = "200" ]; then - echo " ✅ GPU .110 (RTX 5070): healthy" -else - echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})" - FAILED+=("gpu-110") -fi - -# CT 15 (.15) - Strix Halo -STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null) -STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]') -[ -n "$STRIX_CODE" ] || STRIX_CODE="000" - -if [ "$STRIX_CODE" = "200" ]; then - echo " ✅ GPU .15 (Strix Halo): healthy" -else - echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})" - FAILED+=("gpu-15") -fi - -# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY) -echo " Probing Router unified health..." -ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null) -ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]') -[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000" - -if [ "$ROUTER_CODE" = "301" ]; then - echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)" -else - echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})" - FAILED+=("router") -fi - -# 4. LiteLLM health -echo " Probing LiteLLM health..." -LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null) -LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]') -[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000" - -if [ "$LITELLM_CODE" = "200" ]; then - echo " ✅ LiteLLM: alive" -else - echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})" - FAILED+=("litellm") -fi - -# ── Summary ───────────────────────────────────────────────────────────────── -echo "" -if [ ${#FAILED[@]} -eq 0 ]; then - echo " ✅ All legs OK" - exit 0 -else - for f in "${FAILED[@]}"; do - echo " 🔴 FAILED: $f" - done - exit 1 -fi \ No newline at end of file diff --git a/tests/test_mumuni_monitor_removal.py b/tests/test_mumuni_monitor_removal.py index df55352..6e765af 100644 --- a/tests/test_mumuni_monitor_removal.py +++ b/tests/test_mumuni_monitor_removal.py @@ -98,8 +98,8 @@ exit 0 """ CURL_STUB = r"""#!/usr/bin/env bash -# Stub curl: serve the Abiba health fixture and the Zulip server 200, and -# record every call (including notify) payloads. +# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the +# kagentz C3 public URL, and record every call (including notify) payloads. printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls" case "$*" in *:9200/health*) @@ -109,6 +109,8 @@ case "$*" in esac ;; *server_settings*) printf '%s' "$SERVER_HTTP" ;; + *kagentz.sysloggh.net*) + printf '%s' "$KAGENTZ_PUBLIC_CODE" ;; esac exit 0 """ @@ -121,7 +123,8 @@ def _write_exec(path: pathlib.Path, body: str) -> None: def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", - az_a2a_code="401", az_a2a_exit=0): + az_a2a_code="401", az_a2a_exit=0, + kagentz_public_code="302"): """Run the shipped monitor in a sandbox; return (proc, record_dir, log_path). Only the LOG constant is rewritten (to keep the run inside the worktree). @@ -154,6 +157,7 @@ def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200", "PI_HTTP": "200", "PI_BODY": CONNECTED_FIXTURE.read_text(), "SERVER_HTTP": "200", + "KAGENTZ_PUBLIC_CODE": kagentz_public_code, }) proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env, capture_output=True, text=True) @@ -169,8 +173,9 @@ def test_healthy_run_is_quiet_and_never_reaches_mumuni(tmp_path): assert "Server: ✅ HTTP 200" in log assert "Abiba: ✅ Connected" in log assert "Tanko: ✅ service=active http=200" in log - assert "kagentz: ✅ A2A alive (HTTP 401)" in log - assert "Result: ✅ All healthy" in log + assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log + assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log + assert "Result: ✅ 0 issues (all healthy)" in log # A healthy run emits no notify at all — and certainly no Mumuni one. assert proc.stdout == "" @@ -207,8 +212,8 @@ def test_failing_run_alerts_on_tanko_but_never_on_mumuni(tmp_path): # The rest of the monitor still ran alongside the failing Tanko leg. log = log_path.read_text() assert "Abiba: ✅ Connected" in log - assert "kagentz: ✅ A2A alive" in log - assert "Result: 🔴 1 issue(s) found" in log + assert "kagentz C1: ✅ A2A alive" in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path): @@ -216,9 +221,9 @@ def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path): assert proc.returncode == 0, proc.stderr log = log_path.read_text() - assert "kagentz: 🟡 A2A unexpected http=500 (running, warning)" in log - assert "kagentz: ✅ A2A alive" not in log - assert "Result: 🔴 1 issue(s) found" in log + assert "kagentz C1: 🟡 A2A unexpected http=500 (running, warning)" in log + assert "kagentz C1: ✅ A2A alive" not in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log assert "kagentz A2A server answered HTTP 500" in proc.stdout @@ -230,10 +235,10 @@ def test_a2a_connection_failure_is_down_not_unexpected(tmp_path): assert proc.returncode == 0, proc.stderr log = log_path.read_text() - assert "kagentz: ❌ A2A down (HTTP 000)" in log - assert "kagentz: ✅ A2A alive" not in log + assert "kagentz C1: ❌ A2A down (HTTP 000)" in log + assert "kagentz C1: ✅ A2A alive" not in log assert "unexpected" not in log - assert "Result: 🔴 1 issue(s) found" in log + assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log assert "kagentz A2A server DOWN (connection failed)" in proc.stdout diff --git a/tests/test_zulip_kagentz_legs.py b/tests/test_zulip_kagentz_legs.py index 0254a86..98e8dbc 100644 --- a/tests/test_zulip_kagentz_legs.py +++ b/tests/test_zulip_kagentz_legs.py @@ -30,8 +30,6 @@ import pathlib import stat import subprocess -import pytest - ROOT = pathlib.Path(__file__).resolve().parents[1] ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh" CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json" @@ -209,25 +207,3 @@ def test_c1_000_is_incident(tmp_path): assert "all healthy" not in log # The notify fired for the A2A down. assert "kagentz A2A server DOWN" in proc.stdout - - -# ── Prose assertions (kept as secondary, do not replace behavioural) ── - -def test_prose_c1_no_credential_needed(): - """C1 header should say 'no credential needed'.""" - prose = (ROOT / "zulip-health.prose.md").read_text() - assert "C1: A2A Server Health (no credential needed)" in prose - - -def test_prose_c2_requires_litellm_key(): - """C2 header should say 'requires LITELLM_KEY'.""" - prose = (ROOT / "zulip-health.prose.md").read_text() - assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose - - -def test_prose_c3_public_access_path(): - """C3 section should exist and document 502/000 as incidents.""" - prose = (ROOT / "zulip-health.prose.md").read_text() - assert "C3: Public Access Path" in prose - assert "https://kagentz.sysloggh.net/" in prose - assert "502" in prose diff --git a/zulip-health.prose.md b/zulip-health.prose.md index 097e2ec..bce42d6 100644 --- a/zulip-health.prose.md +++ b/zulip-health.prose.md @@ -528,12 +528,19 @@ If any bot processes >50 bot-originated messages in 15min → warning. ### Step 6: Compile and Report -1. Compile all platform checks and severity -2. Determine `overall_severity` from worst per-agent severity -3. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart -4. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp -5. If any agent critical or >2 degraded: send relay message to user -6. Update `last_check` timestamp in `### Maintains` snapshot +1. Run `scripts/zulip-monitor.sh` and take its final `Result:` line as the + authoritative run verdict. The verdict line is either + `Result: ✅ 0 issues (all healthy)` or + `Result: 🔴 INCIDENT — N issue(s) found`. +2. Quote that `Result:` line verbatim in the status report. When it says + `INCIDENT`, the run MUST be reported as an incident — never summarised as + OK/healthy and never annotated as "expected". +3. Compile all platform checks and severity +4. Determine `overall_severity` from worst per-agent severity +5. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart +6. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp +7. If any agent critical or >2 degraded: send relay message to user +8. Update `last_check` timestamp in `### Maintains` snapshot ### Restart Debounce