no-mistakes(review): Fix review findings: test regression, contract verdict, scope trim
This commit is contained in:
@@ -1,71 +0,0 @@
|
||||
#!/bin/bash
|
||||
# daily-health-digest.sh — Daily Fleet Health Digest
|
||||
# Runs all major health checks and summarizes results
|
||||
#
|
||||
# Legs:
|
||||
# - infrastructure-monitoring (13 legs)
|
||||
# - gpu-monitor (6 legs)
|
||||
# - litellm-health (11 checks)
|
||||
# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys)
|
||||
# - proxmox-monitor (4 exporters)
|
||||
# - pm2-self-heal (PM2 services)
|
||||
# - zulip-health (Zulip mesh)
|
||||
#
|
||||
# Exit 0 if all pass, 1 if any fail.
|
||||
#
|
||||
# Run: bash scripts/daily-health-digest.sh
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
||||
SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
FAILED=()
|
||||
|
||||
echo "=== Daily Health Digest — $TIMESTAMP ==="
|
||||
echo "Executed from: $(pwd -P)"
|
||||
echo ""
|
||||
|
||||
# Run each health check and capture exit code
|
||||
run_check() {
|
||||
local name="$1"
|
||||
local script="$2"
|
||||
local runner="${3:-bash}"
|
||||
local output
|
||||
|
||||
echo "── Running $name ──"
|
||||
output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1)
|
||||
local exit_code=$?
|
||||
|
||||
if [ $exit_code -eq 0 ]; then
|
||||
# Extract just the summary lines (lines with ✅ or "All")
|
||||
echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3
|
||||
echo ""
|
||||
else
|
||||
echo " ❌ $name failed (exit $exit_code)"
|
||||
echo "$output" | grep -E "🔴|❌" | tail -5
|
||||
echo ""
|
||||
FAILED+=("$name")
|
||||
fi
|
||||
}
|
||||
|
||||
# Run all checks
|
||||
run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh"
|
||||
run_check "GPU Monitor" "scripts/gpu-monitor.sh"
|
||||
run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3"
|
||||
run_check "Agent Health Check" "scripts/agent-health-check.py" "python3"
|
||||
run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh"
|
||||
run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh"
|
||||
run_check "Zulip Health" "scripts/zulip-monitor.sh"
|
||||
|
||||
# Summary
|
||||
echo "─────────────────────────────────────────────"
|
||||
if [ ${#FAILED[@]} -eq 0 ]; then
|
||||
echo "✅ ALL HEALTH CHECKS PASSED"
|
||||
exit 0
|
||||
else
|
||||
echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:"
|
||||
for f in "${FAILED[@]}"; do
|
||||
echo " - $f"
|
||||
done
|
||||
exit 1
|
||||
fi
|
||||
@@ -1,118 +0,0 @@
|
||||
#!/bin/bash
|
||||
# gpu-monitor.sh — GPU Fleet Health Monitor
|
||||
# Implements gpu-monitor.prose.md (check-health section)
|
||||
#
|
||||
# Legs:
|
||||
# - GPU Monitor health (localhost:9100/health)
|
||||
# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080)
|
||||
# - IMPORTANT: Always probe :8080, NEVER bare :80
|
||||
# - Router unified health (192.168.68.116/health/unified, expects 301)
|
||||
# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200)
|
||||
#
|
||||
# Exit 0 if all probes pass, 1 if any fails.
|
||||
#
|
||||
# Run: bash scripts/gpu-monitor.sh
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
CT116_HOST="192.168.68.116"
|
||||
CT8_HOST="192.168.68.8"
|
||||
CT110_HOST="192.168.68.110"
|
||||
CT15_HOST="192.168.68.15"
|
||||
FAILED=()
|
||||
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
||||
|
||||
echo "=== GPU Monitor — $TIMESTAMP ==="
|
||||
echo "Executed from: $(pwd -P)"
|
||||
echo ""
|
||||
|
||||
# 1. GPU Monitor health
|
||||
echo " Probing GPU Monitor health..."
|
||||
GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null)
|
||||
if [ $? -ne 0 ]; then
|
||||
echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)"
|
||||
FAILED+=("gpu-monitor")
|
||||
elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then
|
||||
CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ')
|
||||
echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)"
|
||||
else
|
||||
echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)"
|
||||
FAILED+=("gpu-monitor")
|
||||
fi
|
||||
|
||||
# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80)
|
||||
echo " Probing GPU host health..."
|
||||
|
||||
# CT 8 (.8)
|
||||
RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null)
|
||||
RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]')
|
||||
[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000"
|
||||
|
||||
if [ "$RTX3090_CODE" = "200" ]; then
|
||||
echo " ✅ GPU .8 (RTX 3090): healthy"
|
||||
else
|
||||
echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})"
|
||||
FAILED+=("gpu-8")
|
||||
fi
|
||||
|
||||
# CT 110 (.110)
|
||||
RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null)
|
||||
RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]')
|
||||
[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000"
|
||||
|
||||
if [ "$RTX5070_CODE" = "200" ]; then
|
||||
echo " ✅ GPU .110 (RTX 5070): healthy"
|
||||
else
|
||||
echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})"
|
||||
FAILED+=("gpu-110")
|
||||
fi
|
||||
|
||||
# CT 15 (.15) - Strix Halo
|
||||
STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null)
|
||||
STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]')
|
||||
[ -n "$STRIX_CODE" ] || STRIX_CODE="000"
|
||||
|
||||
if [ "$STRIX_CODE" = "200" ]; then
|
||||
echo " ✅ GPU .15 (Strix Halo): healthy"
|
||||
else
|
||||
echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})"
|
||||
FAILED+=("gpu-15")
|
||||
fi
|
||||
|
||||
# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY)
|
||||
echo " Probing Router unified health..."
|
||||
ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null)
|
||||
ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]')
|
||||
[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000"
|
||||
|
||||
if [ "$ROUTER_CODE" = "301" ]; then
|
||||
echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)"
|
||||
else
|
||||
echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})"
|
||||
FAILED+=("router")
|
||||
fi
|
||||
|
||||
# 4. LiteLLM health
|
||||
echo " Probing LiteLLM health..."
|
||||
LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null)
|
||||
LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]')
|
||||
[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000"
|
||||
|
||||
if [ "$LITELLM_CODE" = "200" ]; then
|
||||
echo " ✅ LiteLLM: alive"
|
||||
else
|
||||
echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})"
|
||||
FAILED+=("litellm")
|
||||
fi
|
||||
|
||||
# ── Summary ─────────────────────────────────────────────────────────────────
|
||||
echo ""
|
||||
if [ ${#FAILED[@]} -eq 0 ]; then
|
||||
echo " ✅ All legs OK"
|
||||
exit 0
|
||||
else
|
||||
for f in "${FAILED[@]}"; do
|
||||
echo " 🔴 FAILED: $f"
|
||||
done
|
||||
exit 1
|
||||
fi
|
||||
@@ -98,8 +98,8 @@ exit 0
|
||||
"""
|
||||
|
||||
CURL_STUB = r"""#!/usr/bin/env bash
|
||||
# Stub curl: serve the Abiba health fixture and the Zulip server 200, and
|
||||
# record every call (including notify) payloads.
|
||||
# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the
|
||||
# kagentz C3 public URL, and record every call (including notify) payloads.
|
||||
printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls"
|
||||
case "$*" in
|
||||
*:9200/health*)
|
||||
@@ -109,6 +109,8 @@ case "$*" in
|
||||
esac ;;
|
||||
*server_settings*)
|
||||
printf '%s' "$SERVER_HTTP" ;;
|
||||
*kagentz.sysloggh.net*)
|
||||
printf '%s' "$KAGENTZ_PUBLIC_CODE" ;;
|
||||
esac
|
||||
exit 0
|
||||
"""
|
||||
@@ -121,7 +123,8 @@ def _write_exec(path: pathlib.Path, body: str) -> None:
|
||||
|
||||
|
||||
def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
|
||||
az_a2a_code="401", az_a2a_exit=0):
|
||||
az_a2a_code="401", az_a2a_exit=0,
|
||||
kagentz_public_code="302"):
|
||||
"""Run the shipped monitor in a sandbox; return (proc, record_dir, log_path).
|
||||
|
||||
Only the LOG constant is rewritten (to keep the run inside the worktree).
|
||||
@@ -154,6 +157,7 @@ def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
|
||||
"PI_HTTP": "200",
|
||||
"PI_BODY": CONNECTED_FIXTURE.read_text(),
|
||||
"SERVER_HTTP": "200",
|
||||
"KAGENTZ_PUBLIC_CODE": kagentz_public_code,
|
||||
})
|
||||
proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env,
|
||||
capture_output=True, text=True)
|
||||
@@ -169,8 +173,9 @@ def test_healthy_run_is_quiet_and_never_reaches_mumuni(tmp_path):
|
||||
assert "Server: ✅ HTTP 200" in log
|
||||
assert "Abiba: ✅ Connected" in log
|
||||
assert "Tanko: ✅ service=active http=200" in log
|
||||
assert "kagentz: ✅ A2A alive (HTTP 401)" in log
|
||||
assert "Result: ✅ All healthy" in log
|
||||
assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log
|
||||
assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log
|
||||
assert "Result: ✅ 0 issues (all healthy)" in log
|
||||
|
||||
# A healthy run emits no notify at all — and certainly no Mumuni one.
|
||||
assert proc.stdout == ""
|
||||
@@ -207,8 +212,8 @@ def test_failing_run_alerts_on_tanko_but_never_on_mumuni(tmp_path):
|
||||
# The rest of the monitor still ran alongside the failing Tanko leg.
|
||||
log = log_path.read_text()
|
||||
assert "Abiba: ✅ Connected" in log
|
||||
assert "kagentz: ✅ A2A alive" in log
|
||||
assert "Result: 🔴 1 issue(s) found" in log
|
||||
assert "kagentz C1: ✅ A2A alive" in log
|
||||
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
|
||||
|
||||
|
||||
def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
|
||||
@@ -216,9 +221,9 @@ def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
|
||||
assert proc.returncode == 0, proc.stderr
|
||||
log = log_path.read_text()
|
||||
|
||||
assert "kagentz: 🟡 A2A unexpected http=500 (running, warning)" in log
|
||||
assert "kagentz: ✅ A2A alive" not in log
|
||||
assert "Result: 🔴 1 issue(s) found" in log
|
||||
assert "kagentz C1: 🟡 A2A unexpected http=500 (running, warning)" in log
|
||||
assert "kagentz C1: ✅ A2A alive" not in log
|
||||
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
|
||||
assert "kagentz A2A server answered HTTP 500" in proc.stdout
|
||||
|
||||
|
||||
@@ -230,10 +235,10 @@ def test_a2a_connection_failure_is_down_not_unexpected(tmp_path):
|
||||
assert proc.returncode == 0, proc.stderr
|
||||
log = log_path.read_text()
|
||||
|
||||
assert "kagentz: ❌ A2A down (HTTP 000)" in log
|
||||
assert "kagentz: ✅ A2A alive" not in log
|
||||
assert "kagentz C1: ❌ A2A down (HTTP 000)" in log
|
||||
assert "kagentz C1: ✅ A2A alive" not in log
|
||||
assert "unexpected" not in log
|
||||
assert "Result: 🔴 1 issue(s) found" in log
|
||||
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
|
||||
assert "kagentz A2A server DOWN (connection failed)" in proc.stdout
|
||||
|
||||
|
||||
|
||||
@@ -30,8 +30,6 @@ import pathlib
|
||||
import stat
|
||||
import subprocess
|
||||
|
||||
import pytest
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh"
|
||||
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
|
||||
@@ -209,25 +207,3 @@ def test_c1_000_is_incident(tmp_path):
|
||||
assert "all healthy" not in log
|
||||
# The notify fired for the A2A down.
|
||||
assert "kagentz A2A server DOWN" in proc.stdout
|
||||
|
||||
|
||||
# ── Prose assertions (kept as secondary, do not replace behavioural) ──
|
||||
|
||||
def test_prose_c1_no_credential_needed():
|
||||
"""C1 header should say 'no credential needed'."""
|
||||
prose = (ROOT / "zulip-health.prose.md").read_text()
|
||||
assert "C1: A2A Server Health (no credential needed)" in prose
|
||||
|
||||
|
||||
def test_prose_c2_requires_litellm_key():
|
||||
"""C2 header should say 'requires LITELLM_KEY'."""
|
||||
prose = (ROOT / "zulip-health.prose.md").read_text()
|
||||
assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose
|
||||
|
||||
|
||||
def test_prose_c3_public_access_path():
|
||||
"""C3 section should exist and document 502/000 as incidents."""
|
||||
prose = (ROOT / "zulip-health.prose.md").read_text()
|
||||
assert "C3: Public Access Path" in prose
|
||||
assert "https://kagentz.sysloggh.net/" in prose
|
||||
assert "502" in prose
|
||||
|
||||
+13
-6
@@ -528,12 +528,19 @@ If any bot processes >50 bot-originated messages in 15min → warning.
|
||||
|
||||
### Step 6: Compile and Report
|
||||
|
||||
1. Compile all platform checks and severity
|
||||
2. Determine `overall_severity` from worst per-agent severity
|
||||
3. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart
|
||||
4. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp
|
||||
5. If any agent critical or >2 degraded: send relay message to user
|
||||
6. Update `last_check` timestamp in `### Maintains` snapshot
|
||||
1. Run `scripts/zulip-monitor.sh` and take its final `Result:` line as the
|
||||
authoritative run verdict. The verdict line is either
|
||||
`Result: ✅ 0 issues (all healthy)` or
|
||||
`Result: 🔴 INCIDENT — N issue(s) found`.
|
||||
2. Quote that `Result:` line verbatim in the status report. When it says
|
||||
`INCIDENT`, the run MUST be reported as an incident — never summarised as
|
||||
OK/healthy and never annotated as "expected".
|
||||
3. Compile all platform checks and severity
|
||||
4. Determine `overall_severity` from worst per-agent severity
|
||||
5. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart
|
||||
6. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp
|
||||
7. If any agent critical or >2 degraded: send relay message to user
|
||||
8. Update `last_check` timestamp in `### Maintains` snapshot
|
||||
|
||||
### Restart Debounce
|
||||
|
||||
|
||||
Reference in New Issue
Block a user