no-mistakes(review): Fix review findings: test regression, contract verdict, scope trim

This commit is contained in:
root
2026-09-19 22:31:25 +00:00
parent 7a5ddb46a9
commit 63990b84f7
5 changed files with 31 additions and 232 deletions
-71
View File
@@ -1,71 +0,0 @@
#!/bin/bash
# daily-health-digest.sh — Daily Fleet Health Digest
# Runs all major health checks and summarizes results
#
# Legs:
# - infrastructure-monitoring (13 legs)
# - gpu-monitor (6 legs)
# - litellm-health (11 checks)
# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys)
# - proxmox-monitor (4 exporters)
# - pm2-self-heal (PM2 services)
# - zulip-health (Zulip mesh)
#
# Exit 0 if all pass, 1 if any fail.
#
# Run: bash scripts/daily-health-digest.sh
set -uo pipefail
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)"
FAILED=()
echo "=== Daily Health Digest — $TIMESTAMP ==="
echo "Executed from: $(pwd -P)"
echo ""
# Run each health check and capture exit code
run_check() {
local name="$1"
local script="$2"
local runner="${3:-bash}"
local output
echo "── Running $name ──"
output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1)
local exit_code=$?
if [ $exit_code -eq 0 ]; then
# Extract just the summary lines (lines with ✅ or "All")
echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3
echo ""
else
echo " ❌ $name failed (exit $exit_code)"
echo "$output" | grep -E "🔴|❌" | tail -5
echo ""
FAILED+=("$name")
fi
}
# Run all checks
run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh"
run_check "GPU Monitor" "scripts/gpu-monitor.sh"
run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3"
run_check "Agent Health Check" "scripts/agent-health-check.py" "python3"
run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh"
run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh"
run_check "Zulip Health" "scripts/zulip-monitor.sh"
# Summary
echo "─────────────────────────────────────────────"
if [ ${#FAILED[@]} -eq 0 ]; then
echo "✅ ALL HEALTH CHECKS PASSED"
exit 0
else
echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:"
for f in "${FAILED[@]}"; do
echo " - $f"
done
exit 1
fi
-118
View File
@@ -1,118 +0,0 @@
#!/bin/bash
# gpu-monitor.sh — GPU Fleet Health Monitor
# Implements gpu-monitor.prose.md (check-health section)
#
# Legs:
# - GPU Monitor health (localhost:9100/health)
# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080)
# - IMPORTANT: Always probe :8080, NEVER bare :80
# - Router unified health (192.168.68.116/health/unified, expects 301)
# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200)
#
# Exit 0 if all probes pass, 1 if any fails.
#
# Run: bash scripts/gpu-monitor.sh
set -uo pipefail
CT116_HOST="192.168.68.116"
CT8_HOST="192.168.68.8"
CT110_HOST="192.168.68.110"
CT15_HOST="192.168.68.15"
FAILED=()
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
echo "=== GPU Monitor — $TIMESTAMP ==="
echo "Executed from: $(pwd -P)"
echo ""
# 1. GPU Monitor health
echo " Probing GPU Monitor health..."
GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null)
if [ $? -ne 0 ]; then
echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)"
FAILED+=("gpu-monitor")
elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then
CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ')
echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)"
else
echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)"
FAILED+=("gpu-monitor")
fi
# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80)
echo " Probing GPU host health..."
# CT 8 (.8)
RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null)
RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]')
[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000"
if [ "$RTX3090_CODE" = "200" ]; then
echo " ✅ GPU .8 (RTX 3090): healthy"
else
echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})"
FAILED+=("gpu-8")
fi
# CT 110 (.110)
RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null)
RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]')
[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000"
if [ "$RTX5070_CODE" = "200" ]; then
echo " ✅ GPU .110 (RTX 5070): healthy"
else
echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})"
FAILED+=("gpu-110")
fi
# CT 15 (.15) - Strix Halo
STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null)
STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]')
[ -n "$STRIX_CODE" ] || STRIX_CODE="000"
if [ "$STRIX_CODE" = "200" ]; then
echo " ✅ GPU .15 (Strix Halo): healthy"
else
echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})"
FAILED+=("gpu-15")
fi
# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY)
echo " Probing Router unified health..."
ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null)
ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]')
[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000"
if [ "$ROUTER_CODE" = "301" ]; then
echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)"
else
echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})"
FAILED+=("router")
fi
# 4. LiteLLM health
echo " Probing LiteLLM health..."
LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null)
LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]')
[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000"
if [ "$LITELLM_CODE" = "200" ]; then
echo " ✅ LiteLLM: alive"
else
echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})"
FAILED+=("litellm")
fi
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
if [ ${#FAILED[@]} -eq 0 ]; then
echo " ✅ All legs OK"
exit 0
else
for f in "${FAILED[@]}"; do
echo " 🔴 FAILED: $f"
done
exit 1
fi
+18 -13
View File
@@ -98,8 +98,8 @@ exit 0
"""
CURL_STUB = r"""#!/usr/bin/env bash
# Stub curl: serve the Abiba health fixture and the Zulip server 200, and
# record every call (including notify) payloads.
# Stub curl: serve the Abiba health fixture, the Zulip server 200, and the
# kagentz C3 public URL, and record every call (including notify) payloads.
printf '%s\n' "$*" >> "$RECORD_DIR/curl.calls"
case "$*" in
*:9200/health*)
@@ -109,6 +109,8 @@ case "$*" in
esac ;;
*server_settings*)
printf '%s' "$SERVER_HTTP" ;;
*kagentz.sysloggh.net*)
printf '%s' "$KAGENTZ_PUBLIC_CODE" ;;
esac
exit 0
"""
@@ -121,7 +123,8 @@ def _write_exec(path: pathlib.Path, body: str) -> None:
def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
az_a2a_code="401", az_a2a_exit=0):
az_a2a_code="401", az_a2a_exit=0,
kagentz_public_code="302"):
"""Run the shipped monitor in a sandbox; return (proc, record_dir, log_path).
Only the LOG constant is rewritten (to keep the run inside the worktree).
@@ -154,6 +157,7 @@ def _run_monitor(tmp_path, *, tanko_svc="active", tanko_http="200",
"PI_HTTP": "200",
"PI_BODY": CONNECTED_FIXTURE.read_text(),
"SERVER_HTTP": "200",
"KAGENTZ_PUBLIC_CODE": kagentz_public_code,
})
proc = subprocess.run(["bash", str(script)], cwd=sandbox, env=env,
capture_output=True, text=True)
@@ -169,8 +173,9 @@ def test_healthy_run_is_quiet_and_never_reaches_mumuni(tmp_path):
assert "Server: ✅ HTTP 200" in log
assert "Abiba: ✅ Connected" in log
assert "Tanko: ✅ service=active http=200" in log
assert "kagentz: ✅ A2A alive (HTTP 401)" in log
assert "Result: ✅ All healthy" in log
assert "kagentz C1: ✅ A2A alive (HTTP 401)" in log
assert "kagentz C3: ✅ public URL alive (HTTP 302)" in log
assert "Result: ✅ 0 issues (all healthy)" in log
# A healthy run emits no notify at all — and certainly no Mumuni one.
assert proc.stdout == ""
@@ -207,8 +212,8 @@ def test_failing_run_alerts_on_tanko_but_never_on_mumuni(tmp_path):
# The rest of the monitor still ran alongside the failing Tanko leg.
log = log_path.read_text()
assert "Abiba: ✅ Connected" in log
assert "kagentz: ✅ A2A alive" in log
assert "Result: 🔴 1 issue(s) found" in log
assert "kagentz C1: ✅ A2A alive" in log
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
@@ -216,9 +221,9 @@ def test_unexpected_a2a_status_is_an_issue_not_healthy(tmp_path):
assert proc.returncode == 0, proc.stderr
log = log_path.read_text()
assert "kagentz: 🟡 A2A unexpected http=500 (running, warning)" in log
assert "kagentz: ✅ A2A alive" not in log
assert "Result: 🔴 1 issue(s) found" in log
assert "kagentz C1: 🟡 A2A unexpected http=500 (running, warning)" in log
assert "kagentz C1: ✅ A2A alive" not in log
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
assert "kagentz A2A server answered HTTP 500" in proc.stdout
@@ -230,10 +235,10 @@ def test_a2a_connection_failure_is_down_not_unexpected(tmp_path):
assert proc.returncode == 0, proc.stderr
log = log_path.read_text()
assert "kagentz: ❌ A2A down (HTTP 000)" in log
assert "kagentz: ✅ A2A alive" not in log
assert "kagentz C1: ❌ A2A down (HTTP 000)" in log
assert "kagentz C1: ✅ A2A alive" not in log
assert "unexpected" not in log
assert "Result: 🔴 1 issue(s) found" in log
assert "Result: 🔴 INCIDENT — 1 issue(s) found" in log
assert "kagentz A2A server DOWN (connection failed)" in proc.stdout
-24
View File
@@ -30,8 +30,6 @@ import pathlib
import stat
import subprocess
import pytest
ROOT = pathlib.Path(__file__).resolve().parents[1]
ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh"
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
@@ -209,25 +207,3 @@ def test_c1_000_is_incident(tmp_path):
assert "all healthy" not in log
# The notify fired for the A2A down.
assert "kagentz A2A server DOWN" in proc.stdout
# ── Prose assertions (kept as secondary, do not replace behavioural) ──
def test_prose_c1_no_credential_needed():
"""C1 header should say 'no credential needed'."""
prose = (ROOT / "zulip-health.prose.md").read_text()
assert "C1: A2A Server Health (no credential needed)" in prose
def test_prose_c2_requires_litellm_key():
"""C2 header should say 'requires LITELLM_KEY'."""
prose = (ROOT / "zulip-health.prose.md").read_text()
assert "C2: A2A Response Verification (requires LITELLM_KEY)" in prose
def test_prose_c3_public_access_path():
"""C3 section should exist and document 502/000 as incidents."""
prose = (ROOT / "zulip-health.prose.md").read_text()
assert "C3: Public Access Path" in prose
assert "https://kagentz.sysloggh.net/" in prose
assert "502" in prose
+13 -6
View File
@@ -528,12 +528,19 @@ If any bot processes >50 bot-originated messages in 15min → warning.
### Step 6: Compile and Report
1. Compile all platform checks and severity
2. Determine `overall_severity` from worst per-agent severity
3. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart
4. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp
5. If any agent critical or >2 degraded: send relay message to user
6. Update `last_check` timestamp in `### Maintains` snapshot
1. Run `scripts/zulip-monitor.sh` and take its final `Result:` line as the
authoritative run verdict. The verdict line is either
`Result: ✅ 0 issues (all healthy)` or
`Result: 🔴 INCIDENT — N issue(s) found`.
2. Quote that `Result:` line verbatim in the status report. When it says
`INCIDENT`, the run MUST be reported as an incident — never summarised as
OK/healthy and never annotated as "expected".
3. Compile all platform checks and severity
4. Determine `overall_severity` from worst per-agent severity
5. If restart action needed, check `/tmp/zulip-monitor-debounce` — apply only if >300s since last restart
6. Log full diagnostic to `/root/zulip-health-monitor.log` with timestamp
7. If any agent critical or >2 degraded: send relay message to user
8. Update `last_check` timestamp in `### Maintains` snapshot
### Restart Debounce