From 7a5ddb46a9692346e7398e94e3f6246943e8547c Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 22:27:19 +0000 Subject: [PATCH] chore: Add local monitoring tooling scripts --- scripts/daily-health-digest.sh | 71 ++++++++++++++++++++ scripts/gpu-monitor.sh | 118 +++++++++++++++++++++++++++++++++ 2 files changed, 189 insertions(+) create mode 100755 scripts/daily-health-digest.sh create mode 100755 scripts/gpu-monitor.sh diff --git a/scripts/daily-health-digest.sh b/scripts/daily-health-digest.sh new file mode 100755 index 0000000..7663a35 --- /dev/null +++ b/scripts/daily-health-digest.sh @@ -0,0 +1,71 @@ +#!/bin/bash +# daily-health-digest.sh — Daily Fleet Health Digest +# Runs all major health checks and summarizes results +# +# Legs: +# - infrastructure-monitoring (13 legs) +# - gpu-monitor (6 legs) +# - litellm-health (11 checks) +# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys) +# - proxmox-monitor (4 exporters) +# - pm2-self-heal (PM2 services) +# - zulip-health (Zulip mesh) +# +# Exit 0 if all pass, 1 if any fail. +# +# Run: bash scripts/daily-health-digest.sh + +set -uo pipefail + +TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') +SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)" +FAILED=() + +echo "=== Daily Health Digest — $TIMESTAMP ===" +echo "Executed from: $(pwd -P)" +echo "" + +# Run each health check and capture exit code +run_check() { + local name="$1" + local script="$2" + local runner="${3:-bash}" + local output + + echo "── Running $name ──" + output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1) + local exit_code=$? + + if [ $exit_code -eq 0 ]; then + # Extract just the summary lines (lines with ✅ or "All") + echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3 + echo "" + else + echo " ❌ $name failed (exit $exit_code)" + echo "$output" | grep -E "🔴|❌" | tail -5 + echo "" + FAILED+=("$name") + fi +} + +# Run all checks +run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh" +run_check "GPU Monitor" "scripts/gpu-monitor.sh" +run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3" +run_check "Agent Health Check" "scripts/agent-health-check.py" "python3" +run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh" +run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh" +run_check "Zulip Health" "scripts/zulip-monitor.sh" + +# Summary +echo "─────────────────────────────────────────────" +if [ ${#FAILED[@]} -eq 0 ]; then + echo "✅ ALL HEALTH CHECKS PASSED" + exit 0 +else + echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:" + for f in "${FAILED[@]}"; do + echo " - $f" + done + exit 1 +fi \ No newline at end of file diff --git a/scripts/gpu-monitor.sh b/scripts/gpu-monitor.sh new file mode 100755 index 0000000..b7976d9 --- /dev/null +++ b/scripts/gpu-monitor.sh @@ -0,0 +1,118 @@ +#!/bin/bash +# gpu-monitor.sh — GPU Fleet Health Monitor +# Implements gpu-monitor.prose.md (check-health section) +# +# Legs: +# - GPU Monitor health (localhost:9100/health) +# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080) +# - IMPORTANT: Always probe :8080, NEVER bare :80 +# - Router unified health (192.168.68.116/health/unified, expects 301) +# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200) +# +# Exit 0 if all probes pass, 1 if any fails. +# +# Run: bash scripts/gpu-monitor.sh + +set -uo pipefail + +CT116_HOST="192.168.68.116" +CT8_HOST="192.168.68.8" +CT110_HOST="192.168.68.110" +CT15_HOST="192.168.68.15" +FAILED=() +TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') + +echo "=== GPU Monitor — $TIMESTAMP ===" +echo "Executed from: $(pwd -P)" +echo "" + +# 1. GPU Monitor health +echo " Probing GPU Monitor health..." +GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null) +if [ $? -ne 0 ]; then + echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)" + FAILED+=("gpu-monitor") +elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then + CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ') + echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)" +else + echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)" + FAILED+=("gpu-monitor") +fi + +# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80) +echo " Probing GPU host health..." + +# CT 8 (.8) +RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null) +RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]') +[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000" + +if [ "$RTX3090_CODE" = "200" ]; then + echo " ✅ GPU .8 (RTX 3090): healthy" +else + echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})" + FAILED+=("gpu-8") +fi + +# CT 110 (.110) +RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null) +RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]') +[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000" + +if [ "$RTX5070_CODE" = "200" ]; then + echo " ✅ GPU .110 (RTX 5070): healthy" +else + echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})" + FAILED+=("gpu-110") +fi + +# CT 15 (.15) - Strix Halo +STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null) +STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]') +[ -n "$STRIX_CODE" ] || STRIX_CODE="000" + +if [ "$STRIX_CODE" = "200" ]; then + echo " ✅ GPU .15 (Strix Halo): healthy" +else + echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})" + FAILED+=("gpu-15") +fi + +# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY) +echo " Probing Router unified health..." +ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null) +ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]') +[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000" + +if [ "$ROUTER_CODE" = "301" ]; then + echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)" +else + echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})" + FAILED+=("router") +fi + +# 4. LiteLLM health +echo " Probing LiteLLM health..." +LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null) +LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]') +[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000" + +if [ "$LITELLM_CODE" = "200" ]; then + echo " ✅ LiteLLM: alive" +else + echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})" + FAILED+=("litellm") +fi + +# ── Summary ───────────────────────────────────────────────────────────────── +echo "" +if [ ${#FAILED[@]} -eq 0 ]; then + echo " ✅ All legs OK" + exit 0 +else + for f in "${FAILED[@]}"; do + echo " 🔴 FAILED: $f" + done + exit 1 +fi \ No newline at end of file