chore: Add local monitoring tooling scripts
This commit is contained in:
Executable
+71
@@ -0,0 +1,71 @@
|
||||
#!/bin/bash
|
||||
# daily-health-digest.sh — Daily Fleet Health Digest
|
||||
# Runs all major health checks and summarizes results
|
||||
#
|
||||
# Legs:
|
||||
# - infrastructure-monitoring (13 legs)
|
||||
# - gpu-monitor (6 legs)
|
||||
# - litellm-health (11 checks)
|
||||
# - agent-health-check (4 agents + 3 GPUs + 4 LiteLLM keys)
|
||||
# - proxmox-monitor (4 exporters)
|
||||
# - pm2-self-heal (PM2 services)
|
||||
# - zulip-health (Zulip mesh)
|
||||
#
|
||||
# Exit 0 if all pass, 1 if any fail.
|
||||
#
|
||||
# Run: bash scripts/daily-health-digest.sh
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
||||
SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
FAILED=()
|
||||
|
||||
echo "=== Daily Health Digest — $TIMESTAMP ==="
|
||||
echo "Executed from: $(pwd -P)"
|
||||
echo ""
|
||||
|
||||
# Run each health check and capture exit code
|
||||
run_check() {
|
||||
local name="$1"
|
||||
local script="$2"
|
||||
local runner="${3:-bash}"
|
||||
local output
|
||||
|
||||
echo "── Running $name ──"
|
||||
output=$(cd "$SCRIPTS_DIR/.." && $runner "$script" 2>&1)
|
||||
local exit_code=$?
|
||||
|
||||
if [ $exit_code -eq 0 ]; then
|
||||
# Extract just the summary lines (lines with ✅ or "All")
|
||||
echo "$output" | grep -E "✅|All legs OK|All checks passed|All healthy" | tail -3
|
||||
echo ""
|
||||
else
|
||||
echo " ❌ $name failed (exit $exit_code)"
|
||||
echo "$output" | grep -E "🔴|❌" | tail -5
|
||||
echo ""
|
||||
FAILED+=("$name")
|
||||
fi
|
||||
}
|
||||
|
||||
# Run all checks
|
||||
run_check "Infrastructure Monitoring" "scripts/infra-monitoring.sh"
|
||||
run_check "GPU Monitor" "scripts/gpu-monitor.sh"
|
||||
run_check "LiteLLM Health" "scripts/litellm-health-check.py" "python3"
|
||||
run_check "Agent Health Check" "scripts/agent-health-check.py" "python3"
|
||||
run_check "Proxmox Monitor" "scripts/proxmox-monitor.sh"
|
||||
run_check "PM2 Self-Heal" "scripts/pm2-self-heal.sh"
|
||||
run_check "Zulip Health" "scripts/zulip-monitor.sh"
|
||||
|
||||
# Summary
|
||||
echo "─────────────────────────────────────────────"
|
||||
if [ ${#FAILED[@]} -eq 0 ]; then
|
||||
echo "✅ ALL HEALTH CHECKS PASSED"
|
||||
exit 0
|
||||
else
|
||||
echo "❌ ${#FAILED[@]} HEALTH CHECK(S) FAILED:"
|
||||
for f in "${FAILED[@]}"; do
|
||||
echo " - $f"
|
||||
done
|
||||
exit 1
|
||||
fi
|
||||
Executable
+118
@@ -0,0 +1,118 @@
|
||||
#!/bin/bash
|
||||
# gpu-monitor.sh — GPU Fleet Health Monitor
|
||||
# Implements gpu-monitor.prose.md (check-health section)
|
||||
#
|
||||
# Legs:
|
||||
# - GPU Monitor health (localhost:9100/health)
|
||||
# - GPU host health (192.168.68.8:8080, 192.168.68.110:8080, 192.168.68.15:8080)
|
||||
# - IMPORTANT: Always probe :8080, NEVER bare :80
|
||||
# - Router unified health (192.168.68.116/health/unified, expects 301)
|
||||
# - LiteLLM health (192.168.68.116/litellm/health/liveliness, expects 200)
|
||||
#
|
||||
# Exit 0 if all probes pass, 1 if any fails.
|
||||
#
|
||||
# Run: bash scripts/gpu-monitor.sh
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
CT116_HOST="192.168.68.116"
|
||||
CT8_HOST="192.168.68.8"
|
||||
CT110_HOST="192.168.68.110"
|
||||
CT15_HOST="192.168.68.15"
|
||||
FAILED=()
|
||||
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
||||
|
||||
echo "=== GPU Monitor — $TIMESTAMP ==="
|
||||
echo "Executed from: $(pwd -P)"
|
||||
echo ""
|
||||
|
||||
# 1. GPU Monitor health
|
||||
echo " Probing GPU Monitor health..."
|
||||
GPU_MONITOR_RESP=$(curl -s --connect-timeout 10 http://localhost:9100/health 2>/dev/null)
|
||||
if [ $? -ne 0 ]; then
|
||||
echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (curl timeout/error)"
|
||||
FAILED+=("gpu-monitor")
|
||||
elif echo "$GPU_MONITOR_RESP" | grep -q '"status".*"healthy"'; then
|
||||
CACHE_AGE=$(echo "$GPU_MONITOR_RESP" | grep -o '"cache_age_seconds":[[:space:]]*[0-9]*' | cut -d: -f2 | tr -d ' ')
|
||||
echo " ✅ GPU Monitor: healthy (cache_age=$CACHE_AGE)"
|
||||
else
|
||||
echo " 🔴 GPU Monitor: probe-failed: localhost:9100 (expected healthy, got: $GPU_MONITOR_RESP)"
|
||||
FAILED+=("gpu-monitor")
|
||||
fi
|
||||
|
||||
# 2. GPU host health — DIRECT on :8080 (NEVER bare port 80)
|
||||
echo " Probing GPU host health..."
|
||||
|
||||
# CT 8 (.8)
|
||||
RTX3090_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT8_HOST}:8080/health 2>/dev/null)
|
||||
RTX3090_CODE=$(printf '%s' "$RTX3090_CODE" | tr -d '[:space:]')
|
||||
[ -n "$RTX3090_CODE" ] || RTX3090_CODE="000"
|
||||
|
||||
if [ "$RTX3090_CODE" = "200" ]; then
|
||||
echo " ✅ GPU .8 (RTX 3090): healthy"
|
||||
else
|
||||
echo " 🔴 GPU .8 (RTX 3090): probe-failed: ${CT8_HOST}:8080/health (expected 200, got ${RTX3090_CODE})"
|
||||
FAILED+=("gpu-8")
|
||||
fi
|
||||
|
||||
# CT 110 (.110)
|
||||
RTX5070_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT110_HOST}:8080/health 2>/dev/null)
|
||||
RTX5070_CODE=$(printf '%s' "$RTX5070_CODE" | tr -d '[:space:]')
|
||||
[ -n "$RTX5070_CODE" ] || RTX5070_CODE="000"
|
||||
|
||||
if [ "$RTX5070_CODE" = "200" ]; then
|
||||
echo " ✅ GPU .110 (RTX 5070): healthy"
|
||||
else
|
||||
echo " 🔴 GPU .110 (RTX 5070): probe-failed: ${CT110_HOST}:8080/health (expected 200, got ${RTX5070_CODE})"
|
||||
FAILED+=("gpu-110")
|
||||
fi
|
||||
|
||||
# CT 15 (.15) - Strix Halo
|
||||
STRIX_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT15_HOST}:8080/health 2>/dev/null)
|
||||
STRIX_CODE=$(printf '%s' "$STRIX_CODE" | tr -d '[:space:]')
|
||||
[ -n "$STRIX_CODE" ] || STRIX_CODE="000"
|
||||
|
||||
if [ "$STRIX_CODE" = "200" ]; then
|
||||
echo " ✅ GPU .15 (Strix Halo): healthy"
|
||||
else
|
||||
echo " 🔴 GPU .15 (Strix Halo): probe-failed: ${CT15_HOST}:8080/health (expected 200, got ${STRIX_CODE})"
|
||||
FAILED+=("gpu-15")
|
||||
fi
|
||||
|
||||
# 3. Router unified health (source of truth; 301 → /gpu/gpu-data is HEALTHY)
|
||||
echo " Probing Router unified health..."
|
||||
ROUTER_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/health/unified 2>/dev/null)
|
||||
ROUTER_CODE=$(printf '%s' "$ROUTER_CODE" | tr -d '[:space:]')
|
||||
[ -n "$ROUTER_CODE" ] || ROUTER_CODE="000"
|
||||
|
||||
if [ "$ROUTER_CODE" = "301" ]; then
|
||||
echo " ✅ Router Unified: alive (301 → /gpu/gpu-data)"
|
||||
else
|
||||
echo " 🔴 Router Unified: probe-failed: ${CT116_HOST}/health/unified (expected 301, got ${ROUTER_CODE})"
|
||||
FAILED+=("router")
|
||||
fi
|
||||
|
||||
# 4. LiteLLM health
|
||||
echo " Probing LiteLLM health..."
|
||||
LITELLM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}/litellm/health/liveliness 2>/dev/null)
|
||||
LITELLM_CODE=$(printf '%s' "$LITELLM_CODE" | tr -d '[:space:]')
|
||||
[ -n "$LITELLM_CODE" ] || LITELLM_CODE="000"
|
||||
|
||||
if [ "$LITELLM_CODE" = "200" ]; then
|
||||
echo " ✅ LiteLLM: alive"
|
||||
else
|
||||
echo " 🔴 LiteLLM: probe-failed: ${CT116_HOST}/litellm/health/liveliness (expected 200, got ${LITELLM_CODE})"
|
||||
FAILED+=("litellm")
|
||||
fi
|
||||
|
||||
# ── Summary ─────────────────────────────────────────────────────────────────
|
||||
echo ""
|
||||
if [ ${#FAILED[@]} -eq 0 ]; then
|
||||
echo " ✅ All legs OK"
|
||||
exit 0
|
||||
else
|
||||
for f in "${FAILED[@]}"; do
|
||||
echo " 🔴 FAILED: $f"
|
||||
done
|
||||
exit 1
|
||||
fi
|
||||
Reference in New Issue
Block a user