#!/bin/bash # infrastructure-monitoring.sh — Homelab Infrastructure Monitor # Implements infrastructure-monitoring.prose.md (check-health section) # # Legs: Grafana, Prometheus, LiteLLM, PVE API (5 nodes), GPU exporters, # Docker Stats, PVE Exporter # # Design: # - Every target, port, path, and expected status is defined in code # - Liveness rule: any HTTP status = ALIVE for auth-gated/redirect endpoints; # only connection failures (000/timeout) = probe-failed # - Bare-200 rule: expected status must match exactly (200); anything else = alert # - PVE API uses -k flag (self-signed certs), probes /api2/json/version # - Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116, probed via SSH # with one retry at longer timeout (25s connect, 30s max) to distinguish # transient timeout from host-down # - Non-zero exit naming every failed target; no "OK" summary when any leg failed # # Output shape per leg: # ✅ : alive # 🔴 : probe-failed: : (expected ) () # # Failure kinds: timeout | refused | tls | unexpected: (printed in the failure line) set -uo pipefail # ── Configuration (documented in infrastructure-monitoring.prose.md) ──────── # Change these in ONE place; test_infra_monitoring.sh asserts against these. GRAFANA_HOST="192.168.68.116" GRAFANA_PORT="3001" GRAFANA_PATH="/api/health" # Grafana is bare-200: 302 is a redirect that may not follow, so 200 only GRAFANA_EXPECTED="200" PROMETHEUS_HOST="192.168.68.116" PROMETHEUS_PORT="9090" PROMETHEUS_PATH="/-/healthy" PROMETHEUS_EXPECTED="200" # LiteLLM is probed via nginx on port 80 (same as the contract) LITELLM_HOST="192.168.68.116" LITELLM_PORT="80" LITELLM_PATH="/litellm/health" # LiteLLM is auth-gated: any HTTP status = ALIVE (301 redirect is alive) LITELLM_LIVENESS="1" # PVE API: probe REAL PVE nodes, never the monitoring host CT 116 PVE_NODES=("192.168.68.9" "192.168.68.12" "192.168.68.6" "192.168.68.15" "192.168.68.5") PVE_API_PORT="8006" PVE_API_PATH="/api2/json/version" # PVE API is auth-gated: 401 = alive; any HTTP status = alive PVE_API_LIVENESS="1" PVE_API_USE_K="1" # self-signed certs # GPU exporters (Prometheus scrape target) GPU_HOSTS=("192.168.68.8" "192.168.68.110" "192.168.68.15") GPU_PORT="9400" GPU_PATH="/metrics" GPU_EXPECTED="200" # Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116 DOCKER_STATS_PORT="9324" # harness-docker-stats (docker_container_* metrics) PVE_EXPORTER_PORT="9221" # harness-pve-exporter (5 pve_* metrics) CT116_SSH_HOST="192.168.68.116" # Both are bare-200: 404 = container not yet started DOCKER_STATS_EXPECTED="200|404" PVE_EXPORTER_EXPECTED="200|404" # ── Probe Functions ───────────────────────────────────────────────────────── # probe_http [use_k] [ssh_host] [scheme] [liveness] # Returns 0 if probe succeeds (matches expected or liveness), 1 if probe-failed. # Prints the result line. # # FIX C1: The kind value is computed and printed in the failure line. # FIX C2: SSH retry logic is in the first attempt branch (not unreachable). LAST_KIND="" probe_http() { local host="$1" port="$2" path="$3" expected="$4" local use_k="${5:-}" ssh_host="${6:-}" scheme="${7:-http}" liveness="${8:-0}" local url="${scheme}://${host}:${port}${path}" local code="" kind="" LAST_KIND="" # Single invocation that captures both output and status if [ -n "$ssh_host" ]; then out=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \ "curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} ${url}" 2>/dev/null) rc=$? else out=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} "$url" 2>/dev/null) rc=$? fi code=$(printf '%s' "$out" | tr -d '[:space:]') # Classify failure kind and retry if needed if [ -z "$code" ] || [ "$code" = "000" ]; then # Distinguish timeout from TLS error from refused case "$rc" in 35|51|58|59|60|77|83) kind="tls" ;; *) kind="timeout" ;; esac # Retry once at longer timeout (25s connect, 30s max) if [ -n "$ssh_host" ]; then code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \ "curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} ${url}" 2>/dev/null) rc=$? code=$(printf '%s' "$code" | tr -d '[:space:]') else code=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} "$url" 2>/dev/null) rc=$? code=$(printf '%s' "$code" | tr -d '[:space:]') # On retry, classify: still 000 = keep existing kind (or timeout if empty), unexpected status = refused if [ -z "$code" ] || [ "$code" = "000" ]; then [ -z "$kind" ] && kind="timeout" elif ! echo "$code" | grep -qE "^(${expected})$"; then kind="refused" fi fi fi # Check result if [ -n "$code" ] && [ "$code" != "000" ]; then if [ "$liveness" = "1" ]; then # Any HTTP status = ALIVE for auth-gated/redirect endpoints return 0 else # Bare-200 or specific expected pattern if echo "$code" | grep -qE "^(${expected})$"; then return 0 else kind="unexpected:$code" LAST_KIND="$kind" return 1 fi fi else [ -z "$kind" ] && kind="refused" LAST_KIND="$kind" return 1 fi } # ── Main ──────────────────────────────────────────────────────────────────── FAILED=() FAILED_KIND=() TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC') echo "=== Infrastructure Monitoring — $TIMESTAMP ===" echo "Executed from: $(pwd -P)" echo "" # 1. Grafana (CT 116 :3001 /api/health) — bare-200 if probe_http "$GRAFANA_HOST" "$GRAFANA_PORT" "$GRAFANA_PATH" "$GRAFANA_EXPECTED"; then echo " ✅ Grafana: alive" else echo " 🔴 Grafana: probe-failed: ${GRAFANA_HOST}:${GRAFANA_PORT} (expected ${GRAFANA_EXPECTED}) (<${LAST_KIND}>)" FAILED+=("grafana") fi # 2. Prometheus (CT 116 :9090 /-/healthy) — bare-200 if probe_http "$PROMETHEUS_HOST" "$PROMETHEUS_PORT" "$PROMETHEUS_PATH" "$PROMETHEUS_EXPECTED"; then echo " ✅ Prometheus: alive" else echo " 🔴 Prometheus: probe-failed: ${PROMETHEUS_HOST}:${PROMETHEUS_PORT} (expected ${PROMETHEUS_EXPECTED}) (<${LAST_KIND}>)" FAILED+=("prometheus") fi # 3. LiteLLM (CT 116 :80/litellm/health via nginx) — liveness (any HTTP = alive) if probe_http "$LITELLM_HOST" "$LITELLM_PORT" "$LITELLM_PATH" "" "" "" "http" "$LITELLM_LIVENESS"; then echo " ✅ LiteLLM: alive" else echo " 🔴 LiteLLM: probe-failed: ${LITELLM_HOST}:${LITELLM_PORT}${LITELLM_PATH} (any-HTTP liveness) (<${LAST_KIND}>)" FAILED+=("litellm") fi # 4. PVE API (5 real nodes :8006 /api2/json/version, -k, liveness) PVE_FAILED=() for node in "${PVE_NODES[@]}"; do if probe_http "$node" "$PVE_API_PORT" "$PVE_API_PATH" "" "$PVE_API_USE_K" "" "https" "$PVE_API_LIVENESS"; then echo " ✅ PVE API ${node}: alive" else echo " 🔴 PVE API ${node}: probe-failed: ${node}:${PVE_API_PORT} (any-HTTP liveness, -k for self-signed) (<${LAST_KIND}>)" PVE_FAILED+=("$node") fi done if [ ${#PVE_FAILED[@]} -gt 0 ]; then FAILED+=("pve-api: ${PVE_FAILED[*]}") fi # 5. GPU exporters (:9400/metrics) — bare-200 GPU_FAILED=() for host in "${GPU_HOSTS[@]}"; do if probe_http "$host" "$GPU_PORT" "$GPU_PATH" "$GPU_EXPECTED"; then echo " ✅ GPU exporter ${host}: alive" else echo " 🔴 GPU exporter ${host}: probe-failed: ${host}:${GPU_PORT} (expected 200) (<${LAST_KIND}>)" GPU_FAILED+=("$host") fi done if [ ${#GPU_FAILED[@]} -gt 0 ]; then FAILED+=("gpu-exporters: ${GPU_FAILED[*]}") fi # 6. Docker Stats (CT 116 :9324, 127.0.0.1 via SSH) — 200|404 if probe_http "127.0.0.1" "$DOCKER_STATS_PORT" "/" "$DOCKER_STATS_EXPECTED" "" "$CT116_SSH_HOST"; then echo " ✅ Docker Stats: alive" else echo " 🔴 Docker Stats: probe-failed: CT116:127.0.0.1:${DOCKER_STATS_PORT} (expected 200|404) (<${LAST_KIND}>)" FAILED+=("docker-stats") fi # 7. PVE Exporter (CT 116 :9221, 127.0.0.1 via SSH) — 200|404 if probe_http "127.0.0.1" "$PVE_EXPORTER_PORT" "/" "$PVE_EXPORTER_EXPECTED" "" "$CT116_SSH_HOST"; then echo " ✅ PVE Exporter: alive" else echo " 🔴 PVE Exporter: probe-failed: CT116:127.0.0.1:${PVE_EXPORTER_PORT} (expected 200|404) (<${LAST_KIND}>)" FAILED+=("pve-exporter") fi # ── Summary ───────────────────────────────────────────────────────────────── echo "" if [ ${#FAILED[@]} -eq 0 ]; then echo " ✅ All legs OK" exit 0 else for f in "${FAILED[@]}"; do echo " 🔴 FAILED: $f" done exit 1 fi