PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 9s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 1s
The leg comments were wrong: - Line 207: Docker Stats showed :9323 (dockerd port) but should be :9324 - Line 215: PVE Exporter showed :9324 (docker-stats port) but should be :9221 These were the exact pairing this PR exists to correct. Read back the changed lines to verify: scripts/infra-monitoring.sh:207 shows Docker Stats (CT 116 :9324, 127.0.0.1 via SSH) scripts/infra-monitoring.sh:215 shows PVE Exporter (CT 116 :9221, 127.0.0.1 via SSH) Branch: fix/infra-monitoring-probe-ports-20260919
235 lines
9.2 KiB
Bash
Executable File
235 lines
9.2 KiB
Bash
Executable File
#!/bin/bash
|
|
# infrastructure-monitoring.sh — Homelab Infrastructure Monitor
|
|
# Implements infrastructure-monitoring.prose.md (check-health section)
|
|
#
|
|
# Legs: Grafana, Prometheus, LiteLLM, PVE API (5 nodes), GPU exporters,
|
|
# Docker Stats, PVE Exporter
|
|
#
|
|
# Design:
|
|
# - Every target, port, path, and expected status is defined in code
|
|
# - Liveness rule: any HTTP status = ALIVE for auth-gated/redirect endpoints;
|
|
# only connection failures (000/timeout) = probe-failed
|
|
# - Bare-200 rule: expected status must match exactly (200); anything else = alert
|
|
# - PVE API uses -k flag (self-signed certs), probes /api2/json/version
|
|
# - Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116, probed via SSH
|
|
# with one retry at longer timeout (25s connect, 30s max) to distinguish
|
|
# transient timeout from host-down
|
|
# - Non-zero exit naming every failed target; no "OK" summary when any leg failed
|
|
#
|
|
# Output shape per leg:
|
|
# ✅ <name>: alive
|
|
# 🔴 <name>: probe-failed: <host>:<port> (expected <pattern>) (<kind>)
|
|
#
|
|
# Failure kinds: timeout | refused | tls | unexpected:<code> (printed in the failure line)
|
|
|
|
set -uo pipefail
|
|
|
|
# ── Configuration (documented in infrastructure-monitoring.prose.md) ────────
|
|
# Change these in ONE place; test_infra_monitoring.sh asserts against these.
|
|
|
|
GRAFANA_HOST="192.168.68.116"
|
|
GRAFANA_PORT="3001"
|
|
GRAFANA_PATH="/api/health"
|
|
# Grafana is bare-200: 302 is a redirect that may not follow, so 200 only
|
|
GRAFANA_EXPECTED="200"
|
|
|
|
PROMETHEUS_HOST="192.168.68.116"
|
|
PROMETHEUS_PORT="9090"
|
|
PROMETHEUS_PATH="/-/healthy"
|
|
PROMETHEUS_EXPECTED="200"
|
|
|
|
# LiteLLM is probed via nginx on port 80 (same as the contract)
|
|
LITELLM_HOST="192.168.68.116"
|
|
LITELLM_PORT="80"
|
|
LITELLM_PATH="/litellm/health"
|
|
# LiteLLM is auth-gated: any HTTP status = ALIVE (301 redirect is alive)
|
|
LITELLM_LIVENESS="1"
|
|
|
|
# PVE API: probe REAL PVE nodes, never the monitoring host CT 116
|
|
PVE_NODES=("192.168.68.9" "192.168.68.12" "192.168.68.6" "192.168.68.15" "192.168.68.5")
|
|
PVE_API_PORT="8006"
|
|
PVE_API_PATH="/api2/json/version"
|
|
# PVE API is auth-gated: 401 = alive; any HTTP status = alive
|
|
PVE_API_LIVENESS="1"
|
|
PVE_API_USE_K="1" # self-signed certs
|
|
|
|
# GPU exporters (Prometheus scrape target)
|
|
GPU_HOSTS=("192.168.68.8" "192.168.68.110" "192.168.68.15")
|
|
GPU_PORT="9400"
|
|
GPU_PATH="/metrics"
|
|
GPU_EXPECTED="200"
|
|
|
|
# Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116
|
|
DOCKER_STATS_PORT="9324" # harness-docker-stats (docker_container_* metrics)
|
|
PVE_EXPORTER_PORT="9221" # harness-pve-exporter (5 pve_* metrics)
|
|
CT116_SSH_HOST="192.168.68.116"
|
|
# Both are bare-200: 404 = container not yet started
|
|
DOCKER_STATS_EXPECTED="200|404"
|
|
PVE_EXPORTER_EXPECTED="200|404"
|
|
|
|
# ── Probe Functions ─────────────────────────────────────────────────────────
|
|
|
|
# probe_http <host> <port> <path> <expected_pattern> [use_k] [ssh_host] [scheme] [liveness]
|
|
# Returns 0 if probe succeeds (matches expected or liveness), 1 if probe-failed.
|
|
# Prints the result line.
|
|
#
|
|
# FIX C1: The kind value is computed and printed in the failure line.
|
|
# FIX C2: SSH retry logic is in the first attempt branch (not unreachable).
|
|
|
|
LAST_KIND=""
|
|
probe_http() {
|
|
local host="$1" port="$2" path="$3" expected="$4"
|
|
local use_k="${5:-}" ssh_host="${6:-}" scheme="${7:-http}" liveness="${8:-0}"
|
|
local url="${scheme}://${host}:${port}${path}"
|
|
local code="" kind=""
|
|
LAST_KIND=""
|
|
|
|
# Single invocation that captures both output and status
|
|
if [ -n "$ssh_host" ]; then
|
|
out=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
|
|
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} ${url}" 2>/dev/null)
|
|
rc=$?
|
|
else
|
|
out=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} "$url" 2>/dev/null)
|
|
rc=$?
|
|
fi
|
|
code=$(printf '%s' "$out" | tr -d '[:space:]')
|
|
|
|
# Classify failure kind and retry if needed
|
|
if [ -z "$code" ] || [ "$code" = "000" ]; then
|
|
# Distinguish timeout from TLS error from refused
|
|
case "$rc" in
|
|
35|51|58|59|60|77|83) kind="tls" ;;
|
|
*) kind="timeout" ;;
|
|
esac
|
|
# Retry once at longer timeout (25s connect, 30s max)
|
|
if [ -n "$ssh_host" ]; then
|
|
code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
|
|
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} ${url}" 2>/dev/null)
|
|
rc=$?
|
|
code=$(printf '%s' "$code" | tr -d '[:space:]')
|
|
else
|
|
code=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} "$url" 2>/dev/null)
|
|
rc=$?
|
|
code=$(printf '%s' "$code" | tr -d '[:space:]')
|
|
# On retry, classify: still 000 = keep existing kind (or timeout if empty), unexpected status = refused
|
|
if [ -z "$code" ] || [ "$code" = "000" ]; then
|
|
[ -z "$kind" ] && kind="timeout"
|
|
elif ! echo "$code" | grep -qE "^(${expected})$"; then
|
|
kind="refused"
|
|
fi
|
|
fi
|
|
fi
|
|
|
|
# Check result
|
|
if [ -n "$code" ] && [ "$code" != "000" ]; then
|
|
if [ "$liveness" = "1" ]; then
|
|
# Any HTTP status = ALIVE for auth-gated/redirect endpoints
|
|
return 0
|
|
else
|
|
# Bare-200 or specific expected pattern
|
|
if echo "$code" | grep -qE "^(${expected})$"; then
|
|
return 0
|
|
else
|
|
kind="unexpected:$code"
|
|
LAST_KIND="$kind"
|
|
return 1
|
|
fi
|
|
fi
|
|
else
|
|
[ -z "$kind" ] && kind="refused"
|
|
LAST_KIND="$kind"
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
# ── Main ────────────────────────────────────────────────────────────────────
|
|
|
|
FAILED=()
|
|
FAILED_KIND=()
|
|
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
|
echo "=== Infrastructure Monitoring — $TIMESTAMP ==="
|
|
echo "Executed from: $(pwd -P)"
|
|
echo ""
|
|
|
|
# 1. Grafana (CT 116 :3001 /api/health) — bare-200
|
|
if probe_http "$GRAFANA_HOST" "$GRAFANA_PORT" "$GRAFANA_PATH" "$GRAFANA_EXPECTED"; then
|
|
echo " ✅ Grafana: alive"
|
|
else
|
|
echo " 🔴 Grafana: probe-failed: ${GRAFANA_HOST}:${GRAFANA_PORT} (expected ${GRAFANA_EXPECTED}) (<${LAST_KIND}>)"
|
|
FAILED+=("grafana")
|
|
fi
|
|
|
|
# 2. Prometheus (CT 116 :9090 /-/healthy) — bare-200
|
|
if probe_http "$PROMETHEUS_HOST" "$PROMETHEUS_PORT" "$PROMETHEUS_PATH" "$PROMETHEUS_EXPECTED"; then
|
|
echo " ✅ Prometheus: alive"
|
|
else
|
|
echo " 🔴 Prometheus: probe-failed: ${PROMETHEUS_HOST}:${PROMETHEUS_PORT} (expected ${PROMETHEUS_EXPECTED}) (<${LAST_KIND}>)"
|
|
FAILED+=("prometheus")
|
|
fi
|
|
|
|
# 3. LiteLLM (CT 116 :80/litellm/health via nginx) — liveness (any HTTP = alive)
|
|
if probe_http "$LITELLM_HOST" "$LITELLM_PORT" "$LITELLM_PATH" "" "" "" "http" "$LITELLM_LIVENESS"; then
|
|
echo " ✅ LiteLLM: alive"
|
|
else
|
|
echo " 🔴 LiteLLM: probe-failed: ${LITELLM_HOST}:${LITELLM_PORT}${LITELLM_PATH} (any-HTTP liveness) (<${LAST_KIND}>)"
|
|
FAILED+=("litellm")
|
|
fi
|
|
|
|
# 4. PVE API (5 real nodes :8006 /api2/json/version, -k, liveness)
|
|
PVE_FAILED=()
|
|
for node in "${PVE_NODES[@]}"; do
|
|
if probe_http "$node" "$PVE_API_PORT" "$PVE_API_PATH" "" "$PVE_API_USE_K" "" "https" "$PVE_API_LIVENESS"; then
|
|
echo " ✅ PVE API ${node}: alive"
|
|
else
|
|
echo " 🔴 PVE API ${node}: probe-failed: ${node}:${PVE_API_PORT} (any-HTTP liveness, -k for self-signed) (<${LAST_KIND}>)"
|
|
PVE_FAILED+=("$node")
|
|
fi
|
|
done
|
|
if [ ${#PVE_FAILED[@]} -gt 0 ]; then
|
|
FAILED+=("pve-api: ${PVE_FAILED[*]}")
|
|
fi
|
|
|
|
# 5. GPU exporters (:9400/metrics) — bare-200
|
|
GPU_FAILED=()
|
|
for host in "${GPU_HOSTS[@]}"; do
|
|
if probe_http "$host" "$GPU_PORT" "$GPU_PATH" "$GPU_EXPECTED"; then
|
|
echo " ✅ GPU exporter ${host}: alive"
|
|
else
|
|
echo " 🔴 GPU exporter ${host}: probe-failed: ${host}:${GPU_PORT} (expected 200) (<${LAST_KIND}>)"
|
|
GPU_FAILED+=("$host")
|
|
fi
|
|
done
|
|
if [ ${#GPU_FAILED[@]} -gt 0 ]; then
|
|
FAILED+=("gpu-exporters: ${GPU_FAILED[*]}")
|
|
fi
|
|
|
|
# 6. Docker Stats (CT 116 :9324, 127.0.0.1 via SSH) — 200|404
|
|
if probe_http "127.0.0.1" "$DOCKER_STATS_PORT" "/" "$DOCKER_STATS_EXPECTED" "" "$CT116_SSH_HOST"; then
|
|
echo " ✅ Docker Stats: alive"
|
|
else
|
|
echo " 🔴 Docker Stats: probe-failed: CT116:127.0.0.1:${DOCKER_STATS_PORT} (expected 200|404) (<${LAST_KIND}>)"
|
|
FAILED+=("docker-stats")
|
|
fi
|
|
|
|
# 7. PVE Exporter (CT 116 :9221, 127.0.0.1 via SSH) — 200|404
|
|
if probe_http "127.0.0.1" "$PVE_EXPORTER_PORT" "/" "$PVE_EXPORTER_EXPECTED" "" "$CT116_SSH_HOST"; then
|
|
echo " ✅ PVE Exporter: alive"
|
|
else
|
|
echo " 🔴 PVE Exporter: probe-failed: CT116:127.0.0.1:${PVE_EXPORTER_PORT} (expected 200|404) (<${LAST_KIND}>)"
|
|
FAILED+=("pve-exporter")
|
|
fi
|
|
|
|
# ── Summary ─────────────────────────────────────────────────────────────────
|
|
|
|
echo ""
|
|
if [ ${#FAILED[@]} -eq 0 ]; then
|
|
echo " ✅ All legs OK"
|
|
exit 0
|
|
else
|
|
for f in "${FAILED[@]}"; do
|
|
echo " 🔴 FAILED: $f"
|
|
done
|
|
exit 1
|
|
fi
|