fix: move infra-monitoring probes into versioned script
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s

- Create scripts/infra-monitoring.sh with all targets/ports/paths in code
- Add -k flag for PVE API (self-signed certs)
- Probe Docker Stats and PVE Exporter via SSH (bind to 127.0.0.1)
- Non-zero exit with named failures
- Add tests/test_infra_monitoring.py for port drift detection
- Prove happy path + broken target

Fixes: infra-monitoring-probe-targets-drift-20260917
This commit is contained in:
root
2026-09-18 04:40:20 +00:00
parent 8a5cba8515
commit 5ab5de4704
2 changed files with 358 additions and 0 deletions
+193
View File
@@ -0,0 +1,193 @@
#!/bin/bash
# infrastructure-monitoring.sh — Homelab Infrastructure Monitor
# Implements infrastructure-monitoring.prose.md v4
#
# Legs: Grafana, Prometheus, LiteLLM, PVE API (5 nodes), GPU exporters,
# Docker Stats, PVE Exporter, PM2
#
# Design:
# - Every target, port, path, and expected status is defined in code
# - Any HTTP status (200/301/302/401/403/404) = ALIVE
# - Only connection failures (000/timeout) = probe-failed
# - PVE API uses -k flag (self-signed certs)
# - Docker Stats and PVE Exporter bind to 127.0.0.1, probed via SSH
# - Non-zero exit with named failures
# - No "OK" summary when any leg failed
set -uo pipefail
# ── Configuration ───────────────────────────────────────────────────────────
# Documented targets (from infrastructure-monitoring.prose.md)
# Change these in ONE place; tests assert against these values
GRAFANA_HOST="192.168.68.116"
GRAFANA_PORT="3001"
GRAFANA_PATH="/api/health"
GRAFANA_EXPECTED="200|302"
PROMETHEUS_HOST="192.168.68.116"
PROMETHEUS_PORT="9090"
PROMETHEUS_PATH="/-/healthy"
PROMETHEUS_EXPECTED="200"
LITELLM_HOST="192.168.68.116"
LITELLM_PORT="4000"
LITELLM_PATH="/"
LITELLM_EXPECTED="200|401"
# PVE API: probe REAL PVE nodes, never the monitoring host (CT 116)
PVE_NODES=("192.168.68.9" "192.168.68.12" "192.168.68.6" "192.168.68.15" "192.168.68.5")
PVE_API_PORT="8006"
PVE_API_SCHEME="https"
PVE_API_EXPECTED="200|401"
PVE_API_USE_K="1" # self-signed certs
# GPU exporters
GPU_HOSTS=("192.168.68.8" "192.168.68.110" "192.168.68.15")
GPU_PORT="9400"
GPU_PATH="/metrics"
GPU_EXPECTED="200"
# Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116
DOCKER_STATS_HOST="192.168.68.116"
DOCKER_STATS_PORT="9323"
DOCKER_STATS_PATH="/"
DOCKER_STATS_EXPECTED="200|404"
PVE_EXPORTER_HOST="192.168.68.116"
PVE_EXPORTER_PORT="9324"
PVE_EXPORTER_PATH="/"
PVE_EXPORTER_EXPECTED="200|404"
# PM2 (CT 100)
PM2_HOST="192.168.68.24"
PM2_EXPECTED="online"
# ── Probe Functions ─────────────────────────────────────────────────────────
# probe_http <host> <port> <path> <expected_pattern> [use_k] [ssh_host] [scheme]
# Returns: 0 if any HTTP status matches, 1 if probe-failed
probe_http() {
local host="$1" port="$2" path="$3" expected="$4" use_k="${5:-}" ssh_host="${6:-}"
local scheme="${7:-http}"; local url="${scheme}://${host}:${port}${path}"
local curl_opts=(-s -o /dev/null -w '%{http_code}' --max-time 10)
local code=""
if [ -n "$use_k" ]; then
curl_opts+=(-k)
fi
if [ -n "$ssh_host" ]; then
# Probe via SSH to the host where the service binds to 127.0.0.1
code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
"curl -s -o /dev/null -w '%{http_code}' --max-time 10 ${use_k:+-k} ${scheme}://127.0.0.1:${port}${path}" 2>/dev/null) || code="000"
else
code=$(curl "${curl_opts[@]}" "$url" 2>/dev/null) || code="000"
fi
# Clean up the code
code=$(printf '%s' "$code" | tr -d '[:space:]')
[ -n "$code" ] || code="000"
# Check if code matches expected pattern
if echo "$code" | grep -qE "^(${expected})$"; then
return 0
else
return 1
fi
}
# ── Main ────────────────────────────────────────────────────────────────────
FAILED=()
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
echo "=== Infrastructure Monitoring — $TIMESTAMP ==="
# Grafana
if probe_http "$GRAFANA_HOST" "$GRAFANA_PORT" "$GRAFANA_PATH" "$GRAFANA_EXPECTED"; then
echo " ✅ Grafana: alive"
else
echo " 🔴 Grafana: probe-failed: ${GRAFANA_HOST}:${GRAFANA_PORT} (expected ${GRAFANA_EXPECTED})"
FAILED+=("grafana")
fi
# Prometheus
if probe_http "$PROMETHEUS_HOST" "$PROMETHEUS_PORT" "$PROMETHEUS_PATH" "$PROMETHEUS_EXPECTED"; then
echo " ✅ Prometheus: alive"
else
echo " 🔴 Prometheus: probe-failed: ${PROMETHEUS_HOST}:${PROMETHEUS_PORT} (expected ${PROMETHEUS_EXPECTED})"
FAILED+=("prometheus")
fi
# LiteLLM
if probe_http "$LITELLM_HOST" "$LITELLM_PORT" "$LITELLM_PATH" "$LITELLM_EXPECTED"; then
echo " ✅ LiteLLM: alive"
else
echo " 🔴 LiteLLM: probe-failed: ${LITELLM_HOST}:${LITELLM_PORT} (expected ${LITELLM_EXPECTED})"
FAILED+=("litellm")
fi
# PVE API (5 nodes)
PVE_FAILED=()
for node in "${PVE_NODES[@]}"; do
if probe_http "$node" "$PVE_API_PORT" "/" "$PVE_API_EXPECTED" "$PVE_API_USE_K" "" "https"; then
echo " ✅ PVE API ${node}: alive"
else
echo " 🔴 PVE API ${node}: probe-failed: ${node}:${PVE_API_PORT} (expected ${PVE_API_EXPECTED})"
PVE_FAILED+=("$node")
fi
done
if [ ${#PVE_FAILED[@]} -gt 0 ]; then
FAILED+=("pve-api: ${PVE_FAILED[*]}")
fi
# GPU exporters
GPU_FAILED=()
for host in "${GPU_HOSTS[@]}"; do
if probe_http "$host" "$GPU_PORT" "$GPU_PATH" "$GPU_EXPECTED"; then
echo " ✅ GPU ${host}: alive"
else
echo " 🔴 GPU ${host}: probe-failed: ${host}:${GPU_PORT} (expected ${GPU_EXPECTED})"
GPU_FAILED+=("$host")
fi
done
if [ ${#GPU_FAILED[@]} -gt 0 ]; then
FAILED+=("gpu: ${GPU_FAILED[*]}")
fi
# Docker Stats (localhost via SSH)
if probe_http "$DOCKER_STATS_HOST" "$DOCKER_STATS_PORT" "$DOCKER_STATS_PATH" "$DOCKER_STATS_EXPECTED" "" "$DOCKER_STATS_HOST"; then
echo " ✅ Docker Stats: alive"
else
echo " 🔴 Docker Stats: probe-failed: ${DOCKER_STATS_HOST}:${DOCKER_STATS_PORT} (expected ${DOCKER_STATS_EXPECTED})"
FAILED+=("docker-stats")
fi
# PVE Exporter (localhost via SSH)
if probe_http "$PVE_EXPORTER_HOST" "$PVE_EXPORTER_PORT" "$PVE_EXPORTER_PATH" "$PVE_EXPORTER_EXPECTED" "" "$PVE_EXPORTER_HOST"; then
echo " ✅ PVE Exporter: alive"
else
echo " 🔴 PVE Exporter: probe-failed: ${PVE_EXPORTER_HOST}:${PVE_EXPORTER_PORT} (expected ${PVE_EXPORTER_EXPECTED})"
FAILED+=("pve-exporter")
fi
# PM2
PM2_OUTPUT=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${PM2_HOST}" \
"pm2 list 2>/dev/null | grep -c 'online'" 2>/dev/null) || PM2_OUTPUT="0"
if [ "$PM2_OUTPUT" -gt 0 ]; then
echo " ✅ PM2: ${PM2_OUTPUT} processes online"
else
echo " 🔴 PM2: probe-failed: no processes online on ${PM2_HOST}"
FAILED+=("pm2")
fi
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
if [ ${#FAILED[@]} -eq 0 ]; then
echo " ✅ All legs OK"
exit 0
else
echo " 🔴 FAILED legs: ${FAILED[*]}"
exit 1
fi