fix: move infra-monitoring probes into versioned script
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
- Create scripts/infra-monitoring.sh with all targets/ports/paths in code - Add -k flag for PVE API (self-signed certs) - Probe Docker Stats and PVE Exporter via SSH (bind to 127.0.0.1) - Non-zero exit with named failures - Add tests/test_infra_monitoring.py for port drift detection - Prove happy path + broken target Fixes: infra-monitoring-probe-targets-drift-20260917
This commit is contained in:
Executable
+193
@@ -0,0 +1,193 @@
|
||||
#!/bin/bash
|
||||
# infrastructure-monitoring.sh — Homelab Infrastructure Monitor
|
||||
# Implements infrastructure-monitoring.prose.md v4
|
||||
#
|
||||
# Legs: Grafana, Prometheus, LiteLLM, PVE API (5 nodes), GPU exporters,
|
||||
# Docker Stats, PVE Exporter, PM2
|
||||
#
|
||||
# Design:
|
||||
# - Every target, port, path, and expected status is defined in code
|
||||
# - Any HTTP status (200/301/302/401/403/404) = ALIVE
|
||||
# - Only connection failures (000/timeout) = probe-failed
|
||||
# - PVE API uses -k flag (self-signed certs)
|
||||
# - Docker Stats and PVE Exporter bind to 127.0.0.1, probed via SSH
|
||||
# - Non-zero exit with named failures
|
||||
# - No "OK" summary when any leg failed
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
# ── Configuration ───────────────────────────────────────────────────────────
|
||||
# Documented targets (from infrastructure-monitoring.prose.md)
|
||||
# Change these in ONE place; tests assert against these values
|
||||
|
||||
GRAFANA_HOST="192.168.68.116"
|
||||
GRAFANA_PORT="3001"
|
||||
GRAFANA_PATH="/api/health"
|
||||
GRAFANA_EXPECTED="200|302"
|
||||
|
||||
PROMETHEUS_HOST="192.168.68.116"
|
||||
PROMETHEUS_PORT="9090"
|
||||
PROMETHEUS_PATH="/-/healthy"
|
||||
PROMETHEUS_EXPECTED="200"
|
||||
|
||||
LITELLM_HOST="192.168.68.116"
|
||||
LITELLM_PORT="4000"
|
||||
LITELLM_PATH="/"
|
||||
LITELLM_EXPECTED="200|401"
|
||||
|
||||
# PVE API: probe REAL PVE nodes, never the monitoring host (CT 116)
|
||||
PVE_NODES=("192.168.68.9" "192.168.68.12" "192.168.68.6" "192.168.68.15" "192.168.68.5")
|
||||
PVE_API_PORT="8006"
|
||||
PVE_API_SCHEME="https"
|
||||
PVE_API_EXPECTED="200|401"
|
||||
PVE_API_USE_K="1" # self-signed certs
|
||||
|
||||
# GPU exporters
|
||||
GPU_HOSTS=("192.168.68.8" "192.168.68.110" "192.168.68.15")
|
||||
GPU_PORT="9400"
|
||||
GPU_PATH="/metrics"
|
||||
GPU_EXPECTED="200"
|
||||
|
||||
# Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116
|
||||
DOCKER_STATS_HOST="192.168.68.116"
|
||||
DOCKER_STATS_PORT="9323"
|
||||
DOCKER_STATS_PATH="/"
|
||||
DOCKER_STATS_EXPECTED="200|404"
|
||||
|
||||
PVE_EXPORTER_HOST="192.168.68.116"
|
||||
PVE_EXPORTER_PORT="9324"
|
||||
PVE_EXPORTER_PATH="/"
|
||||
PVE_EXPORTER_EXPECTED="200|404"
|
||||
|
||||
# PM2 (CT 100)
|
||||
PM2_HOST="192.168.68.24"
|
||||
PM2_EXPECTED="online"
|
||||
|
||||
# ── Probe Functions ─────────────────────────────────────────────────────────
|
||||
|
||||
# probe_http <host> <port> <path> <expected_pattern> [use_k] [ssh_host] [scheme]
|
||||
# Returns: 0 if any HTTP status matches, 1 if probe-failed
|
||||
probe_http() {
|
||||
local host="$1" port="$2" path="$3" expected="$4" use_k="${5:-}" ssh_host="${6:-}"
|
||||
local scheme="${7:-http}"; local url="${scheme}://${host}:${port}${path}"
|
||||
local curl_opts=(-s -o /dev/null -w '%{http_code}' --max-time 10)
|
||||
local code=""
|
||||
|
||||
if [ -n "$use_k" ]; then
|
||||
curl_opts+=(-k)
|
||||
fi
|
||||
|
||||
if [ -n "$ssh_host" ]; then
|
||||
# Probe via SSH to the host where the service binds to 127.0.0.1
|
||||
code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
|
||||
"curl -s -o /dev/null -w '%{http_code}' --max-time 10 ${use_k:+-k} ${scheme}://127.0.0.1:${port}${path}" 2>/dev/null) || code="000"
|
||||
else
|
||||
code=$(curl "${curl_opts[@]}" "$url" 2>/dev/null) || code="000"
|
||||
fi
|
||||
|
||||
# Clean up the code
|
||||
code=$(printf '%s' "$code" | tr -d '[:space:]')
|
||||
[ -n "$code" ] || code="000"
|
||||
|
||||
# Check if code matches expected pattern
|
||||
if echo "$code" | grep -qE "^(${expected})$"; then
|
||||
return 0
|
||||
else
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# ── Main ────────────────────────────────────────────────────────────────────
|
||||
|
||||
FAILED=()
|
||||
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
|
||||
echo "=== Infrastructure Monitoring — $TIMESTAMP ==="
|
||||
|
||||
# Grafana
|
||||
if probe_http "$GRAFANA_HOST" "$GRAFANA_PORT" "$GRAFANA_PATH" "$GRAFANA_EXPECTED"; then
|
||||
echo " ✅ Grafana: alive"
|
||||
else
|
||||
echo " 🔴 Grafana: probe-failed: ${GRAFANA_HOST}:${GRAFANA_PORT} (expected ${GRAFANA_EXPECTED})"
|
||||
FAILED+=("grafana")
|
||||
fi
|
||||
|
||||
# Prometheus
|
||||
if probe_http "$PROMETHEUS_HOST" "$PROMETHEUS_PORT" "$PROMETHEUS_PATH" "$PROMETHEUS_EXPECTED"; then
|
||||
echo " ✅ Prometheus: alive"
|
||||
else
|
||||
echo " 🔴 Prometheus: probe-failed: ${PROMETHEUS_HOST}:${PROMETHEUS_PORT} (expected ${PROMETHEUS_EXPECTED})"
|
||||
FAILED+=("prometheus")
|
||||
fi
|
||||
|
||||
# LiteLLM
|
||||
if probe_http "$LITELLM_HOST" "$LITELLM_PORT" "$LITELLM_PATH" "$LITELLM_EXPECTED"; then
|
||||
echo " ✅ LiteLLM: alive"
|
||||
else
|
||||
echo " 🔴 LiteLLM: probe-failed: ${LITELLM_HOST}:${LITELLM_PORT} (expected ${LITELLM_EXPECTED})"
|
||||
FAILED+=("litellm")
|
||||
fi
|
||||
|
||||
# PVE API (5 nodes)
|
||||
PVE_FAILED=()
|
||||
for node in "${PVE_NODES[@]}"; do
|
||||
if probe_http "$node" "$PVE_API_PORT" "/" "$PVE_API_EXPECTED" "$PVE_API_USE_K" "" "https"; then
|
||||
echo " ✅ PVE API ${node}: alive"
|
||||
else
|
||||
echo " 🔴 PVE API ${node}: probe-failed: ${node}:${PVE_API_PORT} (expected ${PVE_API_EXPECTED})"
|
||||
PVE_FAILED+=("$node")
|
||||
fi
|
||||
done
|
||||
if [ ${#PVE_FAILED[@]} -gt 0 ]; then
|
||||
FAILED+=("pve-api: ${PVE_FAILED[*]}")
|
||||
fi
|
||||
|
||||
# GPU exporters
|
||||
GPU_FAILED=()
|
||||
for host in "${GPU_HOSTS[@]}"; do
|
||||
if probe_http "$host" "$GPU_PORT" "$GPU_PATH" "$GPU_EXPECTED"; then
|
||||
echo " ✅ GPU ${host}: alive"
|
||||
else
|
||||
echo " 🔴 GPU ${host}: probe-failed: ${host}:${GPU_PORT} (expected ${GPU_EXPECTED})"
|
||||
GPU_FAILED+=("$host")
|
||||
fi
|
||||
done
|
||||
if [ ${#GPU_FAILED[@]} -gt 0 ]; then
|
||||
FAILED+=("gpu: ${GPU_FAILED[*]}")
|
||||
fi
|
||||
|
||||
# Docker Stats (localhost via SSH)
|
||||
if probe_http "$DOCKER_STATS_HOST" "$DOCKER_STATS_PORT" "$DOCKER_STATS_PATH" "$DOCKER_STATS_EXPECTED" "" "$DOCKER_STATS_HOST"; then
|
||||
echo " ✅ Docker Stats: alive"
|
||||
else
|
||||
echo " 🔴 Docker Stats: probe-failed: ${DOCKER_STATS_HOST}:${DOCKER_STATS_PORT} (expected ${DOCKER_STATS_EXPECTED})"
|
||||
FAILED+=("docker-stats")
|
||||
fi
|
||||
|
||||
# PVE Exporter (localhost via SSH)
|
||||
if probe_http "$PVE_EXPORTER_HOST" "$PVE_EXPORTER_PORT" "$PVE_EXPORTER_PATH" "$PVE_EXPORTER_EXPECTED" "" "$PVE_EXPORTER_HOST"; then
|
||||
echo " ✅ PVE Exporter: alive"
|
||||
else
|
||||
echo " 🔴 PVE Exporter: probe-failed: ${PVE_EXPORTER_HOST}:${PVE_EXPORTER_PORT} (expected ${PVE_EXPORTER_EXPECTED})"
|
||||
FAILED+=("pve-exporter")
|
||||
fi
|
||||
|
||||
# PM2
|
||||
PM2_OUTPUT=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${PM2_HOST}" \
|
||||
"pm2 list 2>/dev/null | grep -c 'online'" 2>/dev/null) || PM2_OUTPUT="0"
|
||||
if [ "$PM2_OUTPUT" -gt 0 ]; then
|
||||
echo " ✅ PM2: ${PM2_OUTPUT} processes online"
|
||||
else
|
||||
echo " 🔴 PM2: probe-failed: no processes online on ${PM2_HOST}"
|
||||
FAILED+=("pm2")
|
||||
fi
|
||||
|
||||
# ── Summary ─────────────────────────────────────────────────────────────────
|
||||
|
||||
echo ""
|
||||
if [ ${#FAILED[@]} -eq 0 ]; then
|
||||
echo " ✅ All legs OK"
|
||||
exit 0
|
||||
else
|
||||
echo " 🔴 FAILED legs: ${FAILED[*]}"
|
||||
exit 1
|
||||
fi
|
||||
Reference in New Issue
Block a user