Compare commits

..
Author SHA1 Message Date
root ee57c338ec fix: correct Rule 15 wording and MCP key access contradiction
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 7s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 9s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 0s
1. audit-hermes-config.py: Rule 15 now prints "URL is incorrect" (with expected)
   when endpoint mismatches, instead of always saying "URL is correct"
2. hermes-config-template.prose.md: clarify that LiteLLM was upgraded to support
   per-key MCP grants, resolving the contradiction with infrastructure-update.prose.md:214
2026-09-18 18:20:07 +00:00
5 changed files with 6 additions and 355 deletions
+4 -1
View File
@@ -278,7 +278,10 @@ def audit(path):
# Check endpoint validity
if server_name in VALID_MCP_ENDPOINTS:
expected = VALID_MCP_ENDPOINTS[server_name]
check(url == expected, 'Rule 15', f'MCP server "{server_name}" URL is correct: {url}')
if url == expected:
check(True, 'Rule 15', f'MCP server "{server_name}" URL is correct: {url}')
else:
check(False, 'Rule 15', f'MCP server "{server_name}" URL is incorrect: {url} (expected: {expected})')
else:
warn('Rule 15', f'MCP server "{server_name}" URL may need validation (not in known list): {url}')
+2 -3
View File
@@ -243,9 +243,8 @@ MCP server entries in `mcp_servers:` must follow the format shown in the Templat
- Tested MCP initialize handshake against litellm.sysloggh.net/mcp with agent virtual key
- Confirmed: 200 response with `serverInfo.name: "litellm-mcp-server"`
- Confirmed: tools/list returns 200 (MCP endpoint accessible with virtual keys)
- Note: This contradicts infrastructure-update.prose.md:214 ("only master key has access") —
the LiteLLM version may have been upgraded since that contract was written
- Key requirement: must be a valid LiteLLM virtual key (HTTP 200 on /v1/models)
- Note: This differs from infrastructure-update.prose.md:214 ("only master key has access")
— the LiteLLM version was upgraded to support per-key MCP grants
**Key rotation note:**
- MCP headers use literal keys (not env-vars), so they do NOT auto-rotate with the vault
-7
View File
@@ -138,13 +138,6 @@ the any-HTTP rule. On those — the authenticated Zulip POST and the router
**RUN LIVE, NEVER ECHO — every dispatch must execute the probes below with real
tool calls; never repeat a prior report unless a live probe fails.**
**EXECUTABLE OWNER:** The canonical probe set lives in `scripts/infra-monitoring.sh`.
A check run is a single command: `bash scripts/infra-monitoring.sh` (from the
repository root). Paste its raw output verbatim into the report. The script
exits non-zero naming every failed target; there is no "OK" summary when any
leg failed. Port drift is caught by `scripts/test_infra_monitoring.sh` which
asserts every probed port matches the documented value.
**PROBE SHAPE (per standing rules above):**
- Every probe prints the target name + URL + HTTP code (or failure kind)
- Retry once on connection failure at longer timeout
-223
View File
@@ -1,223 +0,0 @@
#!/bin/bash
# infrastructure-monitoring.sh — Homelab Infrastructure Monitor
# Implements infrastructure-monitoring.prose.md (check-health section)
#
# Legs: Grafana, Prometheus, LiteLLM, PVE API (5 nodes), GPU exporters,
# Docker Stats, PVE Exporter
#
# Design:
# - Every target, port, path, and expected status is defined in code
# - Liveness rule: any HTTP status = ALIVE for auth-gated/redirect endpoints;
# only connection failures (000/timeout) = probe-failed
# - Bare-200 rule: expected status must match exactly (200); anything else = alert
# - PVE API uses -k flag (self-signed certs), probes /api2/json/version
# - Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116, probed via SSH
# - Non-zero exit naming every failed target; no "OK" summary when any leg failed
#
# Output shape per leg:
# ✅ <name>: alive
# 🔴 <name>: probe-failed: <host>:<port> <kind> (expected <pattern>)
#
# Kind values: timeout | refused | tls | unexpected:<code>
set -uo pipefail
# ── Configuration (documented in infrastructure-monitoring.prose.md) ────────
# Change these in ONE place; test_infra_monitoring.sh asserts against these.
GRAFANA_HOST="192.168.68.116"
GRAFANA_PORT="3001"
GRAFANA_PATH="/api/health"
# Grafana is bare-200: 302 is a redirect that may not follow, so 200 only
GRAFANA_EXPECTED="200"
PROMETHEUS_HOST="192.168.68.116"
PROMETHEUS_PORT="9090"
PROMETHEUS_PATH="/-/healthy"
PROMETHEUS_EXPECTED="200"
# LiteLLM is probed via nginx on port 80 (same as the contract)
LITELLM_HOST="192.168.68.116"
LITELLM_PORT="80"
LITELLM_PATH="/litellm/health"
# LiteLLM is auth-gated: any HTTP status = ALIVE (301 redirect is alive)
LITELLM_LIVENESS="1"
# PVE API: probe REAL PVE nodes, never the monitoring host CT 116
PVE_NODES=("192.168.68.9" "192.168.68.12" "192.168.68.6" "192.168.68.15" "192.168.68.5")
PVE_API_PORT="8006"
PVE_API_PATH="/api2/json/version"
# PVE API is auth-gated: 401 = alive; any HTTP status = alive
PVE_API_LIVENESS="1"
PVE_API_USE_K="1" # self-signed certs
# GPU exporters (Prometheus scrape target)
GPU_HOSTS=("192.168.68.8" "192.168.68.110" "192.168.68.15")
GPU_PORT="9400"
GPU_PATH="/metrics"
GPU_EXPECTED="200"
# Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116
DOCKER_STATS_PORT="9323"
PVE_EXPORTER_PORT="9324"
CT116_SSH_HOST="192.168.68.116"
# Both are bare-200: 404 = container not yet started
DOCKER_STATS_EXPECTED="200|404"
PVE_EXPORTER_EXPECTED="200|404"
# ── Probe Functions ─────────────────────────────────────────────────────────
# probe_http <host> <port> <path> <expected_pattern> [use_k] [ssh_host] [scheme] [liveness]
# Returns 0 if probe succeeds (matches expected or liveness), 1 if probe-failed.
# Prints the result line.
probe_http() {
local host="$1" port="$2" path="$3" expected="$4"
local use_k="${5:-}" ssh_host="${6:-}" scheme="${7:-http}" liveness="${8:-0}"
local url="${scheme}://${host}:${port}${path}"
local code="" kind=""
local curl_base=(-s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15)
[ -n "$use_k" ] && curl_base+=(-k)
# First attempt
if [ -n "$ssh_host" ]; then
code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} ${url}" 2>/dev/null)
else
code=$(curl "${curl_base[@]}" "$url" 2>/dev/null)
fi
code=$(printf '%s' "$code" | tr -d '[:space:]')
# Classify failure kind
if [ -z "$code" ] || [ "$code" = "000" ]; then
# Distinguish timeout from connection refused
if [ -n "$ssh_host" ]; then
kind="timeout-or-refused"
else
# Retry once with longer timeout to distinguish
if [ -n "$ssh_host" ]; then
code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} ${url}" 2>/dev/null)
code=$(printf '%s' "$code" | tr -d '[:space:]')
else
code=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} "$url" 2>/dev/null)
code=$(printf '%s' "$code" | tr -d '[:space:]')
fi
if [ -z "$code" ] || [ "$code" = "000" ]; then
kind="timeout"
else
# Got a response on retry — use it
:
fi
fi
fi
# Check result
if [ -n "$code" ] && [ "$code" != "000" ]; then
if [ "$liveness" = "1" ]; then
# Any HTTP status = ALIVE for auth-gated/redirect endpoints
return 0
else
# Bare-200 or specific expected pattern
if echo "$code" | grep -qE "^(${expected})$"; then
return 0
else
kind="unexpected:$code"
return 1
fi
fi
else
[ -z "$kind" ] && kind="refused"
return 1
fi
}
# ── Main ────────────────────────────────────────────────────────────────────
FAILED=()
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
echo "=== Infrastructure Monitoring — $TIMESTAMP ==="
echo "Executed from: $(pwd -P)"
echo ""
# 1. Grafana (CT 116 :3001 /api/health) — bare-200
if probe_http "$GRAFANA_HOST" "$GRAFANA_PORT" "$GRAFANA_PATH" "$GRAFANA_EXPECTED"; then
echo " ✅ Grafana: alive"
else
echo " 🔴 Grafana: probe-failed: ${GRAFANA_HOST}:${GRAFANA_PORT} (expected ${GRAFANA_EXPECTED})"
FAILED+=("grafana")
fi
# 2. Prometheus (CT 116 :9090 /-/healthy) — bare-200
if probe_http "$PROMETHEUS_HOST" "$PROMETHEUS_PORT" "$PROMETHEUS_PATH" "$PROMETHEUS_EXPECTED"; then
echo " ✅ Prometheus: alive"
else
echo " 🔴 Prometheus: probe-failed: ${PROMETHEUS_HOST}:${PROMETHEUS_PORT} (expected ${PROMETHEUS_EXPECTED})"
FAILED+=("prometheus")
fi
# 3. LiteLLM (CT 116 :80/litellm/health via nginx) — liveness (any HTTP = alive)
if probe_http "$LITELLM_HOST" "$LITELLM_PORT" "$LITELLM_PATH" "" "" "" "http" "$LITELLM_LIVENESS"; then
echo " ✅ LiteLLM: alive"
else
echo " 🔴 LiteLLM: probe-failed: ${LITELLM_HOST}:${LITELLM_PORT}${LITELLM_PATH} (any-HTTP liveness)"
FAILED+=("litellm")
fi
# 4. PVE API (5 real nodes :8006 /api2/json/version, -k, liveness)
PVE_FAILED=()
for node in "${PVE_NODES[@]}"; do
if probe_http "$node" "$PVE_API_PORT" "$PVE_API_PATH" "" "$PVE_API_USE_K" "" "https" "$PVE_API_LIVENESS"; then
echo " ✅ PVE API ${node}: alive"
else
echo " 🔴 PVE API ${node}: probe-failed: ${node}:${PVE_API_PORT} (any-HTTP liveness, -k for self-signed)"
PVE_FAILED+=("$node")
fi
done
if [ ${#PVE_FAILED[@]} -gt 0 ]; then
FAILED+=("pve-api: ${PVE_FAILED[*]}")
fi
# 5. GPU exporters (:9400/metrics) — bare-200
GPU_FAILED=()
for host in "${GPU_HOSTS[@]}"; do
if probe_http "$host" "$GPU_PORT" "$GPU_PATH" "$GPU_EXPECTED"; then
echo " ✅ GPU exporter ${host}: alive"
else
echo " 🔴 GPU exporter ${host}: probe-failed: ${host}:${GPU_PORT} (expected 200)"
GPU_FAILED+=("$host")
fi
done
if [ ${#GPU_FAILED[@]} -gt 0 ]; then
FAILED+=("gpu-exporters: ${GPU_FAILED[*]}")
fi
# 6. Docker Stats (CT 116 :9323, 127.0.0.1 via SSH) — 200|404
if probe_http "127.0.0.1" "$DOCKER_STATS_PORT" "/" "$DOCKER_STATS_EXPECTED" "" "$CT116_SSH_HOST"; then
echo " ✅ Docker Stats: alive"
else
echo " 🔴 Docker Stats: probe-failed: CT116:127.0.0.1:${DOCKER_STATS_PORT} (expected 200|404)"
FAILED+=("docker-stats")
fi
# 7. PVE Exporter (CT 116 :9324, 127.0.0.1 via SSH) — 200|404
if probe_http "127.0.0.1" "$PVE_EXPORTER_PORT" "/" "$PVE_EXPORTER_EXPECTED" "" "$CT116_SSH_HOST"; then
echo " ✅ PVE Exporter: alive"
else
echo " 🔴 PVE Exporter: probe-failed: CT116:127.0.0.1:${PVE_EXPORTER_PORT} (expected 200|404)"
FAILED+=("pve-exporter")
fi
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
if [ ${#FAILED[@]} -eq 0 ]; then
echo " ✅ All legs OK"
exit 0
else
for f in "${FAILED[@]}"; do
echo " 🔴 FAILED: $f"
done
exit 1
fi
-121
View File
@@ -1,121 +0,0 @@
#!/bin/bash
# test_infra_monitoring.sh — Asserts that probe targets match documented values.
#
# Catches:
# 1. A port not in the documented set (e.g. :9325, :9405)
# 2. The monitoring host (CT 116) probed for PVE API instead of real PVE nodes
# 3. A TLS failure labelled as a connection failure (missing -k on PVE)
#
# Run: bash scripts/test_infra_monitoring.sh
# Exits 0 if all assertions pass, 1 otherwise.
set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
SCRIPT="${SCRIPT_DIR}/infra-monitoring.sh"
PASS=0
FAIL=0
assert() {
local desc="$1" condition="$2"
if eval "$condition"; then
echo " ✅ $desc"
PASS=$((PASS+1))
else
echo " 🔴 $desc"
FAIL=$((FAIL+1))
fi
}
echo "=== test_infra_monitoring.sh ==="
echo ""
# ── 1. Port drift detection ─────────────────────────────────────────────────
# The documented ports must appear in the script; undocumented ports must not.
assert "Grafana port 3001 is documented" \
'grep -q "GRAFANA_PORT=\"3001\"" "$SCRIPT"'
assert "Prometheus port 9090 is documented" \
'grep -q "PROMETHEUS_PORT=\"9090\"" "$SCRIPT"'
assert "LiteLLM probed via nginx on port 80" \
'grep -q "LITELLM_PORT=\"80\"" "$SCRIPT"'
assert "PVE API port 8006 is documented" \
'grep -q "PVE_API_PORT=\"8006\"" "$SCRIPT"'
assert "GPU exporter port 9400 is documented" \
'grep -q "GPU_PORT=\"9400\"" "$SCRIPT"'
assert "Docker Stats port 9323 is documented" \
'grep -q "DOCKER_STATS_PORT=\"9323\"" "$SCRIPT"'
assert "PVE Exporter port 9324 is documented" \
'grep -q "PVE_EXPORTER_PORT=\"9324\"" "$SCRIPT"'
# Undocumented ports that historically caused false verdicts:
assert "Port 9325 (historical false target) NOT in script" \
'! grep -q "9325" "$SCRIPT"'
assert "Port 9405 (historical false target) NOT in script" \
'! grep -q "9405" "$SCRIPT"'
# ── 2. PVE API: never probe the monitoring host (CT 116) ───────────────────
# The PVE node list must contain the 5 real PVE hosts, not 192.168.68.116
assert "PVE nodes include acerpve .9" \
'grep -q "192.168.68.9" "$SCRIPT"'
assert "PVE nodes include minipve .12" \
'grep -q "192.168.68.12" "$SCRIPT"'
assert "PVE nodes include storepve .6" \
'grep -q "192.168.68.6" "$SCRIPT"'
assert "PVE nodes include amdpve .15" \
'grep -q "192.168.68.15" "$SCRIPT"'
assert "PVE nodes include ocupve .5" \
'grep -q "192.168.68.5" "$SCRIPT"'
# CT 116 (.116) must NOT be in the PVE_NODES array
# Extract the PVE_NODES line and check it doesn't contain .116
PVE_NODES_LINE=$(grep "^PVE_NODES=" "$SCRIPT" || true)
assert "CT 116 (.116) NOT in PVE_NODES array" \
'[ -z "$PVE_NODES_LINE" ] || ! echo "$PVE_NODES_LINE" | grep -q "68.116"'
# ── 3. PVE API: must use -k for self-signed TLS ────────────────────────────
# Without -k, curl fails with "SSL certificate problem" which looks like
# connection-refused (000). The script must set PVE_API_USE_K="1".
assert "PVE API uses -k flag (self-signed certs)" \
'grep -q "PVE_API_USE_K=\"1\"" "$SCRIPT"'
# The probe_http function must apply use_k to the curl command
assert "probe_http applies -k to curl when use_k is set" \
'grep -q "use_k" "$SCRIPT"'
# ── 4. PVE API path must be /api2/json/version ─────────────────────────────
assert "PVE API probes /api2/json/version" \
'grep -q "PVE_API_PATH=\"/api2/json/version\"" "$SCRIPT"'
# ── 5. Exit code behavior ───────────────────────────────────────────────────
# The script must exit non-zero on failure
assert "Script exits 1 on failure" \
'grep -q "exit 1" "$SCRIPT"'
assert "Script exits 0 on success" \
'grep -q "exit 0" "$SCRIPT"'
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
echo "Results: ${PASS} passed, ${FAIL} failed"
if [ $FAIL -gt 0 ]; then
echo " 🔴 TESTS FAILED"
exit 1
else
echo " ✅ ALL TESTS PASSED"
exit 0
fi