Compare commits

..
Author SHA1 Message Date
root 1bfab85a25 docs: add PBS GC schedule to proxmox-monitor contract
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 12s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 0s
- GC runs daily at 20:00 UTC (after backup window closes at 06:40Z)
- Volume is /tank/pbs-backup on ZFS pool 'tank' (12.7T, 10% used)
- NOT /media/easystore2 (media library, 3.7T, 96% HOST-RED)
- Cron: /etc/cron.d/pbs-gc on storepve (192.168.68.6)
- Last run: 2026-09-18 00:08:34 UTC (removed 551.768 GiB)
2026-09-18 04:46:44 +00:00
14 changed files with 65 additions and 974 deletions
+7 -20
View File
@@ -114,22 +114,12 @@ def audit(path):
)
# --- Rule 5: Main Config Base URL ---
# Canonical internal base (hermes-key-enforcement.prose.md:19/38/57/84/91/96)
# and public base (serves /v1 only, per 2026-09-19 probe from CT 116).
# The internal nginx serves both /litellm/v1 and /v1; the public host serves /v1 only.
# FAIL anything else (do not widen to accept any path ending in /v1).
# Internal /v1 is non-canonical but working (authenticated via nginx), so WARN not FAIL.
canonical_internal = "http://192.168.68.116/litellm/v1"
public_host = "https://litellm.sysloggh.net/v1"
non_canonical_internal = "http://192.168.68.116/v1"
allowed_bases = (canonical_internal, public_host)
actual_base = model.get("base_url")
if actual_base in allowed_bases:
check(True, "Rule 5", f"model.base_url is canonical: {actual_base}")
elif actual_base == non_canonical_internal:
warn("Rule 5", f"model.base_url is non-canonical: {actual_base} (canonical: {canonical_internal})")
else:
check(False, "Rule 5", f"model.base_url must be one of {allowed_bases} (got {actual_base!r})")
expected_base = "http://192.168.68.116/v1"
check(
model.get("base_url") == expected_base,
"Rule 5",
f"model.base_url must be {expected_base} (got {model.get('base_url')!r}) — /v1 not /litellm/v1",
)
# --- Rule 6: max_tokens Is Required ---
check(
@@ -288,10 +278,7 @@ def audit(path):
# Check endpoint validity
if server_name in VALID_MCP_ENDPOINTS:
expected = VALID_MCP_ENDPOINTS[server_name]
if url == expected:
check(True, 'Rule 15', f'MCP server "{server_name}" URL is correct: {url}')
else:
check(False, 'Rule 15', f'MCP server "{server_name}" URL is incorrect: {url} (expected: {expected})')
check(url == expected, 'Rule 15', f'MCP server "{server_name}" URL is correct: {url}')
else:
warn('Rule 15', f'MCP server "{server_name}" URL may need validation (not in known list): {url}')
+1 -16
View File
@@ -61,7 +61,7 @@ Host filesystems have their own risk profile and their own bands. A host root ne
|-------|-----------|----------|------------|
| **HOST-WARN** | 85% | Name the volume + % + absolute free space in the scan output | None |
| **HOST-AMBER** | 90% | Name the volume + % + absolute free space; flag for owner attention | Zulip DM to owner (state-change only) |
| **HOST-RED** | 95% | Name the volume + % + absolute free space; **media volumes: "capacity decision — owner to decide"; pbs-datastore/host-root: "immediate owner attention"** | Zulip DM + channel alert (state-change only) |
| **HOST-RED** | 95% | Name the volume + % + absolute free space; flag for immediate owner attention | Zulip DM + channel alert (state-change only) |
**Escalations are STATE-CHANGE driven, not per-run.** A volume alerts ONCE when it enters a higher band (GREEN->WARN, WARN->AMBER, AMBER->RED) and ONCE when it drops back down (a recovery notice). While a volume stays in the same band, it is reported in the scan output only — no DM, no channel alert. This prevents the same 96% easystore2 from re-DMing the owner on every 6-hour scan and drowning a real warning in noise.
@@ -228,21 +228,6 @@ call summary-reporter
plan: plan
```
## GC SCHEDULE (PBS datastore only)
The PBS GC schedule is defined in ONE authoritative place: `/etc/cron.d/pbs-gc` on storepve.
The schedule is `0 20 * * *` (20:00 LOCAL = 00:00 UTC, since host timezone is America/New_York).
This applies **only** to the PBS datastore (`/tank/pbs-backup`), NOT to media volumes.
Media volumes (/media/*) are report-only at all threat levels.
The cron runs `/usr/local/bin/pbs-gc.sh` which executes:
```bash
proxmox-backup-manager garbage-collection start storepve-datastore
```
This is NOT a `prune` operation; it is a GC pass that reclaims unreferenced chunks.
There is no `--keep-daily` flag; retention is governed by jobs.cfg (keep-daily=35).
The GC does not touch media volumes or any other filesystem.
## GC Strategies by Host Type
### Docker Hosts (kagentz 105, syslog-api 116, docker-vm 109, amdpve .15)
+2 -2
View File
@@ -243,8 +243,8 @@ MCP server entries in `mcp_servers:` must follow the format shown in the Templat
- Tested MCP initialize handshake against litellm.sysloggh.net/mcp with agent virtual key
- Confirmed: 200 response with `serverInfo.name: "litellm-mcp-server"`
- Confirmed: tools/list returns 200 (MCP endpoint accessible with virtual keys)
- Note: Per-key MCP grants are now supported (verified 2026-09-18), resolving the earlier
contradiction with infrastructure-update.prose.md (which now reflects the update)
- Note: This contradicts infrastructure-update.prose.md:214 ("only master key has access") —
the LiteLLM version may have been upgraded since that contract was written
- Key requirement: must be a valid LiteLLM virtual key (HTTP 200 on /v1/models)
**Key rotation note:**
+4 -4
View File
@@ -16,7 +16,7 @@ author: Abiba (pi agent)
## Rule (One Sentence)
**All harness/litellm providers MUST use `api_key_env: LITELLM_API_KEY` with canonical internal path `http://192.168.68.116/litellm/v1` (Hermes appends `/v1/responses`) or public path `https://litellm.sysloggh.net/v1` — hardcoded keys AND direct `:4000` access are both forbidden. Internal `/v1` still works but is non-canonical (WARN, not FAIL).**
**All harness/litellm providers MUST use `api_key_env: LITELLM_API_KEY` with authenticated path `http://192.168.68.116/litellm/v1/responses` — hardcoded keys AND unauthenticated `/v1` direct access are both forbidden.**
## Scope
@@ -38,8 +38,8 @@ Syslog is migrating away from **unauthenticated direct access** to the shared in
| `http://192.168.68.116/litellm/v1` | Bearer `sk-*` key (nginx-fronted) | ✅ **CURRENT / CANONICAL** — captain-approved migration target; 600s proxy_read_timeout (verified) |
| `http://192.168.68.116:4000/v1` | Bearer `sk-*` key (direct container) | ❌ **FORBIDDEN** — bypasses nginx; port 4000 direct is not a config path |
All harness/litellm providers MUST use an authenticated nginx-fronted path (`/litellm/v1` canonical, `/v1` non-canonical but working).
The public host `https://litellm.sysloggh.net` serves `/v1` ONLY (404 on `/litellm/v1`).
All harness/litellm providers MUST use an authenticated nginx-fronted path (`/litellm/v1` canonical, `/v1` legacy-valid).
Any `base_url` pointing at `:4000` or a bare IP without nginx is a **migration violation**.
### 🔥 CRITICAL: Double-Path Bug (2026-07-10)
@@ -113,7 +113,7 @@ model:
model:
provider: harness
base_url: http://192.168.68.116/v1 # ← NON-CANONICAL but WORKING (authenticated via nginx, WARN not FAIL)
base_url: http://192.168.68.116/v1 # ← RULE VIOLATION: unauthenticated path
api_key_env: LITELLM_API_KEY
```
+3 -32
View File
@@ -138,19 +138,11 @@ the any-HTTP rule. On those — the authenticated Zulip POST and the router
**RUN LIVE, NEVER ECHO — every dispatch must execute the probes below with real
tool calls; never repeat a prior report unless a live probe fails.**
**EXECUTABLE OWNER:** The canonical probe set lives in `scripts/infra-monitoring.sh`.
A check run is a single command: `bash scripts/infra-monitoring.sh` (from the
repository root). Paste its raw output verbatim into the report. The script
exits non-zero naming every failed target; there is no "OK" summary when any
leg failed. Port drift is caught by `scripts/test_infra_monitoring.sh` which
asserts every probed port matches the documented value.
**PROBE SHAPE (per standing rules above):**
- Every probe prints: `✅ <name>: alive` on success, or `🔴 <name>: probe-failed: <host>:<port> (expected <pattern>) (<kind>)` on failure
- PVE API failures include `(any-HTTP liveness, -k for self-signed)` to distinguish TLS vs connection
- Retry once on connection failure at longer timeout (25s connect, 30s max)
- Every probe prints the target name + URL + HTTP code (or failure kind)
- Retry once on connection failure at longer timeout
- Any HTTP status = ALIVE; only 000/timeout/refused = probe-failed
- Report the actual probe output, not a summary verdict
- Report the actual probe command and its result, not a summary verdict
```bash
# Provenance — run first; paste the absolute path into the report
@@ -277,27 +269,6 @@ code (or failure kind with retry details). Apply the standing probe rules: any
HTTP status = ALIVE; only 000/timeout/refused = probe-failed. A redirect is not
a failure.
### Docker Stats and PVE Exporter Ports
These two exporters bind to 127.0.0.1 on CT 116 (localhost-only) and must be probed via SSH:
| Exporter | Port | Container | Metrics |
|----------|------|-----------|---------|
| **Docker Stats** | **9324** | harness-docker-stats | `docker_container_*` (per-container CPU/mem/network) |
| **PVE Exporter** | **9221** | harness-pve-exporter | `pve_*` (5 cluster-level metrics) |
**IMPORTANT**: Do not confuse with port 9323, which is owned by dockerd and serves the Docker Engine's own metrics (`builder_builds_*`, `containerd_build_info_*`).
```bash
# Docker Stats (harness-docker-stats)
ssh root@192.168.68.116 "curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://127.0.0.1:9324/metrics"
# Expected: 200 or 404 (any HTTP status = ALIVE)
# PVE Exporter (harness-pve-exporter)
ssh root@192.168.68.116 "curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://127.0.0.1:9221/metrics"
# Expected: 200 or 404 (any HTTP status = ALIVE)
```
### Phase 1: GPU Exporters
**NVIDIA (.8 and .110)**:
+7 -7
View File
@@ -208,19 +208,19 @@ mcp_servers:
| Key | MCP Access |
|-----|-----------|
| Master key | ✅ Full — 90 tools (vault-injected) |
| Agent keys (mumuni, tanko, etc.) | ✅ Per-key grants supported (as of 2026-09-18 verification) |
| Agent keys (mumuni, tanko, etc.) | ❌ Per-key grants not supported in v1.99.1 |
### Known Limitations
- ~~Per-key MCP server grants not functional — only master key has access~~ (resolved 2026-09-18: per-key grants now work)
- Per-key MCP server grants not functional — only master key has access
- Responses API (`/v1/responses`) with MCP tools broken on llama.cpp backends
- HTTP 307 redirect on `/mcp` → use `/mcp/` (trailing slash) or `/mcp-rest/` endpoints
- `api_mode: responses` in Hermes appends `/v1/responses` to base_url → **base_url must end at `/v1`, never `/responses`** (double-path bug)
### Migration Path (COMPLETED 2026-09-18)
Per-key MCP grants are now supported:
1. ✅ Agent keys granted MCP access via `allowed_mcp_servers` field
2. ✅ Hermes `mcp_servers.litellm.url` set to `https://litellm.sysloggh.net/mcp`
3. ✅ `headers: {x-litellm-api-key: "Bearer <literal_key>"}` added to MCP config
### Migration Path
When LiteLLM is upgraded to a version supporting per-key MCP grants:
1. Grant agent keys `mcp_servers: ["ra_h_os"]`
2. Update Hermes `mcp_servers.ra-h-os.url` from `http://192.168.68.65:3100/mcp` → `http://192.168.68.116:4000/mcp/`
3. Add `headers: {x-litellm-api-key: "Bearer $LITELLM_API_KEY"}` to MCP config
## Security-Specific Updates
+18 -42
View File
@@ -89,48 +89,6 @@ agent: abiba
| minipve | 192.168.68.12 | PVE |
| amdpve | 192.168.68.15 | PVE + Strix Halo LLM (strix-moe) |
## PBS GC (Proxmox Backup Server)
### Schedule
Cron `0 20 * * *` on the **storepve HOST** (192.168.68.6) = 20:00 America/New_York local = **00:00 UTC**.
**NOTE**: The closed PR #116 said "20:00 UTC" — this is WRONG by four hours. Do not copy it.
### What Actually Runs
Host script `/usr/local/bin/pbs-gc.sh` runs `pct exec 107 -- proxmox-backup-manager garbage-collection start storepve-datastore`.
**IMPORTANT**: The tool `proxmox-backup-manager` exists only inside CT 107 (where the PBS server runs). The storepve host has only `proxmox-backup-client`. This is why the job had never worked before 2026-09-19 00:00 UTC.
### Datastore Location
- **Datastore**: CT 107's `/mnt/pbs-backup` on the storepve ZFS dataset `/tank/pbs-backup` (pool `tank`, ~11T free)
- **NOT** `/media/easystore2` (media library, 3.7T, 96% used — separate volume)
### Liveness Check
The new `proxmox-monitor.sh` leg checks storepve-datastore GC health:
- Reads GC state from CT 107: `pct exec 107 -- proxmox-backup-manager garbage-collection list --output-format json`
- **FAILS** if `last-run-endtime` is older than 48 hours
- Reports age in hours and pending-bytes
**All six verdict shapes** (exactly as emitted by the script):
1. **Healthy** (fresh GC, 0 B pending):
`✅ PBS GC: healthy (last run 1h ago, pending-bytes: 0 B)`
2. **Stale** (GC ran >48h ago):
`🔴 PBS GC: stale (last run 49h ago, pending-bytes: 1048576 B)`
3. **Probe-failed: empty read** (000/timeout/unreadable):
`🔴 PBS GC: probe-failed: storepve:192.168.68.6 (expected JSON, got 000)`
4. **Probe-failed: unparseable** (non-empty but invalid JSON — the "command not found" case):
`🔴 PBS GC: probe-failed: storepve:192.168.68.6 (unparseable JSON)`
5. **Never-run: datastore absent** (valid JSON but storepve-datastore not in list):
`🔴 PBS GC: never-run (storepve-datastore not found in GC list)`
6. **Never-run: no endtime** (valid JSON with datastore present but last-run-endtime is null/0):
`🔴 PBS GC: never-run (storepve-datastore has no last-run-endtime)`
## Operations
### view-dashboards
@@ -184,3 +142,21 @@ each probe. If any probe returns non-200, flag as alert.
- **PVE exporter metric schema**: NOT name-prefixed. `pve_cpu_usage_ratio`, `pve_memory_usage_bytes`, `pve_disk_usage_bytes`, `pve_uptime_seconds` are GUEST-level only (24 series, `id=lxc/100` etc). Node-level host metrics come from node_exporter. Storage pool usage: `pve_storage_info` (info only, no usage bytes — use node_filesystem_* for actual disk usage).
- **grafana piechart plugin removed** from `GF_INSTALL_PLUGINS` (Angular, unsupported in Grafana 13).
- **Single pve-exporter points at amdpve .15** — if amdpve API is down, cluster metrics gap (other node_exporters still report host metrics). Acceptable; amdpve is primary.
## PBS Garbage Collection Schedule (storepve-datastore)
**Schedule:** Daily at 20:00 UTC (4:00 PM EDT)
**Rationale:** The nightly backup window runs 04:00–06:40 UTC (local backups 04:00–06:40Z, S3 sync 04:15Z, S3 trim 05:15Z). Running GC during this window causes avoidable I/O contention on the same datastore and host. 20:00 UTC lands well after the backup window closes and before the next day's backups begin.
**Volume:** `/tank/pbs-backup` on the ZFS pool `tank` (12.7T total, 11.3T free, 10% used). Inside CT 107, this appears as 1.5T total / 413G used / 1.1T avail = 28%.
**Important:** This is NOT the same as `/media/easystore2` (3.7T, 96% used, HOST-RED), which is a media library ("4K MOVIES") on the storepve host root filesystem. The PBS datastore lives on a separate ZFS pool.
**Cron:** `/etc/cron.d/pbs-gc` on storepve (192.168.68.6):
```
0 20 * * * root /usr/local/bin/pbs-gc.sh
```
**Last GC run:** 2026-09-18 00:08:34 UTC (3h 19m 9s, removed 551.768 GiB)
**Schedule owner:** The `gc-schedule` field in the PBS datastore config is the durable owner of GC behavior. The cron is a fallback because `proxmox-backup-manager datastore update --gc-schedule` failed to parse the calendar event (Nom(Eof) error).
-234
View File
@@ -1,234 +0,0 @@
#!/bin/bash
# infrastructure-monitoring.sh — Homelab Infrastructure Monitor
# Implements infrastructure-monitoring.prose.md (check-health section)
#
# Legs: Grafana, Prometheus, LiteLLM, PVE API (5 nodes), GPU exporters,
# Docker Stats, PVE Exporter
#
# Design:
# - Every target, port, path, and expected status is defined in code
# - Liveness rule: any HTTP status = ALIVE for auth-gated/redirect endpoints;
# only connection failures (000/timeout) = probe-failed
# - Bare-200 rule: expected status must match exactly (200); anything else = alert
# - PVE API uses -k flag (self-signed certs), probes /api2/json/version
# - Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116, probed via SSH
# with one retry at longer timeout (25s connect, 30s max) to distinguish
# transient timeout from host-down
# - Non-zero exit naming every failed target; no "OK" summary when any leg failed
#
# Output shape per leg:
# ✅ <name>: alive
# 🔴 <name>: probe-failed: <host>:<port> (expected <pattern>) (<kind>)
#
# Failure kinds: timeout | refused | tls | unexpected:<code> (printed in the failure line)
set -uo pipefail
# ── Configuration (documented in infrastructure-monitoring.prose.md) ────────
# Change these in ONE place; test_infra_monitoring.sh asserts against these.
GRAFANA_HOST="192.168.68.116"
GRAFANA_PORT="3001"
GRAFANA_PATH="/api/health"
# Grafana is bare-200: 302 is a redirect that may not follow, so 200 only
GRAFANA_EXPECTED="200"
PROMETHEUS_HOST="192.168.68.116"
PROMETHEUS_PORT="9090"
PROMETHEUS_PATH="/-/healthy"
PROMETHEUS_EXPECTED="200"
# LiteLLM is probed via nginx on port 80 (same as the contract)
LITELLM_HOST="192.168.68.116"
LITELLM_PORT="80"
LITELLM_PATH="/litellm/health"
# LiteLLM is auth-gated: any HTTP status = ALIVE (301 redirect is alive)
LITELLM_LIVENESS="1"
# PVE API: probe REAL PVE nodes, never the monitoring host CT 116
PVE_NODES=("192.168.68.9" "192.168.68.12" "192.168.68.6" "192.168.68.15" "192.168.68.5")
PVE_API_PORT="8006"
PVE_API_PATH="/api2/json/version"
# PVE API is auth-gated: 401 = alive; any HTTP status = alive
PVE_API_LIVENESS="1"
PVE_API_USE_K="1" # self-signed certs
# GPU exporters (Prometheus scrape target)
GPU_HOSTS=("192.168.68.8" "192.168.68.110" "192.168.68.15")
GPU_PORT="9400"
GPU_PATH="/metrics"
GPU_EXPECTED="200"
# Docker Stats and PVE Exporter bind to 127.0.0.1 on CT 116
DOCKER_STATS_PORT="9324" # harness-docker-stats (docker_container_* metrics)
PVE_EXPORTER_PORT="9221" # harness-pve-exporter (5 pve_* metrics)
CT116_SSH_HOST="192.168.68.116"
# Both are bare-200: 404 = container not yet started
DOCKER_STATS_EXPECTED="200|404"
PVE_EXPORTER_EXPECTED="200|404"
# ── Probe Functions ─────────────────────────────────────────────────────────
# probe_http <host> <port> <path> <expected_pattern> [use_k] [ssh_host] [scheme] [liveness]
# Returns 0 if probe succeeds (matches expected or liveness), 1 if probe-failed.
# Prints the result line.
#
# FIX C1: The kind value is computed and printed in the failure line.
# FIX C2: SSH retry logic is in the first attempt branch (not unreachable).
LAST_KIND=""
probe_http() {
local host="$1" port="$2" path="$3" expected="$4"
local use_k="${5:-}" ssh_host="${6:-}" scheme="${7:-http}" liveness="${8:-0}"
local url="${scheme}://${host}:${port}${path}"
local code="" kind=""
LAST_KIND=""
# Single invocation that captures both output and status
if [ -n "$ssh_host" ]; then
out=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} ${url}" 2>/dev/null)
rc=$?
else
out=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 --max-time 15 ${use_k:+-k} "$url" 2>/dev/null)
rc=$?
fi
code=$(printf '%s' "$out" | tr -d '[:space:]')
# Classify failure kind and retry if needed
if [ -z "$code" ] || [ "$code" = "000" ]; then
# Distinguish timeout from TLS error from refused
case "$rc" in
35|51|58|59|60|77|83) kind="tls" ;;
*) kind="timeout" ;;
esac
# Retry once at longer timeout (25s connect, 30s max)
if [ -n "$ssh_host" ]; then
code=$(ssh -o ConnectTimeout=5 -o BatchMode=yes "root@${ssh_host}" \
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} ${url}" 2>/dev/null)
rc=$?
code=$(printf '%s' "$code" | tr -d '[:space:]')
else
code=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 25 --max-time 30 ${use_k:+-k} "$url" 2>/dev/null)
rc=$?
code=$(printf '%s' "$code" | tr -d '[:space:]')
# On retry, classify: still 000 = keep existing kind (or timeout if empty), unexpected status = refused
if [ -z "$code" ] || [ "$code" = "000" ]; then
[ -z "$kind" ] && kind="timeout"
elif ! echo "$code" | grep -qE "^(${expected})$"; then
kind="refused"
fi
fi
fi
# Check result
if [ -n "$code" ] && [ "$code" != "000" ]; then
if [ "$liveness" = "1" ]; then
# Any HTTP status = ALIVE for auth-gated/redirect endpoints
return 0
else
# Bare-200 or specific expected pattern
if echo "$code" | grep -qE "^(${expected})$"; then
return 0
else
kind="unexpected:$code"
LAST_KIND="$kind"
return 1
fi
fi
else
[ -z "$kind" ] && kind="refused"
LAST_KIND="$kind"
return 1
fi
}
# ── Main ────────────────────────────────────────────────────────────────────
FAILED=()
FAILED_KIND=()
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
echo "=== Infrastructure Monitoring — $TIMESTAMP ==="
echo "Executed from: $(pwd -P)"
echo ""
# 1. Grafana (CT 116 :3001 /api/health) — bare-200
if probe_http "$GRAFANA_HOST" "$GRAFANA_PORT" "$GRAFANA_PATH" "$GRAFANA_EXPECTED"; then
echo " ✅ Grafana: alive"
else
echo " 🔴 Grafana: probe-failed: ${GRAFANA_HOST}:${GRAFANA_PORT} (expected ${GRAFANA_EXPECTED}) (<${LAST_KIND}>)"
FAILED+=("grafana")
fi
# 2. Prometheus (CT 116 :9090 /-/healthy) — bare-200
if probe_http "$PROMETHEUS_HOST" "$PROMETHEUS_PORT" "$PROMETHEUS_PATH" "$PROMETHEUS_EXPECTED"; then
echo " ✅ Prometheus: alive"
else
echo " 🔴 Prometheus: probe-failed: ${PROMETHEUS_HOST}:${PROMETHEUS_PORT} (expected ${PROMETHEUS_EXPECTED}) (<${LAST_KIND}>)"
FAILED+=("prometheus")
fi
# 3. LiteLLM (CT 116 :80/litellm/health via nginx) — liveness (any HTTP = alive)
if probe_http "$LITELLM_HOST" "$LITELLM_PORT" "$LITELLM_PATH" "" "" "" "http" "$LITELLM_LIVENESS"; then
echo " ✅ LiteLLM: alive"
else
echo " 🔴 LiteLLM: probe-failed: ${LITELLM_HOST}:${LITELLM_PORT}${LITELLM_PATH} (any-HTTP liveness) (<${LAST_KIND}>)"
FAILED+=("litellm")
fi
# 4. PVE API (5 real nodes :8006 /api2/json/version, -k, liveness)
PVE_FAILED=()
for node in "${PVE_NODES[@]}"; do
if probe_http "$node" "$PVE_API_PORT" "$PVE_API_PATH" "" "$PVE_API_USE_K" "" "https" "$PVE_API_LIVENESS"; then
echo " ✅ PVE API ${node}: alive"
else
echo " 🔴 PVE API ${node}: probe-failed: ${node}:${PVE_API_PORT} (any-HTTP liveness, -k for self-signed) (<${LAST_KIND}>)"
PVE_FAILED+=("$node")
fi
done
if [ ${#PVE_FAILED[@]} -gt 0 ]; then
FAILED+=("pve-api: ${PVE_FAILED[*]}")
fi
# 5. GPU exporters (:9400/metrics) — bare-200
GPU_FAILED=()
for host in "${GPU_HOSTS[@]}"; do
if probe_http "$host" "$GPU_PORT" "$GPU_PATH" "$GPU_EXPECTED"; then
echo " ✅ GPU exporter ${host}: alive"
else
echo " 🔴 GPU exporter ${host}: probe-failed: ${host}:${GPU_PORT} (expected 200) (<${LAST_KIND}>)"
GPU_FAILED+=("$host")
fi
done
if [ ${#GPU_FAILED[@]} -gt 0 ]; then
FAILED+=("gpu-exporters: ${GPU_FAILED[*]}")
fi
# 6. Docker Stats (CT 116 :9324, 127.0.0.1 via SSH) — 200|404
if probe_http "127.0.0.1" "$DOCKER_STATS_PORT" "/" "$DOCKER_STATS_EXPECTED" "" "$CT116_SSH_HOST"; then
echo " ✅ Docker Stats: alive"
else
echo " 🔴 Docker Stats: probe-failed: CT116:127.0.0.1:${DOCKER_STATS_PORT} (expected 200|404) (<${LAST_KIND}>)"
FAILED+=("docker-stats")
fi
# 7. PVE Exporter (CT 116 :9221, 127.0.0.1 via SSH) — 200|404
if probe_http "127.0.0.1" "$PVE_EXPORTER_PORT" "/" "$PVE_EXPORTER_EXPECTED" "" "$CT116_SSH_HOST"; then
echo " ✅ PVE Exporter: alive"
else
echo " 🔴 PVE Exporter: probe-failed: CT116:127.0.0.1:${PVE_EXPORTER_PORT} (expected 200|404) (<${LAST_KIND}>)"
FAILED+=("pve-exporter")
fi
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
if [ ${#FAILED[@]} -eq 0 ]; then
echo " ✅ All legs OK"
exit 0
else
for f in "${FAILED[@]}"; do
echo " 🔴 FAILED: $f"
done
exit 1
fi
+4 -18
View File
@@ -116,31 +116,17 @@ def check_model_probes():
results = []
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
# Single-host aliases: 30s initial timeout, retry once at 45s on failure
# gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
# Single-host aliases: 30s timeout each
# gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=30)
first_kind = None # Track first attempt's failure kind
if code == 000 and failure_kind:
# Retry once with longer timeout (45s) before declaring failure
first_kind = failure_kind
time.sleep(1)
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=45)
if code == 000 and failure_kind:
# Both attempts failed - report both kinds
if first_kind:
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
else:
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
# Report probe failure with kind, do not assert a service verdict
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)"))
elif code == 200:
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
elif code in (401, 403):
-138
View File
@@ -1,138 +0,0 @@
#!/bin/bash
# proxmox-monitor.sh — Proxmox Cluster + Docker Monitoring Health Check
# Implements proxmox-monitor.prose.md (check-health section)
#
# Legs: Prometheus, Grafana, Docker Stats Exporter, PVE Exporter, PBS GC
# All legs must return 200 for healthy status.
#
# Run: bash scripts/proxmox-monitor.sh
# Exits 0 if all probes pass, 1 if any fails.
set -uo pipefail
CT116_HOST="192.168.68.116"
FAILED=()
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
echo "=== Proxmox Monitor — $TIMESTAMP ==="
echo "Executed from: $(pwd -P)"
echo ""
# 1. Prometheus health (bound to 0.0.0.0:9090 on .116)
PROM_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}:9090/-/healthy 2>/dev/null)
PROM_CODE=$(printf '%s' "$PROM_CODE" | tr -d '[:space:]')
[ -n "$PROM_CODE" ] || PROM_CODE="000"
if [ "$PROM_CODE" = "200" ]; then
echo " ✅ Prometheus: alive"
else
echo " 🔴 Prometheus: probe-failed: ${CT116_HOST}:9090 (expected 200, got ${PROM_CODE})"
FAILED+=("prometheus")
fi
# 2. Grafana health (bound to 0.0.0.0:3001 on .116)
GRAF_CODE=$(curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://${CT116_HOST}:3001/api/health 2>/dev/null)
GRAF_CODE=$(printf '%s' "$GRAF_CODE" | tr -d '[:space:]')
[ -n "$GRAF_CODE" ] || GRAF_CODE="000"
if [ "$GRAF_CODE" = "200" ]; then
echo " ✅ Grafana: alive"
else
echo " 🔴 Grafana: probe-failed: ${CT116_HOST}:3001 (expected 200, got ${GRAF_CODE})"
FAILED+=("grafana")
fi
# 3. Docker Stats exporter (bound to 127.0.0.1:9324 on .116 — probe from .116 localhost)
DOCKER_CODE=$(ssh -o ConnectTimeout=5 -o BatchMode=yes root@${CT116_HOST} \
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://127.0.0.1:9324/metrics" 2>/dev/null)
DOCKER_CODE=$(printf '%s' "$DOCKER_CODE" | tr -d '[:space:]')
[ -n "$DOCKER_CODE" ] || DOCKER_CODE="000"
if [ "$DOCKER_CODE" = "200" ]; then
echo " ✅ Docker Stats: alive"
else
echo " 🔴 Docker Stats: probe-failed: CT116:127.0.0.1:9324 (expected 200, got ${DOCKER_CODE})"
FAILED+=("docker-stats")
fi
# 4. PVE Exporter (bound to 127.0.0.1:9221 on .116 — probe from .116 localhost)
PVE_CODE=$(ssh -o ConnectTimeout=5 -o BatchMode=yes root@${CT116_HOST} \
"curl -s -o /dev/null -w '%{http_code}' --connect-timeout 10 http://127.0.0.1:9221/metrics" 2>/dev/null)
PVE_CODE=$(printf '%s' "$PVE_CODE" | tr -d '[:space:]')
[ -n "$PVE_CODE" ] || PVE_CODE="000"
if [ "$PVE_CODE" = "200" ]; then
echo " ✅ PVE Exporter: alive"
else
echo " 🔴 PVE Exporter: probe-failed: CT116:127.0.0.1:9221 (expected 200, got ${PVE_CODE})"
FAILED+=("pve-exporter")
fi
# 5. PBS GC liveness (storepve-datastore GC must have run within 48h)
PBS_GC_OUTPUT=$(ssh -o ConnectTimeout=5 -o BatchMode=yes root@192.168.68.6 \
"pct exec 107 -- proxmox-backup-manager garbage-collection list --output-format json" 2>/dev/null)
PBS_GC_OUTPUT=$(printf '%s' "$PBS_GC_OUTPUT" | tr -d '[:space:]')
[ -n "$PBS_GC_OUTPUT" ] || PBS_GC_OUTPUT="000"
if [ "$PBS_GC_OUTPUT" = "000" ]; then
echo " 🔴 PBS GC: probe-failed: storepve:192.168.68.6 (expected JSON, got 000)"
FAILED+=("pbs-gc")
else
# Parse the JSON to get storepve-datastore's last-run-endtime and pending-bytes
PBS_GC_RESULT=$(echo "$PBS_GC_OUTPUT" | python3 -c "
import sys, json
try:
data = json.load(sys.stdin)
for store in data:
if store['store'] == 'storepve-datastore':
endtime = store.get('last-run-endtime')
pending = store.get('pending-bytes', 0)
if endtime is None or endtime == 0:
print('never-run')
else:
print(f'{endtime}|{pending}')
break
else:
print('absent')
except json.JSONDecodeError:
print('unparseable')
" 2>/dev/null)
if [ -z "$PBS_GC_RESULT" ] || [ "$PBS_GC_RESULT" = "unparseable" ]; then
echo " 🔴 PBS GC: probe-failed: storepve:192.168.68.6 (unparseable JSON)"
FAILED+=("pbs-gc")
elif [ "$PBS_GC_RESULT" = "absent" ]; then
echo " 🔴 PBS GC: never-run (storepve-datastore not found in GC list)"
FAILED+=("pbs-gc")
elif [ "$PBS_GC_RESULT" = "never-run" ]; then
echo " 🔴 PBS GC: never-run (storepve-datastore has no last-run-endtime)"
FAILED+=("pbs-gc")
else
# Parse the endtime|pending format
LAST_RUN_ENDTIME=$(echo "$PBS_GC_RESULT" | cut -d'|' -f1)
PENDING_BYTES=$(echo "$PBS_GC_RESULT" | cut -d'|' -f2)
# Convert epoch to age in hours
NOW_EPOCH=$(date -u +%s)
AGE_HOURS=$(( (NOW_EPOCH - LAST_RUN_ENDTIME) / 3600 ))
if [ $AGE_HOURS -gt 48 ]; then
echo " 🔴 PBS GC: stale (last run ${AGE_HOURS}h ago, pending-bytes: ${PENDING_BYTES} B)"
FAILED+=("pbs-gc")
else
echo " ✅ PBS GC: healthy (last run ${AGE_HOURS}h ago, pending-bytes: ${PENDING_BYTES} B)"
fi
fi
fi
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
if [ ${#FAILED[@]} -eq 0 ]; then
echo " ✅ All legs OK"
exit 0
else
for f in "${FAILED[@]}"; do
echo " 🔴 FAILED: $f"
done
exit 1
fi
-228
View File
@@ -1,228 +0,0 @@
#!/bin/bash
# test_infra_monitoring.sh — Asserts that probe calls use the documented targets.
#
# Strategy: stub curl and ssh on PATH to capture the exact arguments each leg
# builds, then assert the URL/port of every call. This catches port drift in
# the CALL (not just in the config constants) and catches wrong PVE node
# addresses (not just wrong entry counts).
#
# Run: bash scripts/test_infra_monitoring.sh
# Exits 0 if all assertions pass, 1 otherwise.
set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
SCRIPT="${SCRIPT_DIR}/infra-monitoring.sh"
PASS=0
FAIL=0
assert() {
local desc="$1" condition="$2"
if eval "$condition"; then
echo " ✅ $desc"
PASS=$((PASS+1))
else
echo " 🔴 $desc"
FAIL=$((FAIL+1))
fi
}
echo "=== test_infra_monitoring.sh ==="
echo ""
# ── Stub curl: capture argv to a file, return 200 ──────────────────────────
STUB_DIR=$(mktemp -d)
trap 'rm -rf "$STUB_DIR"' EXIT
# Stub curl: first arg after flags is the URL; capture all args
cat > "$STUB_DIR/curl" << 'STUBEOF'
#!/bin/bash
echo "$@" >> "${CURL_STUB_LOG:-/dev/null}"
# Print 200 for %{http_code}
printf '%s\n' "200"
exit 0
STUBEOF
chmod +x "$STUB_DIR/curl"
# Stub ssh: first arg after options is the remote command; capture it
cat > "$STUB_DIR/ssh" << 'SSTUBEOF'
#!/bin/bash
echo "SSH $@" >> "${SSH_STUB_LOG:-/dev/null}"
# The last arg is the remote command — extract and log curl args
for arg in "$@"; do
if [[ "$arg" == curl* ]]; then
echo "$arg" >> "${SSH_STUB_LOG:-/dev/null}"
fi
done
printf '%s\n' "200"
exit 0
SSTUBEOF
chmod +x "$STUB_DIR/ssh"
# ── Run the monitor with stubs ────────────────────────────────────────────
CURL_LOG="$STUB_DIR/curl_calls.log"
SSH_LOG="$STUB_DIR/ssh_calls.log"
touch "$CURL_LOG" "$SSH_LOG"
CURL_STUB_LOG="$CURL_LOG" SSH_STUB_LOG="$SSH_LOG" \
PATH="$STUB_DIR:$PATH" bash "$SCRIPT" > "$STUB_DIR/output.txt" 2>&1
# ── 1. Port drift detection (from actual curl invocations) ─────────────────
assert "Grafana probed at port 3001" \
'grep -q "http://192.168.68.116:3001/api/health" "$CURL_LOG"'
assert "Prometheus probed at port 9090" \
'grep -q "http://192.168.68.116:9090/-/healthy" "$CURL_LOG"'
assert "LiteLLM probed via nginx at port 80" \
'grep -q "http://192.168.68.116:80/litellm/health" "$CURL_LOG"'
assert "PVE API probed at port 8006" \
'grep -q ":8006/api2/json/version" "$CURL_LOG"'
assert "GPU exporter probed at port 9400" \
'grep -q ":9400/metrics" "$CURL_LOG"'
# ── 2. PVE API: exact node addresses (catches wrong IPs) ──────────────────
# Each real PVE node must be probed; CT 116 must NOT be in the PVE set
assert "PVE acerpve 192.168.68.9 probed" \
'grep -q "https://192.168.68.9:8006/api2/json/version" "$CURL_LOG"'
assert "PVE minipve 192.168.68.12 probed" \
'grep -q "https://192.168.68.12:8006/api2/json/version" "$CURL_LOG"'
assert "PVE storepve 192.168.68.6 probed" \
'grep -q "https://192.168.68.6:8006/api2/json/version" "$CURL_LOG"'
assert "PVE amdpve 192.168.68.15 probed" \
'grep -q "https://192.168.68.15:8006/api2/json/version" "$CURL_LOG"'
assert "PVE ocupve 192.168.68.5 probed" \
'grep -q "https://192.168.68.5:8006/api2/json/version" "$CURL_LOG"'
# CT 116 (.116) must NOT appear as a PVE API target
assert "CT 116 (.116) NOT probed as PVE API node" \
'! grep -q "https://192.168.68.116:8006" "$CURL_LOG"'
# ── 3. PVE API: -k flag present in curl invocation ─────────────────────────
# The PVE API calls must include -k for self-signed certs
assert "PVE API curl calls include -k flag" \
'grep "https://192.168.68.9:8006" "$CURL_LOG" | grep -q -- "-k"'
# ── 4. Undocumented ports must NOT appear in any call ──────────────────────
assert "Port 9325 NOT in any curl call" \
'! grep -q ":9325" "$CURL_LOG"'
assert "Port 9405 NOT in any curl call" \
'! grep -q ":9405" "$CURL_LOG"'
# ── 5. Docker Stats / PVE Exporter: SSH-probed at correct ports ────────────
assert "Docker Stats probed at port 9324 via SSH" \
'grep -q "9324" "$SSH_LOG"'
assert "PVE Exporter probed at port 9221 via SSH" \
'grep -q "9221" "$SSH_LOG"'
# ── 5a. Per-leg assertions (proves which leg owns which port) ──────────────
# The script source must show Docker Stats using $DOCKER_STATS_PORT and
# PVE Exporter using $PVE_EXPORTER_PORT in the correct leg sections
assert "Docker Stats leg uses DOCKER_STATS_PORT constant" \
'grep -A 3 "# 6. Docker Stats" "$SCRIPT" | grep -q "\$DOCKER_STATS_PORT"'
assert "PVE Exporter leg uses PVE_EXPORTER_PORT constant" \
'grep -A 3 "# 7. PVE Exporter" "$SCRIPT" | grep -q "\$PVE_EXPORTER_PORT"'
# Verify the constants themselves are set to the correct values
assert "DOCKER_STATS_PORT constant set to 9324" \
'grep -q "^DOCKER_STATS_PORT=\"9324\"" "$SCRIPT"'
assert "PVE_EXPORTER_PORT constant set to 9221" \
'grep -q "^PVE_EXPORTER_PORT=\"9221\"" "$SCRIPT"'
# ── 5b. Stale port 9323 (dockerd) NOT probed ──────────────────────────────
assert "Port 9323 (dockerd) NOT in SSH log" \
'! grep -q "9323" "$SSH_LOG"'
# ── 6. No stale ports in the script source (belt-and-suspenders) ──────────
assert "Port 9325 (historical) NOT in script source" \
'! grep -q "9325" "$SCRIPT"'
assert "Port 9405 (historical) NOT in script source" \
'! grep -q "9405" "$SCRIPT"'
# ── 7. Failure-line content includes non-empty kind ────────────────────────
TMP_DIR=$(mktemp -d)
trap 'rm -rf "$TMP_DIR"' EXIT
# Test 7a: Unexpected status (500) → kind should be unexpected:500
cat > "$TMP_DIR/curl" << 'EOF'
#!/bin/bash
# Stub: return 500 for Grafana port, 200 otherwise
for arg in "$@"; do
if [[ "$arg" == *":3001"* ]]; then
echo "500"
exit 0
fi
done
echo "200"
exit 0
EOF
chmod +x "$TMP_DIR/curl"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
GRAFANA_FAIL=$(echo "$OUT" | grep "Grafana: probe-failed")
assert "Grafana failure line exists (unexpected status)" \
'[[ -n "$GRAFANA_FAIL" ]]'
KIND=$(echo "$GRAFANA_FAIL" | grep -oP '\(<[^>]+>\)' | tr -d '()<>')
assert "Grafana failure kind is non-empty (unexpected status)" \
'[[ -n "$KIND" ]]'
# Test 7b: TLS error (000 + exit 60) → kind should be tls
cat > "$TMP_DIR/curl" << 'EOF'
#!/bin/bash
# Stub: return 000 and exit 60 for Grafana port (TLS error) on ALL invocations
for arg in "$@"; do
if [[ "$arg" == *":3001"* ]]; then
echo "000"
exit 60
fi
done
echo "200"
exit 0
EOF
chmod +x "$TMP_DIR/curl"
# Also stub ssh to return 000 + exit 60 for the retry
cat > "$TMP_DIR/ssh" << 'EOF'
#!/bin/bash
for arg in "$@"; do
if [[ "$arg" == curl* ]]; then
echo "000"
exit 60
fi
done
echo "200"
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
export PATH="$TMP_DIR:$PATH"
OUT=$(bash "$SCRIPT" 2>&1)
GRAFANA_FAIL=$(echo "$OUT" | grep "Grafana: probe-failed")
assert "Grafana failure line exists (TLS error)" \
'[[ -n "$GRAFANA_FAIL" ]]'
KIND=$(echo "$GRAFANA_FAIL" | grep -oP '\(<[^>]+>\)' | tr -d '()<>')
assert "Grafana failure kind is tls" \
'[[ "$KIND" == "tls" ]]'
# ── Summary ─────────────────────────────────────────────────────────────────
echo ""
echo "Results: ${PASS} passed, ${FAIL} failed"
if [ $FAIL -gt 0 ]; then
echo " 🔴 TESTS FAILED"
exit 1
else
echo " ✅ ALL TESTS PASSED"
exit 0
fi
-125
View File
@@ -1,125 +0,0 @@
#!/bin/bash
# test_proxmox_monitor.sh — Stub-driven tests for PBS GC liveness leg
set -uo pipefail
SCRIPT="$(cd "$(dirname "$0")" && pwd)/proxmox-monitor.sh"
PASS=0
FAIL=0
# ── Helpers ────────────────────────────────────────────────────────────────
assert() {
local desc="$1" cond="$2"
if eval "$cond" 2>/dev/null; then
echo " ✅ $desc"
PASS=$((PASS+1))
else
echo " 🔴 $desc"
FAIL=$((FAIL+1))
fi
}
# ── 1. Healthy: fresh GC, 0 B pending ─────────────────────────────────────
TMP_DIR=$(mktemp -d)
FRESH_ENDTIME=$(( $(date -u +%s) - (1 * 3600) )) # 1 hour ago
cat > "$TMP_DIR/ssh" << EOF
#!/bin/bash
# Stub: return valid JSON with fresh endtime
echo '[{"store":"storepve-datastore","last-run-endtime":$FRESH_ENDTIME,"pending-bytes":0}]'
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
PBS_LINE=$(echo "$OUT" | grep "PBS GC:" || echo "")
assert "Healthy: PBS line exists" '[[ -n "$PBS_LINE" ]]'
assert "Healthy: shows healthy verdict" '[[ "$PBS_LINE" == *"healthy"* ]]'
assert "Healthy: shows pending-bytes 0 B" '[[ "$PBS_LINE" == *"pending-bytes: 0 B"* ]]'
rm -rf "$TMP_DIR"
# ── 2. Stale: GC >48h old ─────────────────────────────────────────────────
TMP_DIR=$(mktemp -d)
OLD_ENDTIME=$(( $(date -u +%s) - (49 * 3600) )) # 49 hours ago
cat > "$TMP_DIR/ssh" << EOF
#!/bin/bash
echo '[{"store":"storepve-datastore","last-run-endtime":$OLD_ENDTIME,"pending-bytes":1048576}]'
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
PBS_LINE=$(echo "$OUT" | grep "PBS GC:" || echo "")
assert "Stale: PBS line exists" '[[ -n "$PBS_LINE" ]]'
assert "Stale: shows stale verdict" '[[ "$PBS_LINE" == *"stale"* ]]'
assert "Stale: shows pending-bytes 1048576 B" '[[ "$PBS_LINE" == *"pending-bytes: 1048576 B"* ]]'
rm -rf "$TMP_DIR"
# ── 3. Probe-failed: empty output ─────────────────────────────────────────
TMP_DIR=$(mktemp -d)
cat > "$TMP_DIR/ssh" << 'EOF'
#!/bin/bash
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
PBS_LINE=$(echo "$OUT" | grep "PBS GC:" || echo "")
assert "Probe-failed (empty): PBS line exists" '[[ -n "$PBS_LINE" ]]'
assert "Probe-failed (empty): shows probe-failed" '[[ "$PBS_LINE" == *"probe-failed"* ]]'
rm -rf "$TMP_DIR"
# ── 4. Probe-failed: unparseable output ───────────────────────────────────
TMP_DIR=$(mktemp -d)
cat > "$TMP_DIR/ssh" << 'EOF'
#!/bin/bash
echo "proxmox-backup-manager: command not found"
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
PBS_LINE=$(echo "$OUT" | grep "PBS GC:" || echo "")
assert "Probe-failed (unparseable): PBS line exists" '[[ -n "$PBS_LINE" ]]'
assert "Probe-failed (unparseable): shows probe-failed" '[[ "$PBS_LINE" == *"probe-failed"* ]]'
rm -rf "$TMP_DIR"
# ── 5. Null endtime: never-run ────────────────────────────────────────────
TMP_DIR=$(mktemp -d)
cat > "$TMP_DIR/ssh" << 'EOF'
#!/bin/bash
echo '[{"store":"storepve-datastore","last-run-endtime":null,"pending-bytes":0}]'
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
PBS_LINE=$(echo "$OUT" | grep "PBS GC:" || echo "")
assert "Null endtime: PBS line exists" '[[ -n "$PBS_LINE" ]]'
assert "Null endtime: shows never-run" '[[ "$PBS_LINE" == *"never-run"* ]]'
rm -rf "$TMP_DIR"
# ── 6. Datastore absent ───────────────────────────────────────────────────
TMP_DIR=$(mktemp -d)
cat > "$TMP_DIR/ssh" << 'EOF'
#!/bin/bash
echo '[{"store":"s3-archive","last-run-endtime":null,"pending-bytes":0}]'
exit 0
EOF
chmod +x "$TMP_DIR/ssh"
OUT=$(PATH="$TMP_DIR:$PATH" bash "$SCRIPT" 2>&1)
PBS_LINE=$(echo "$OUT" | grep "PBS GC:" || echo "")
assert "Datastore absent: PBS line exists" '[[ -n "$PBS_LINE" ]]'
assert "Datastore absent: shows never-run" '[[ "$PBS_LINE" == *"never-run"* ]]'
rm -rf "$TMP_DIR"
# ── Summary ────────────────────────────────────────────────────────────────
echo ""
echo "Results: ${PASS} passed, ${FAIL} failed"
if [ $FAIL -gt 0 ]; then
echo " 🔴 TESTS FAILED"
exit 1
else
echo " ✅ ALL TESTS PASSED"
exit 0
fi
+17 -36
View File
@@ -8,21 +8,12 @@
set -euo pipefail
# Credentials sourced from environment variable ZULIP_API_KEY (set by vault-backed start script)
# Never fall back to a literal key.
# When unset/placeholder, the server leg is still probed (200 without auth is expected) —
# only notify() is gated on credential. The pi/Tanko/kagentz
# legs do not need the Zulip API key. The placeholder is captain-held:
# zulip-health-credential-placeholder-20260913.
ZULIP_API_KEY="${ZULIP_API_KEY:-}"
# Never fall back to a literal key
ZULIP_API_KEY="${ZULIP_API_KEY:?ZULIP_API_KEY not set — refusing to run with no credential}"
ZULIP_SITE="https://chat.sysloggh.net"
ZULIP_EMAIL="abiba-bot@chat.sysloggh.net"
OWNER_ZULIP_ID="9"
# Track whether the Zulip API credential is usable
ZULIP_CRED_OK=1
if [ -z "$ZULIP_API_KEY" ] || [[ "$ZULIP_API_KEY" == *"placeholder"* ]] || [[ "$ZULIP_API_KEY" == *"REDACTED"* ]]; then
ZULIP_CRED_OK=0
fi
LOG="/root/zulip-health-monitor.log"
TIMESTAMP=$(date -u '+%Y-%m-%d %H:%M UTC')
@@ -33,29 +24,23 @@ notify() {
local severity="$1" msg="$2"
echo "[$severity] $msg"
# Zulip DM to owner (skip if no credential)
if [ "$ZULIP_CRED_OK" -eq 1 ]; then
local content="${severity} Zulip Monitor: ${msg}"
local form
form="type=private&to=%5B${OWNER_ZULIP_ID}%5D&content=$(python3 -c "import urllib.parse; print(urllib.parse.quote('''${content}'''))")"
curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" \
-d "${form}" > /dev/null 2>&1 || true
# Zulip stream post to #agent-hub on topic 'zulip-health'
local stream_content="${severity} Zulip Monitor: ${msg}"
curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" \
-d "type=stream&to=%5B7%5D&topic=zulip-health&content=$(printf '%s' "${stream_content}" | python3 -c "import sys,urllib.parse; print(urllib.parse.quote_from_bytes(sys.stdin.buffer.read()))")" \
> /dev/null 2>&1 \
|| echo " WARN: stream alert to #agent-hub (zulip-health) delivery failed (curl exit $?)">> "$LOG"
else
echo " ALERT SUPPRESSED (no credential): ${severity} ${msg}" >> "$LOG"
fi
# Zulip DM to owner
local content="${severity} Zulip Monitor: ${msg}"
local form
form="type=private&to=%5B${OWNER_ZULIP_ID}%5D&content=$(python3 -c "import urllib.parse; print(urllib.parse.quote('''${content}'''))")"
curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" \
-d "${form}" > /dev/null 2>&1 || true
# Zulip stream post to #agent-hub on topic 'zulip-health'
local stream_content="${severity} Zulip Monitor: ${msg}"
curl -sf -X POST "${ZULIP_SITE}/api/v1/messages" \
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" \
-d "type=stream&to=%5B7%5D&topic=zulip-health&content=$(printf '%s' "${stream_content}" | python3 -c "import sys,urllib.parse; print(urllib.parse.quote_from_bytes(sys.stdin.buffer.read()))")" \
> /dev/null 2>&1 \
|| echo " WARN: stream alert to #agent-hub (zulip-health) delivery failed (curl exit $?)" >> "$LOG"
}
# ── Global: Zulip Server ──
# F3: Always probe server regardless of credential — 200 without auth is expected
# (verified live: server_settings returns 200 with no credential or wrong key).
SERVER_CODE=$(curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 \
https://chat.sysloggh.net/api/v1/server_settings \
-u "${ZULIP_EMAIL}:${ZULIP_API_KEY}" 2>/dev/null) || SERVER_CODE="000"
@@ -198,11 +183,7 @@ fi
# ── Summary ──
if [ "$ISSUES" -eq 0 ]; then
if [ "$ZULIP_CRED_OK" -eq 0 ]; then
echo " Result: ✅ All healthy" >> "$LOG"
else
echo " Result: ✅ All healthy" >> "$LOG"
fi
echo " Result: ✅ All healthy" >> "$LOG"
else
echo " Result: 🔴 $ISSUES issue(s) found" >> "$LOG"
notify "🔴" "$ISSUES issue(s) found — check /root/zulip-health-monitor.log"
+2 -72
View File
@@ -24,7 +24,7 @@ BASE = """
model:
api_key: ""
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1
base_url: http://192.168.68.116/v1
max_tokens: 4096
default: syslog-auto
provider: harness
@@ -52,7 +52,7 @@ delegation:
custom_providers:
- name: harness
key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1
base_url: http://192.168.68.116/v1
"""
@@ -189,73 +189,3 @@ def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
assert code == 1, out
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_canonical_internal_path_passes(tmp_path):
"""Rule 5 must accept the canonical internal base from hermes-key-enforcement.prose.md."""
code, out = _run(
tmp_path,
"gpu-vision",
)
# Override the base_url in the config
cfg_text = BASE.format(alias="gpu-vision").replace(
" base_url: http://192.168.68.116/v1",
" base_url: http://192.168.68.116/litellm/v1",
)
code, out = _run_config(tmp_path, "canonical-internal.yaml", cfg_text)
assert code == 0, out
assert "RESULT: PASS" in out
# Verify the correct message is shown
assert "model.base_url is canonical" in out
def test_wrong_base_url_fails(tmp_path):
"""Rule 5 must reject paths outside the allowed list."""
cfg_text = BASE.format(alias="gpu-vision").replace(
" base_url: http://192.168.68.116/litellm/v1",
" base_url: http://192.168.68.116/litellm/v1/responses",
)
code, out = _run_config(tmp_path, "wrong-base.yaml", cfg_text)
assert code == 1, out
assert "RESULT: FAIL" in out
assert "model.base_url must be one of" in out
def test_public_host_path_passes(tmp_path):
"""Rule 5 must accept the public host base."""
cfg_text = BASE.format(alias="gpu-vision").replace(
" base_url: http://192.168.68.116/v1",
" base_url: https://litellm.sysloggh.net/v1",
)
code, out = _run_config(tmp_path, "public-host.yaml", cfg_text)
assert code == 0, out
assert "RESULT: PASS" in out
def test_old_rule5_check_would_fail_canonical(tmp_path):
"""
Proof that the OLD Rule 5 check would fail the canonical internal path.
This proves the bug existed before the fix.
"""
# OLD check expected /v1, so the canonical /litellm/v1 would have failed
canonical_cfg = BASE.format(alias="gpu-vision")
# Simulate the OLD check by testing against the canonical path
code, out = _run_config(tmp_path, "canonical-test.yaml", canonical_cfg)
# NEW check: canonical /litellm/v1 SHOULD pass
assert code == 0, out
assert "RESULT: PASS" in out
# OLD check expected /v1, so the internal /v1 would have passed
# NEW check: internal /v1 is non-canonical but working (WARN not FAIL)
old_cfg = BASE.format(alias="gpu-vision").replace(
" base_url: http://192.168.68.116/litellm/v1",
" base_url: http://192.168.68.116/v1",
)
code, out = _run_config(tmp_path, "old-check-test.yaml", old_cfg)
assert code == 0, out
assert "RESULT: PASS" in out
# NEW check should pass
new_cfg = BASE.format(alias="gpu-vision")
code, out = _run_config(tmp_path, "new-check-test.yaml", new_cfg)
assert code == 0, out
assert "RESULT: PASS" in out