Files
syslog-harness/scripts/litellm-health-check.sh
T

361 lines
15 KiB
Bash
Executable File

#!/bin/bash
LLK="${LITELLM_API_KEY:?LITELLM_API_KEY must be set}"
# LiteLLM Health Check + Self-Heal — Automated (every 6 hours)
# Deployed from litellm-self-heal.prose.md contract
# Writes results to RA-H OS knowledge graph via MCP bridge
set -uo pipefail # -e removed: grep -c returns 1 on no-match, which is valid
RUN_ID="litellm-health-$(date +%Y%m%d-%H%M%S)"
TIMESTAMP=$(date -Iseconds)
BRIDGE="http://192.168.68.65:3100/mcp"
MASTER_KEY="sk-litellm-7f96080dd99b15c36bd4b333b58a6796" # admin only: /key/list. NEVER for inference
LITELLM_HOST="192.168.68.116"
source /etc/litellm-monitor.env 2>/dev/null # dedicated monitor agent key for inference tests
MONITOR_KEY="${LITELLM_MONITOR_KEY:-$MASTER_KEY}" # fallback only if env missing
GPU_DASHBOARD="http://192.168.68.24:9100"
LOG_DIR="/var/log/litellm"
RESULTS=""
ISSUES=0
FIXED=0
ESCALATED=0
DETAILS="[]"
mcp_call() {
local method="$1" tool="$2" args="$3"
curl -s -X POST "$BRIDGE" \
-H "Content-Type: application/json" \
-H "Accept: application/json, text/event-stream" \
-d "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"$method\",\"params\":$args}" 2>/dev/null
}
add_detail() {
local name="$1" status="$2" msg="$3"
DETAILS=$(echo "$DETAILS" | python3 -c "
import sys, json
d = json.load(sys.stdin)
d.append({'check':'$name','status':'$status','detail':'$msg'})
print(json.dumps(d))
" 2>/dev/null || echo "$DETAILS")
}
# ═══════════════════════════════════════════════════════
# PHASE 1: HEALTH CHECKS
# ═══════════════════════════════════════════════════════
echo "[$(date)] Starting LiteLLM health check — $RUN_ID"
# 1. Public endpoints
echo -n " UI... "
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 https://litellm.sysloggh.net/ui/ 2>/dev/null | grep -q 200; then
add_detail "public-ui" "pass" "200 OK"
echo "pass"
else
add_detail "public-ui" "fail" "non-200"
echo "FAIL"; ISSUES=$((ISSUES+1))
fi
echo -n " Docs... "
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 https://litellm.sysloggh.net/docs 2>/dev/null | grep -q 200; then
add_detail "public-docs" "pass" "200 OK"
echo "pass"
else
add_detail "public-docs" "fail" "non-200"
echo "FAIL"; ISSUES=$((ISSUES+1))
fi
# 2. LiteLLM liveliness
echo -n " Liveliness... "
LIVE=$(curl -s --connect-timeout 5 "http://${LITELLM_HOST}:4000/health/liveliness" 2>/dev/null)
if echo "$LIVE" | grep -qi "alive"; then
add_detail "liveliness" "pass" "$LIVE"
echo "pass"
else
add_detail "liveliness" "fail" "$LIVE"
echo "FAIL"; ISSUES=$((ISSUES+1))
fi
# 3. Container health
echo -n " Containers... "
CONTAINERS=$(docker ps --format '{{.Names}}:{{.Status}}' 2>/dev/null)
CT_COUNT=$(echo "$CONTAINERS" | wc -l)
# Only flag containers that are explicitly 'unhealthy', not ones without healthchecks
UNHEALTHY=$(echo "$CONTAINERS" | grep -c '(unhealthy)' 2>/dev/null || true)
UNHEALTHY=${UNHEALTHY:-0}
if [ "$UNHEALTHY" -eq 0 ] && [ "$CT_COUNT" -ge 8 ]; then
add_detail "containers" "pass" "$CT_COUNT containers healthy"
echo "pass ($CT_COUNT up)"
else
add_detail "containers" "fail" "$UNHEALTHY unhealthy of $CT_COUNT"
echo "FAIL ($UNHEALTHY/$CT_COUNT unhealthy)"; ISSUES=$((ISSUES+1))
# Auto-restart unhealthy containers
for ct in $(echo "$CONTAINERS" | grep '(unhealthy)' | cut -d: -f1); do
echo " Restarting $ct..."
docker restart "$ct" 2>/dev/null && FIXED=$((FIXED+1))
done
fi
# 4. GPU Fleet (full telemetry from gpu-monitor on .24:9100)
echo -n " GPU Fleet... "
GPU_DATA=$(curl -s --connect-timeout 10 "$GPU_DASHBOARD/gpu-data" 2>/dev/null)
GPU_HEALTH=$(echo "$GPU_DATA" | python3 -c "
import sys, json
d = json.load(sys.stdin)
s = d.get('summary',{})
alerts = d.get('alerts',[])
gpus = d.get('gpus',[])
strix = d.get('strix',{})
fleet = s.get('fleet_status','unknown')
gpu_count = s.get('gpu_count',0)
errors = s.get('gpu_errors',0)
cb_open = s.get('circuit_breakers_open',0)
strix_ok = strix.get('status','') == 'running'
alert_count = len(alerts)
critical_alerts = len([a for a in alerts if isinstance(a, dict) and a.get('level')=='critical'])
# Per-GPU details
for g in gpus:
if isinstance(g, dict):
n = g.get('gpu_name','?')[:30]
t = g.get('temp_c','?')
u = g.get('gpu_util_pct','?')
v = f\"{g.get('vram_used_mb',0)}/{g.get('vram_total_mb',0)}MB\"
print(f' {n}: {t}°C util={u}% vram={v}')
# Alert details
for a in alerts:
if isinstance(a, dict):
print(f' ⚠ {a.get(\"level\",\"?\")}: {a.get(\"metric\",\"?\")} — {a.get(\"value\",\"?\")}')
# Summary line
status = 'healthy' if fleet == 'healthy' and critical_alerts == 0 and errors == 0 else 'degraded'
print(f'SUMMARY: {status} | {gpu_count} GPUs | {errors} errors | {alert_count} alerts | CB open={cb_open} | Strix={\"running\" if strix_ok else \"down\"}')
" 2>/dev/null)
GPU_STATUS=$(echo "$GPU_HEALTH" | grep 'SUMMARY:' | cut -d' ' -f2-)
if echo "$GPU_STATUS" | grep -q '^healthy'; then
add_detail "gpu-fleet" "pass" "$GPU_STATUS"
echo "pass"
echo "$GPU_HEALTH" | grep -v 'SUMMARY:'
else
add_detail "gpu-fleet" "fail" "$GPU_STATUS"
echo "FAIL"
echo "$GPU_HEALTH"
ISSUES=$((ISSUES+1))
fi
# 5. Model inference tests
MODELS="gpu-dense strix-moe gpu-vision syslog-auto"
MODEL_FAILS=0
for model in $MODELS; do
echo -n " Model $model... "
HTTP_CODE=$(curl -s -o /dev/null -w "%{http_code}" --connect-timeout 30 \
-H "Authorization: Bearer $MONITOR_KEY" \
-H "Content-Type: application/json" \
-d "{\"model\":\"$model\",\"messages\":[{\"role\":\"user\",\"content\":\"ping\"}],\"max_tokens\":5}" \
"http://${LITELLM_HOST}:4000/v1/chat/completions" 2>/dev/null)
if [ "$HTTP_CODE" = "200" ]; then
add_detail "model-$model" "pass" "200 OK"
echo "pass"
else
add_detail "model-$model" "fail" "HTTP $HTTP_CODE"
echo "FAIL ($HTTP_CODE)"; ISSUES=$((ISSUES+1)); MODEL_FAILS=$((MODEL_FAILS+1))
fi
done
# 6. Agent keys
echo -n " Agent Keys... "
KEYS=$(curl -s --connect-timeout 10 \
-H "Authorization: Bearer $MASTER_KEY" \
"http://${LITELLM_HOST}:4000/key/list?return_full_object=true" 2>/dev/null)
AGENT_COUNT=$(echo "$KEYS" | python3 -c "
import sys,json
d = json.load(sys.stdin)
agents = {'mumuni','tanko','kagenz0','koby','koonimo','abiba-pi','baggy'}
keys = d.get('keys',[])
found = sum(1 for k in keys if k.get('key_alias') in agents)
print(found)
" 2>/dev/null || echo 0)
if [ "$AGENT_COUNT" -ge 6 ]; then
add_detail "agent-keys" "pass" "$AGENT_COUNT agent keys present"
echo "pass ($AGENT_COUNT keys)"
else
add_detail "agent-keys" "fail" "only $AGENT_COUNT/7 agent keys found"
echo "FAIL"; ISSUES=$((ISSUES+1))
fi
# 7. Grafana
echo -n " Grafana... "
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 \
"http://${LITELLM_HOST}:3001/api/health" 2>/dev/null | grep -q 200; then
add_detail "grafana" "pass" "200 OK"
echo "pass"
else
add_detail "grafana" "warn" "Grafana unreachable (non-critical)"
echo "warn"
fi
# 8. 401 error count
echo -n " 401 Errors... "
ERR_401_RAW=$(docker logs harness-litellm --since 6h 2>&1 | grep 'Received=' 2>/dev/null || true)
ERR_401=$(echo "$ERR_401_RAW" | grep -c 'Received=' 2>/dev/null); ERR_401=${ERR_401:-0}
if [ "$ERR_401" -eq 0 ]; then
add_detail "401-errors" "pass" "0 auth errors in last 6h"
echo "pass (0)"
else
# Extract key patterns and source IPs from 401 errors
ERR_KEYS=$(echo "$ERR_401_RAW" | grep -oP 'Received=\K[^,]+' | sort -u | tr '\n' ' ')
ERR_IPS=$(docker logs harness-litellm --since 6h 2>&1 | grep -B2 'Received=' | grep -oP '\d+\.\d+\.\d+\.\d+' | sort -u | tr '\n' ' ')
DETAIL="$ERR_401 auth errors in last 6h | Keys: ${ERR_KEYS:-unknown} | Sources: ${ERR_IPS:-unknown}"
add_detail "401-errors" "warn" "$DETAIL"
echo "warn ($ERR_401 — keys: ${ERR_KEYS:-?}, sources: ${ERR_IPS:-?})"
ISSUES=$((ISSUES+1))
# Self-heal: analyze and attempt fix based on source IP type
if echo "$ERR_KEYS" | grep -q 'no-key-required'; then
echo " → 'no-key-required' = GPU-direct key being used as LiteLLM client key."
fi
for src_ip in $ERR_IPS; do
# Case 1: Docker network IPs (172.x) — this is the nginx proxy, meaning an external client
if echo "$src_ip" | grep -q '^172\.'; then
CT_NAME=$(docker inspect -f '{{.Name}}' $(docker ps -q) 2>/dev/null | while read n; do
ip=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' $n 2>/dev/null)
[ "$ip" = "$src_ip" ] && echo "$n"
done)
echo " → Source $src_ip is Docker container: ${CT_NAME:-unknown}"
if echo "$CT_NAME" | grep -q 'nginx'; then
echo " → 401 came through nginx proxy — external client, not locally fixable."
echo " → Checking nginx logs via docker logs for actual source..."
NGINX_SRC=$(timeout 8 docker logs harness-nginx --tail 2000 2>&1 | grep ' 401 ' | grep -oP '^\S+' | sort -u | tr '\n' ' ')
if [ -n "$NGINX_SRC" ]; then
echo " → Nginx upstream source(s): $NGINX_SRC"
fi
ESCALATED=$((ESCALATED+1))
else
echo " → Checking Docker container logs for clue..."
docker logs "$CT_NAME" --tail 50 2>/dev/null | grep -i 'litellm\|api.key\|auth' | tail -5
fi
# Case 2: Localhost or local CT116 IP — check locally
elif [ "$src_ip" = "127.0.0.1" ] || [ "$src_ip" = "192.168.68.116" ]; then
echo " → Source $src_ip is local (CT116). Checking local configs..."
LOCAL_CFG=$(grep -rl 'no-key-required' /root/.hermes/ /opt/inference-harness/ --include='*.yaml' --include='*.yml' --include='*.py' 2>/dev/null | grep -v 'litellm-health-check\|litellm_config\|\.bak' | head -5)
if [ -n "$LOCAL_CFG" ]; then
echo " → Found local references: $LOCAL_CFG"
else
echo " → No local config using no-key-required. Likely external via proxy."
ESCALATED=$((ESCALATED+1))
fi
# Case 3: Known agent host IPs — SSH and check
else
echo " → Checking remote source $src_ip for misconfigured agent..."
AGENT_CONFIGS=$(ssh -o ConnectTimeout=5 -o StrictHostKeyChecking=no root@$src_ip \
"grep -rl 'no-key-required\|api_key.*not-needed' /root/.hermes/config.yaml /home/*/.hermes/config.yaml 2>/dev/null" 2>/dev/null || true)
if [ -n "$AGENT_CONFIGS" ]; then
echo " → FOUND: agent config with GPU-direct key at $src_ip"
for cfg in $AGENT_CONFIGS; do
echo " → Attempting self-heal on $cfg..."
ssh -o ConnectTimeout=5 root@$src_ip \
"python3 -c \"
import yaml
with open('$cfg') as f: c = yaml.safe_load(f)
fixed = False
for section in ['agent', 'auxiliary']:
for sub in c.get(section, {}):
if isinstance(c[section].get(sub), dict):
ak = c[section][sub].get('api_key', '')
if ak in ['no-key-required', 'not-needed', 'no-k']:
c[section][sub]['api_key_env'] = 'LITELLM_API_KEY'
del c[section][sub]['api_key']
fixed = True
if fixed:
with open('$cfg', 'w') as f: yaml.dump(c, f, default_flow_style=False, allow_unicode=True)
print('Fixed. Restart gateway to apply.')
else:
print('No fixable sections.')
\"" 2>/dev/null && FIXED=$((FIXED+1)) || true
done
else
echo " → No misconfigured agent on $src_ip."
ESCALATED=$((ESCALATED+1))
fi
fi
done
fi
# ═══════════════════════════════════════════════════════
# PHASE 2: COMPILE STATUS
# ═══════════════════════════════════════════════════════
if [ "$ISSUES" -eq 0 ]; then
OVERALL="healthy"
elif [ "$ISSUES" -le 2 ]; then
OVERALL="degraded"
else
OVERALL="down"
fi
SUMMARY="LiteLLM Health: $OVERALL | Checks: $(echo "$DETAILS" | python3 -c "import sys,json;print(len(json.load(sys.stdin)))" 2>/dev/null || echo "?") | Issues: $ISSUES | Fixed: $FIXED | Escalated: $ESCALATED"
echo ""
echo "════════════════════════════════════════"
echo " Status: $OVERALL"
echo " Issues: $ISSUES found | $FIXED auto-fixed | $ESCALATED escalated"
echo "════════════════════════════════════════"
# ═══════════════════════════════════════════════════════
# PHASE 3: LOG TO FILESYSTEM
# ═══════════════════════════════════════════════════════
mkdir -p "$LOG_DIR"
REPORT=$(python3 -c "
import json
print(json.dumps({
'run_id': '$RUN_ID',
'timestamp': '$TIMESTAMP',
'overall_status': '$OVERALL',
'issues_found': $ISSUES,
'issues_fixed': $FIXED,
'issues_escalated': $ESCALATED,
'checks': $DETAILS
}, indent=2))
" 2>/dev/null)
echo "$REPORT" > "$LOG_DIR/${RUN_ID}.json"
# 9. GPU Monitor self-check
echo -n " GPU Monitor... "
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 5 "$GPU_DASHBOARD/health" 2>/dev/null | grep -q 200; then
add_detail "gpu-monitor" "pass" "Monitor server health OK"
echo "pass"
else
add_detail "gpu-monitor" "warn" "GPU monitor health endpoint unreachable"
echo "warn"
fi
echo ""
echo " Report saved: $LOG_DIR/${RUN_ID}.json"
# ═══════════════════════════════════════════════════════
# PHASE 4: FEED TO KNOWLEDGE GRAPH
# PHASE 4: LOG TO GITEA (hard rule: health logs NEVER go to knowledge graph)
# Logged to SyslogSolution/health-logs/litellm/{run_id}.json — versioned, searchable.
echo -n " Gitea... "
/opt/inference-harness/scripts/gitea-logger.sh litellm "${RUN_ID}.json" "${LOG_DIR}/${RUN_ID}.json"
# PHASE 5: NOTIFY ON ISSUES
# ═══════════════════════════════════════════════════════
if [ "$ISSUES" -gt 0 ]; then
echo ""
echo "⚠️ $ISSUES issue(s) detected. Creating relay alert..."
ALERT_TITLE="⚠ LiteLLM Health Alert — $OVERALL ($ISSUES issues)"
ALERT_BODY="$SUMMARY
Details:
$DETAILS"
mcp_call "tools/call" "createRelayNode" "{\"name\":\"createRelayNode\",\"arguments\":{\"title\":\"$ALERT_TITLE\",\"source\":\"$ALERT_BODY\",\"description\":\"LiteLLM health check failure alert\"}}" > /dev/null 2>&1
fi
echo ""
echo "[$(date)] Health check complete — $RUN_ID"
exit 0