#!/bin/bash # LiteLLM Health Check + Self-Heal — Automated (every 6 hours) # Deployed from litellm-self-heal.prose.md contract # Writes results to RA-H OS knowledge graph via MCP bridge set -uo pipefail # -e removed: grep -c returns 1 on no-match, which is valid RUN_ID="litellm-health-$(date +%Y%m%d-%H%M%S)" TIMESTAMP=$(date -Iseconds) BRIDGE="http://192.168.68.65:3100/mcp" MASTER_KEY="sk-litellm-7f96080dd99b15c36bd4b333b58a6796" # admin only: /key/list. NEVER for inference LITELLM_HOST="192.168.68.116" source /etc/litellm-monitor.env 2>/dev/null # dedicated monitor agent key for inference tests MONITOR_KEY="${LITELLM_MONITOR_KEY:-$MASTER_KEY}" # fallback only if env missing GPU_DASHBOARD="http://192.168.68.24:9100" LOG_DIR="/var/log/litellm" RESULTS="" ISSUES=0 FIXED=0 ESCALATED=0 DETAILS="[]" mcp_call() { local method="$1" tool="$2" args="$3" curl -s -X POST "$BRIDGE" \ -H "Content-Type: application/json" \ -H "Accept: application/json, text/event-stream" \ -d "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"$method\",\"params\":$args}" 2>/dev/null } add_detail() { local name="$1" status="$2" msg="$3" DETAILS=$(echo "$DETAILS" | python3 -c " import sys, json d = json.load(sys.stdin) d.append({'check':'$name','status':'$status','detail':'$msg'}) print(json.dumps(d)) " 2>/dev/null || echo "$DETAILS") } # ═══════════════════════════════════════════════════════ # PHASE 1: HEALTH CHECKS # ═══════════════════════════════════════════════════════ echo "[$(date)] Starting LiteLLM health check — $RUN_ID" # 1. Public endpoints echo -n " UI... " if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 https://litellm.sysloggh.net/ui/ 2>/dev/null | grep -q 200; then add_detail "public-ui" "pass" "200 OK" echo "pass" else add_detail "public-ui" "fail" "non-200" echo "FAIL"; ISSUES=$((ISSUES+1)) fi echo -n " Docs... " if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 https://litellm.sysloggh.net/docs 2>/dev/null | grep -q 200; then add_detail "public-docs" "pass" "200 OK" echo "pass" else add_detail "public-docs" "fail" "non-200" echo "FAIL"; ISSUES=$((ISSUES+1)) fi # 2. LiteLLM liveliness echo -n " Liveliness... " LIVE=$(curl -s --connect-timeout 5 "http://${LITELLM_HOST}:4000/health/liveliness" 2>/dev/null) if echo "$LIVE" | grep -qi "alive"; then add_detail "liveliness" "pass" "$LIVE" echo "pass" else add_detail "liveliness" "fail" "$LIVE" echo "FAIL"; ISSUES=$((ISSUES+1)) fi # 3. Container health echo -n " Containers... " CONTAINERS=$(docker ps --format '{{.Names}}:{{.Status}}' 2>/dev/null) CT_COUNT=$(echo "$CONTAINERS" | wc -l) # Only flag containers that are explicitly 'unhealthy', not ones without healthchecks UNHEALTHY=$(echo "$CONTAINERS" | grep -c '(unhealthy)' 2>/dev/null || true) UNHEALTHY=${UNHEALTHY:-0} if [ "$UNHEALTHY" -eq 0 ] && [ "$CT_COUNT" -ge 8 ]; then add_detail "containers" "pass" "$CT_COUNT containers healthy" echo "pass ($CT_COUNT up)" else add_detail "containers" "fail" "$UNHEALTHY unhealthy of $CT_COUNT" echo "FAIL ($UNHEALTHY/$CT_COUNT unhealthy)"; ISSUES=$((ISSUES+1)) # Auto-restart unhealthy containers for ct in $(echo "$CONTAINERS" | grep '(unhealthy)' | cut -d: -f1); do echo " Restarting $ct..." docker restart "$ct" 2>/dev/null && FIXED=$((FIXED+1)) done fi # 4. GPU Fleet (full telemetry from gpu-monitor on .24:9100) echo -n " GPU Fleet... " GPU_DATA=$(curl -s --connect-timeout 10 "$GPU_DASHBOARD/gpu-data" 2>/dev/null) GPU_HEALTH=$(echo "$GPU_DATA" | python3 -c " import sys, json d = json.load(sys.stdin) s = d.get('summary',{}) alerts = d.get('alerts',[]) gpus = d.get('gpus',[]) strix = d.get('strix',{}) fleet = s.get('fleet_status','unknown') gpu_count = s.get('gpu_count',0) errors = s.get('gpu_errors',0) cb_open = s.get('circuit_breakers_open',0) strix_ok = strix.get('status','') == 'running' alert_count = len(alerts) critical_alerts = len([a for a in alerts if isinstance(a, dict) and a.get('level')=='critical']) # Per-GPU details for g in gpus: if isinstance(g, dict): n = g.get('gpu_name','?')[:30] t = g.get('temp_c','?') u = g.get('gpu_util_pct','?') v = f\"{g.get('vram_used_mb',0)}/{g.get('vram_total_mb',0)}MB\" print(f' {n}: {t}°C util={u}% vram={v}') # Alert details for a in alerts: if isinstance(a, dict): print(f' ⚠ {a.get(\"level\",\"?\")}: {a.get(\"metric\",\"?\")} — {a.get(\"value\",\"?\")}') # Summary line status = 'healthy' if fleet == 'healthy' and critical_alerts == 0 and errors == 0 else 'degraded' print(f'SUMMARY: {status} | {gpu_count} GPUs | {errors} errors | {alert_count} alerts | CB open={cb_open} | Strix={\"running\" if strix_ok else \"down\"}') " 2>/dev/null) GPU_STATUS=$(echo "$GPU_HEALTH" | grep 'SUMMARY:' | cut -d' ' -f2-) if echo "$GPU_STATUS" | grep -q '^healthy'; then add_detail "gpu-fleet" "pass" "$GPU_STATUS" echo "pass" echo "$GPU_HEALTH" | grep -v 'SUMMARY:' else add_detail "gpu-fleet" "fail" "$GPU_STATUS" echo "FAIL" echo "$GPU_HEALTH" ISSUES=$((ISSUES+1)) fi # 5. Model inference tests MODELS="gpu-dense strix-moe gpu-vision syslog-auto" MODEL_FAILS=0 for model in $MODELS; do echo -n " Model $model... " HTTP_CODE=$(curl -s -o /dev/null -w "%{http_code}" --connect-timeout 30 \ -H "Authorization: Bearer $MONITOR_KEY" \ -H "Content-Type: application/json" \ -d "{\"model\":\"$model\",\"messages\":[{\"role\":\"user\",\"content\":\"ping\"}],\"max_tokens\":5}" \ "http://${LITELLM_HOST}:4000/v1/chat/completions" 2>/dev/null) if [ "$HTTP_CODE" = "200" ]; then add_detail "model-$model" "pass" "200 OK" echo "pass" else add_detail "model-$model" "fail" "HTTP $HTTP_CODE" echo "FAIL ($HTTP_CODE)"; ISSUES=$((ISSUES+1)); MODEL_FAILS=$((MODEL_FAILS+1)) fi done # 6. Agent keys echo -n " Agent Keys... " KEYS=$(curl -s --connect-timeout 10 \ -H "Authorization: Bearer $MASTER_KEY" \ "http://${LITELLM_HOST}:4000/key/list?return_full_object=true" 2>/dev/null) AGENT_COUNT=$(echo "$KEYS" | python3 -c " import sys,json d = json.load(sys.stdin) agents = {'mumuni','tanko','kagenz0','koby','koonimo','abiba-pi','baggy'} keys = d.get('keys',[]) found = sum(1 for k in keys if k.get('key_alias') in agents) print(found) " 2>/dev/null || echo 0) if [ "$AGENT_COUNT" -ge 6 ]; then add_detail "agent-keys" "pass" "$AGENT_COUNT agent keys present" echo "pass ($AGENT_COUNT keys)" else add_detail "agent-keys" "fail" "only $AGENT_COUNT/7 agent keys found" echo "FAIL"; ISSUES=$((ISSUES+1)) fi # 7. Grafana echo -n " Grafana... " if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 \ "http://${LITELLM_HOST}:3001/api/health" 2>/dev/null | grep -q 200; then add_detail "grafana" "pass" "200 OK" echo "pass" else add_detail "grafana" "warn" "Grafana unreachable (non-critical)" echo "warn" fi # 8. 401 error count echo -n " 401 Errors... " ERR_401_RAW=$(docker logs harness-litellm --since 6h 2>&1 | grep 'Received=' 2>/dev/null || true) ERR_401=$(echo "$ERR_401_RAW" | grep -c 'Received=' 2>/dev/null); ERR_401=${ERR_401:-0} if [ "$ERR_401" -eq 0 ]; then add_detail "401-errors" "pass" "0 auth errors in last 6h" echo "pass (0)" else # Extract key patterns and source IPs from 401 errors ERR_KEYS=$(echo "$ERR_401_RAW" | grep -oP 'Received=\K[^,]+' | sort -u | tr '\n' ' ') ERR_IPS=$(docker logs harness-litellm --since 6h 2>&1 | grep -B2 'Received=' | grep -oP '\d+\.\d+\.\d+\.\d+' | sort -u | tr '\n' ' ') DETAIL="$ERR_401 auth errors in last 6h | Keys: ${ERR_KEYS:-unknown} | Sources: ${ERR_IPS:-unknown}" add_detail "401-errors" "warn" "$DETAIL" echo "warn ($ERR_401 — keys: ${ERR_KEYS:-?}, sources: ${ERR_IPS:-?})" ISSUES=$((ISSUES+1)) # Self-heal: analyze and attempt fix based on source IP type if echo "$ERR_KEYS" | grep -q 'no-key-required'; then echo " → 'no-key-required' = GPU-direct key being used as LiteLLM client key." fi for src_ip in $ERR_IPS; do # Case 1: Docker network IPs (172.x) — this is the nginx proxy, meaning an external client if echo "$src_ip" | grep -q '^172\.'; then CT_NAME=$(docker inspect -f '{{.Name}}' $(docker ps -q) 2>/dev/null | while read n; do ip=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' $n 2>/dev/null) [ "$ip" = "$src_ip" ] && echo "$n" done) echo " → Source $src_ip is Docker container: ${CT_NAME:-unknown}" if echo "$CT_NAME" | grep -q 'nginx'; then echo " → 401 came through nginx proxy — external client, not locally fixable." echo " → Checking nginx logs via docker logs for actual source..." NGINX_SRC=$(timeout 8 docker logs harness-nginx --tail 2000 2>&1 | grep ' 401 ' | grep -oP '^\S+' | sort -u | tr '\n' ' ') if [ -n "$NGINX_SRC" ]; then echo " → Nginx upstream source(s): $NGINX_SRC" fi ESCALATED=$((ESCALATED+1)) else echo " → Checking Docker container logs for clue..." docker logs "$CT_NAME" --tail 50 2>/dev/null | grep -i 'litellm\|api.key\|auth' | tail -5 fi # Case 2: Localhost or local CT116 IP — check locally elif [ "$src_ip" = "127.0.0.1" ] || [ "$src_ip" = "192.168.68.116" ]; then echo " → Source $src_ip is local (CT116). Checking local configs..." LOCAL_CFG=$(grep -rl 'no-key-required' /root/.hermes/ /opt/inference-harness/ --include='*.yaml' --include='*.yml' --include='*.py' 2>/dev/null | grep -v 'litellm-health-check\|litellm_config\|\.bak' | head -5) if [ -n "$LOCAL_CFG" ]; then echo " → Found local references: $LOCAL_CFG" else echo " → No local config using no-key-required. Likely external via proxy." ESCALATED=$((ESCALATED+1)) fi # Case 3: Known agent host IPs — SSH and check else echo " → Checking remote source $src_ip for misconfigured agent..." AGENT_CONFIGS=$(ssh -o ConnectTimeout=5 -o StrictHostKeyChecking=no root@$src_ip \ "grep -rl 'no-key-required\|api_key.*not-needed' /root/.hermes/config.yaml /home/*/.hermes/config.yaml 2>/dev/null" 2>/dev/null || true) if [ -n "$AGENT_CONFIGS" ]; then echo " → FOUND: agent config with GPU-direct key at $src_ip" for cfg in $AGENT_CONFIGS; do echo " → Attempting self-heal on $cfg..." ssh -o ConnectTimeout=5 root@$src_ip \ "python3 -c \" import yaml with open('$cfg') as f: c = yaml.safe_load(f) fixed = False for section in ['agent', 'auxiliary']: for sub in c.get(section, {}): if isinstance(c[section].get(sub), dict): ak = c[section][sub].get('api_key', '') if ak in ['no-key-required', 'not-needed', 'no-k']: c[section][sub]['api_key_env'] = 'LITELLM_API_KEY' del c[section][sub]['api_key'] fixed = True if fixed: with open('$cfg', 'w') as f: yaml.dump(c, f, default_flow_style=False, allow_unicode=True) print('Fixed. Restart gateway to apply.') else: print('No fixable sections.') \"" 2>/dev/null && FIXED=$((FIXED+1)) || true done else echo " → No misconfigured agent on $src_ip." ESCALATED=$((ESCALATED+1)) fi fi done fi # ═══════════════════════════════════════════════════════ # PHASE 2: COMPILE STATUS # ═══════════════════════════════════════════════════════ if [ "$ISSUES" -eq 0 ]; then OVERALL="healthy" elif [ "$ISSUES" -le 2 ]; then OVERALL="degraded" else OVERALL="down" fi SUMMARY="LiteLLM Health: $OVERALL | Checks: $(echo "$DETAILS" | python3 -c "import sys,json;print(len(json.load(sys.stdin)))" 2>/dev/null || echo "?") | Issues: $ISSUES | Fixed: $FIXED | Escalated: $ESCALATED" echo "" echo "════════════════════════════════════════" echo " Status: $OVERALL" echo " Issues: $ISSUES found | $FIXED auto-fixed | $ESCALATED escalated" echo "════════════════════════════════════════" # ═══════════════════════════════════════════════════════ # PHASE 3: LOG TO FILESYSTEM # ═══════════════════════════════════════════════════════ mkdir -p "$LOG_DIR" REPORT=$(python3 -c " import json print(json.dumps({ 'run_id': '$RUN_ID', 'timestamp': '$TIMESTAMP', 'overall_status': '$OVERALL', 'issues_found': $ISSUES, 'issues_fixed': $FIXED, 'issues_escalated': $ESCALATED, 'checks': $DETAILS }, indent=2)) " 2>/dev/null) echo "$REPORT" > "$LOG_DIR/${RUN_ID}.json" # 9. GPU Monitor self-check echo -n " GPU Monitor... " if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 5 "$GPU_DASHBOARD/health" 2>/dev/null | grep -q 200; then add_detail "gpu-monitor" "pass" "Monitor server health OK" echo "pass" else add_detail "gpu-monitor" "warn" "GPU monitor health endpoint unreachable" echo "warn" fi echo "" echo " Report saved: $LOG_DIR/${RUN_ID}.json" # ═══════════════════════════════════════════════════════ # PHASE 4: FEED TO KNOWLEDGE GRAPH # PHASE 4: LOG TO GITEA (hard rule: health logs NEVER go to knowledge graph) # Logged to SyslogSolution/health-logs/litellm/{run_id}.json — versioned, searchable. echo -n " Gitea... " /opt/inference-harness/scripts/gitea-logger.sh litellm "${RUN_ID}.json" "${LOG_DIR}/${RUN_ID}.json" # PHASE 5: NOTIFY ON ISSUES # ═══════════════════════════════════════════════════════ if [ "$ISSUES" -gt 0 ]; then echo "" echo "⚠️ $ISSUES issue(s) detected. Creating relay alert..." ALERT_TITLE="⚠ LiteLLM Health Alert — $OVERALL ($ISSUES issues)" ALERT_BODY="$SUMMARY Details: $DETAILS" mcp_call "tools/call" "createRelayNode" "{\"name\":\"createRelayNode\",\"arguments\":{\"title\":\"$ALERT_TITLE\",\"source\":\"$ALERT_BODY\",\"description\":\"LiteLLM health check failure alert\"}}" > /dev/null 2>&1 fi echo "" echo "[$(date)] Health check complete — $RUN_ID" exit 0