361 lines
15 KiB
Bash
Executable File
361 lines
15 KiB
Bash
Executable File
#!/bin/bash
|
|
LLK="${LITELLM_API_KEY:?LITELLM_API_KEY must be set}"
|
|
# LiteLLM Health Check + Self-Heal — Automated (every 6 hours)
|
|
# Deployed from litellm-self-heal.prose.md contract
|
|
# Writes results to RA-H OS knowledge graph via MCP bridge
|
|
set -uo pipefail # -e removed: grep -c returns 1 on no-match, which is valid
|
|
|
|
RUN_ID="litellm-health-$(date +%Y%m%d-%H%M%S)"
|
|
TIMESTAMP=$(date -Iseconds)
|
|
BRIDGE="http://192.168.68.65:3100/mcp"
|
|
MASTER_KEY="sk-litellm-7f96080dd99b15c36bd4b333b58a6796" # admin only: /key/list. NEVER for inference
|
|
LITELLM_HOST="192.168.68.116"
|
|
source /etc/litellm-monitor.env 2>/dev/null # dedicated monitor agent key for inference tests
|
|
MONITOR_KEY="${LITELLM_MONITOR_KEY:-$MASTER_KEY}" # fallback only if env missing
|
|
GPU_DASHBOARD="http://192.168.68.24:9100"
|
|
LOG_DIR="/var/log/litellm"
|
|
RESULTS=""
|
|
ISSUES=0
|
|
FIXED=0
|
|
ESCALATED=0
|
|
DETAILS="[]"
|
|
|
|
mcp_call() {
|
|
local method="$1" tool="$2" args="$3"
|
|
curl -s -X POST "$BRIDGE" \
|
|
-H "Content-Type: application/json" \
|
|
-H "Accept: application/json, text/event-stream" \
|
|
-d "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"$method\",\"params\":$args}" 2>/dev/null
|
|
}
|
|
|
|
add_detail() {
|
|
local name="$1" status="$2" msg="$3"
|
|
DETAILS=$(echo "$DETAILS" | python3 -c "
|
|
import sys, json
|
|
d = json.load(sys.stdin)
|
|
d.append({'check':'$name','status':'$status','detail':'$msg'})
|
|
print(json.dumps(d))
|
|
" 2>/dev/null || echo "$DETAILS")
|
|
}
|
|
|
|
# ═══════════════════════════════════════════════════════
|
|
# PHASE 1: HEALTH CHECKS
|
|
# ═══════════════════════════════════════════════════════
|
|
|
|
echo "[$(date)] Starting LiteLLM health check — $RUN_ID"
|
|
|
|
# 1. Public endpoints
|
|
echo -n " UI... "
|
|
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 https://litellm.sysloggh.net/ui/ 2>/dev/null | grep -q 200; then
|
|
add_detail "public-ui" "pass" "200 OK"
|
|
echo "pass"
|
|
else
|
|
add_detail "public-ui" "fail" "non-200"
|
|
echo "FAIL"; ISSUES=$((ISSUES+1))
|
|
fi
|
|
|
|
echo -n " Docs... "
|
|
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 https://litellm.sysloggh.net/docs 2>/dev/null | grep -q 200; then
|
|
add_detail "public-docs" "pass" "200 OK"
|
|
echo "pass"
|
|
else
|
|
add_detail "public-docs" "fail" "non-200"
|
|
echo "FAIL"; ISSUES=$((ISSUES+1))
|
|
fi
|
|
|
|
# 2. LiteLLM liveliness
|
|
echo -n " Liveliness... "
|
|
LIVE=$(curl -s --connect-timeout 5 "http://${LITELLM_HOST}:4000/health/liveliness" 2>/dev/null)
|
|
if echo "$LIVE" | grep -qi "alive"; then
|
|
add_detail "liveliness" "pass" "$LIVE"
|
|
echo "pass"
|
|
else
|
|
add_detail "liveliness" "fail" "$LIVE"
|
|
echo "FAIL"; ISSUES=$((ISSUES+1))
|
|
fi
|
|
|
|
# 3. Container health
|
|
echo -n " Containers... "
|
|
CONTAINERS=$(docker ps --format '{{.Names}}:{{.Status}}' 2>/dev/null)
|
|
CT_COUNT=$(echo "$CONTAINERS" | wc -l)
|
|
# Only flag containers that are explicitly 'unhealthy', not ones without healthchecks
|
|
UNHEALTHY=$(echo "$CONTAINERS" | grep -c '(unhealthy)' 2>/dev/null || true)
|
|
UNHEALTHY=${UNHEALTHY:-0}
|
|
if [ "$UNHEALTHY" -eq 0 ] && [ "$CT_COUNT" -ge 8 ]; then
|
|
add_detail "containers" "pass" "$CT_COUNT containers healthy"
|
|
echo "pass ($CT_COUNT up)"
|
|
else
|
|
add_detail "containers" "fail" "$UNHEALTHY unhealthy of $CT_COUNT"
|
|
echo "FAIL ($UNHEALTHY/$CT_COUNT unhealthy)"; ISSUES=$((ISSUES+1))
|
|
# Auto-restart unhealthy containers
|
|
for ct in $(echo "$CONTAINERS" | grep '(unhealthy)' | cut -d: -f1); do
|
|
echo " Restarting $ct..."
|
|
docker restart "$ct" 2>/dev/null && FIXED=$((FIXED+1))
|
|
done
|
|
fi
|
|
|
|
# 4. GPU Fleet (full telemetry from gpu-monitor on .24:9100)
|
|
echo -n " GPU Fleet... "
|
|
GPU_DATA=$(curl -s --connect-timeout 10 "$GPU_DASHBOARD/gpu-data" 2>/dev/null)
|
|
GPU_HEALTH=$(echo "$GPU_DATA" | python3 -c "
|
|
import sys, json
|
|
d = json.load(sys.stdin)
|
|
s = d.get('summary',{})
|
|
alerts = d.get('alerts',[])
|
|
gpus = d.get('gpus',[])
|
|
strix = d.get('strix',{})
|
|
|
|
fleet = s.get('fleet_status','unknown')
|
|
gpu_count = s.get('gpu_count',0)
|
|
errors = s.get('gpu_errors',0)
|
|
cb_open = s.get('circuit_breakers_open',0)
|
|
strix_ok = strix.get('status','') == 'running'
|
|
alert_count = len(alerts)
|
|
critical_alerts = len([a for a in alerts if isinstance(a, dict) and a.get('level')=='critical'])
|
|
|
|
# Per-GPU details
|
|
for g in gpus:
|
|
if isinstance(g, dict):
|
|
n = g.get('gpu_name','?')[:30]
|
|
t = g.get('temp_c','?')
|
|
u = g.get('gpu_util_pct','?')
|
|
v = f\"{g.get('vram_used_mb',0)}/{g.get('vram_total_mb',0)}MB\"
|
|
print(f' {n}: {t}°C util={u}% vram={v}')
|
|
|
|
# Alert details
|
|
for a in alerts:
|
|
if isinstance(a, dict):
|
|
print(f' ⚠ {a.get(\"level\",\"?\")}: {a.get(\"metric\",\"?\")} — {a.get(\"value\",\"?\")}')
|
|
|
|
# Summary line
|
|
status = 'healthy' if fleet == 'healthy' and critical_alerts == 0 and errors == 0 else 'degraded'
|
|
print(f'SUMMARY: {status} | {gpu_count} GPUs | {errors} errors | {alert_count} alerts | CB open={cb_open} | Strix={\"running\" if strix_ok else \"down\"}')
|
|
" 2>/dev/null)
|
|
|
|
GPU_STATUS=$(echo "$GPU_HEALTH" | grep 'SUMMARY:' | cut -d' ' -f2-)
|
|
if echo "$GPU_STATUS" | grep -q '^healthy'; then
|
|
add_detail "gpu-fleet" "pass" "$GPU_STATUS"
|
|
echo "pass"
|
|
echo "$GPU_HEALTH" | grep -v 'SUMMARY:'
|
|
else
|
|
add_detail "gpu-fleet" "fail" "$GPU_STATUS"
|
|
echo "FAIL"
|
|
echo "$GPU_HEALTH"
|
|
ISSUES=$((ISSUES+1))
|
|
fi
|
|
|
|
# 5. Model inference tests
|
|
MODELS="gpu-dense strix-moe gpu-vision syslog-auto"
|
|
MODEL_FAILS=0
|
|
for model in $MODELS; do
|
|
echo -n " Model $model... "
|
|
HTTP_CODE=$(curl -s -o /dev/null -w "%{http_code}" --connect-timeout 30 \
|
|
-H "Authorization: Bearer $LLK" \
|
|
-H "Content-Type: application/json" \
|
|
-d "{\"model\":\"$model\",\"messages\":[{\"role\":\"user\",\"content\":\"ping\"}],\"max_tokens\":5}" \
|
|
"http://${LITELLM_HOST}:4000/v1/chat/completions" 2>/dev/null)
|
|
if [ "$HTTP_CODE" = "200" ]; then
|
|
add_detail "model-$model" "pass" "200 OK"
|
|
echo "pass"
|
|
else
|
|
add_detail "model-$model" "fail" "HTTP $HTTP_CODE"
|
|
echo "FAIL ($HTTP_CODE)"; ISSUES=$((ISSUES+1)); MODEL_FAILS=$((MODEL_FAILS+1))
|
|
fi
|
|
done
|
|
|
|
# 6. Agent keys
|
|
echo -n " Agent Keys... "
|
|
KEYS=$(curl -s --connect-timeout 10 \
|
|
-H "Authorization: Bearer $LLK" \
|
|
"http://${LITELLM_HOST}:4000/key/list?return_full_object=true" 2>/dev/null)
|
|
AGENT_COUNT=$(echo "$KEYS" | python3 -c "
|
|
import sys,json
|
|
d = json.load(sys.stdin)
|
|
agents = {'mumuni','tanko','kagenz0','koby','koonimo','abiba-pi','baggy'}
|
|
keys = d.get('keys',[])
|
|
found = sum(1 for k in keys if k.get('key_alias') in agents)
|
|
print(found)
|
|
" 2>/dev/null || echo 0)
|
|
if [ "$AGENT_COUNT" -ge 6 ]; then
|
|
add_detail "agent-keys" "pass" "$AGENT_COUNT agent keys present"
|
|
echo "pass ($AGENT_COUNT keys)"
|
|
else
|
|
add_detail "agent-keys" "fail" "only $AGENT_COUNT/7 agent keys found"
|
|
echo "FAIL"; ISSUES=$((ISSUES+1))
|
|
fi
|
|
|
|
# 7. Grafana
|
|
echo -n " Grafana... "
|
|
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 10 \
|
|
"http://${LITELLM_HOST}:3001/api/health" 2>/dev/null | grep -q 200; then
|
|
add_detail "grafana" "pass" "200 OK"
|
|
echo "pass"
|
|
else
|
|
add_detail "grafana" "warn" "Grafana unreachable (non-critical)"
|
|
echo "warn"
|
|
fi
|
|
|
|
# 8. 401 error count
|
|
echo -n " 401 Errors... "
|
|
ERR_401_RAW=$(docker logs harness-litellm --since 6h 2>&1 | grep 'Received=' 2>/dev/null || true)
|
|
ERR_401=$(echo "$ERR_401_RAW" | grep -c 'Received=' 2>/dev/null); ERR_401=${ERR_401:-0}
|
|
if [ "$ERR_401" -eq 0 ]; then
|
|
add_detail "401-errors" "pass" "0 auth errors in last 6h"
|
|
echo "pass (0)"
|
|
else
|
|
# Extract key patterns and source IPs from 401 errors
|
|
ERR_KEYS=$(echo "$ERR_401_RAW" | grep -oP 'Received=\K[^,]+' | sort -u | tr '\n' ' ')
|
|
ERR_IPS=$(docker logs harness-litellm --since 6h 2>&1 | grep -B2 'Received=' | grep -oP '\d+\.\d+\.\d+\.\d+' | sort -u | tr '\n' ' ')
|
|
|
|
DETAIL="$ERR_401 auth errors in last 6h | Keys: ${ERR_KEYS:-unknown} | Sources: ${ERR_IPS:-unknown}"
|
|
add_detail "401-errors" "warn" "$DETAIL"
|
|
echo "warn ($ERR_401 — keys: ${ERR_KEYS:-?}, sources: ${ERR_IPS:-?})"
|
|
ISSUES=$((ISSUES+1))
|
|
|
|
# Self-heal: analyze and attempt fix based on source IP type
|
|
if echo "$ERR_KEYS" | grep -q 'no-key-required'; then
|
|
echo " → 'no-key-required' = GPU-direct key being used as LiteLLM client key."
|
|
fi
|
|
for src_ip in $ERR_IPS; do
|
|
# Case 1: Docker network IPs (172.x) — this is the nginx proxy, meaning an external client
|
|
if echo "$src_ip" | grep -q '^172\.'; then
|
|
CT_NAME=$(docker inspect -f '{{.Name}}' $(docker ps -q) 2>/dev/null | while read n; do
|
|
ip=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' $n 2>/dev/null)
|
|
[ "$ip" = "$src_ip" ] && echo "$n"
|
|
done)
|
|
echo " → Source $src_ip is Docker container: ${CT_NAME:-unknown}"
|
|
if echo "$CT_NAME" | grep -q 'nginx'; then
|
|
echo " → 401 came through nginx proxy — external client, not locally fixable."
|
|
echo " → Checking nginx logs via docker logs for actual source..."
|
|
NGINX_SRC=$(timeout 8 docker logs harness-nginx --tail 2000 2>&1 | grep ' 401 ' | grep -oP '^\S+' | sort -u | tr '\n' ' ')
|
|
if [ -n "$NGINX_SRC" ]; then
|
|
echo " → Nginx upstream source(s): $NGINX_SRC"
|
|
fi
|
|
ESCALATED=$((ESCALATED+1))
|
|
else
|
|
echo " → Checking Docker container logs for clue..."
|
|
docker logs "$CT_NAME" --tail 50 2>/dev/null | grep -i 'litellm\|api.key\|auth' | tail -5
|
|
fi
|
|
# Case 2: Localhost or local CT116 IP — check locally
|
|
elif [ "$src_ip" = "127.0.0.1" ] || [ "$src_ip" = "192.168.68.116" ]; then
|
|
echo " → Source $src_ip is local (CT116). Checking local configs..."
|
|
LOCAL_CFG=$(grep -rl 'no-key-required' /root/.hermes/ /opt/inference-harness/ --include='*.yaml' --include='*.yml' --include='*.py' 2>/dev/null | grep -v 'litellm-health-check\|litellm_config\|\.bak' | head -5)
|
|
if [ -n "$LOCAL_CFG" ]; then
|
|
echo " → Found local references: $LOCAL_CFG"
|
|
else
|
|
echo " → No local config using no-key-required. Likely external via proxy."
|
|
ESCALATED=$((ESCALATED+1))
|
|
fi
|
|
# Case 3: Known agent host IPs — SSH and check
|
|
else
|
|
echo " → Checking remote source $src_ip for misconfigured agent..."
|
|
AGENT_CONFIGS=$(ssh -o ConnectTimeout=5 -o StrictHostKeyChecking=no root@$src_ip \
|
|
"grep -rl 'no-key-required\|api_key.*not-needed' /root/.hermes/config.yaml /home/*/.hermes/config.yaml 2>/dev/null" 2>/dev/null || true)
|
|
if [ -n "$AGENT_CONFIGS" ]; then
|
|
echo " → FOUND: agent config with GPU-direct key at $src_ip"
|
|
for cfg in $AGENT_CONFIGS; do
|
|
echo " → Attempting self-heal on $cfg..."
|
|
ssh -o ConnectTimeout=5 root@$src_ip \
|
|
"python3 -c \"
|
|
import yaml
|
|
with open('$cfg') as f: c = yaml.safe_load(f)
|
|
fixed = False
|
|
for section in ['agent', 'auxiliary']:
|
|
for sub in c.get(section, {}):
|
|
if isinstance(c[section].get(sub), dict):
|
|
ak = c[section][sub].get('api_key', '')
|
|
if ak in ['no-key-required', 'not-needed', 'no-k']:
|
|
c[section][sub]['api_key_env'] = 'LITELLM_API_KEY'
|
|
del c[section][sub]['api_key']
|
|
fixed = True
|
|
if fixed:
|
|
with open('$cfg', 'w') as f: yaml.dump(c, f, default_flow_style=False, allow_unicode=True)
|
|
print('Fixed. Restart gateway to apply.')
|
|
else:
|
|
print('No fixable sections.')
|
|
\"" 2>/dev/null && FIXED=$((FIXED+1)) || true
|
|
done
|
|
else
|
|
echo " → No misconfigured agent on $src_ip."
|
|
ESCALATED=$((ESCALATED+1))
|
|
fi
|
|
fi
|
|
done
|
|
fi
|
|
|
|
# ═══════════════════════════════════════════════════════
|
|
# PHASE 2: COMPILE STATUS
|
|
# ═══════════════════════════════════════════════════════
|
|
|
|
if [ "$ISSUES" -eq 0 ]; then
|
|
OVERALL="healthy"
|
|
elif [ "$ISSUES" -le 2 ]; then
|
|
OVERALL="degraded"
|
|
else
|
|
OVERALL="down"
|
|
fi
|
|
|
|
SUMMARY="LiteLLM Health: $OVERALL | Checks: $(echo "$DETAILS" | python3 -c "import sys,json;print(len(json.load(sys.stdin)))" 2>/dev/null || echo "?") | Issues: $ISSUES | Fixed: $FIXED | Escalated: $ESCALATED"
|
|
echo ""
|
|
echo "════════════════════════════════════════"
|
|
echo " Status: $OVERALL"
|
|
echo " Issues: $ISSUES found | $FIXED auto-fixed | $ESCALATED escalated"
|
|
echo "════════════════════════════════════════"
|
|
|
|
# ═══════════════════════════════════════════════════════
|
|
# PHASE 3: LOG TO FILESYSTEM
|
|
# ═══════════════════════════════════════════════════════
|
|
|
|
mkdir -p "$LOG_DIR"
|
|
REPORT=$(python3 -c "
|
|
import json
|
|
print(json.dumps({
|
|
'run_id': '$RUN_ID',
|
|
'timestamp': '$TIMESTAMP',
|
|
'overall_status': '$OVERALL',
|
|
'issues_found': $ISSUES,
|
|
'issues_fixed': $FIXED,
|
|
'issues_escalated': $ESCALATED,
|
|
'checks': $DETAILS
|
|
}, indent=2))
|
|
" 2>/dev/null)
|
|
|
|
echo "$REPORT" > "$LOG_DIR/${RUN_ID}.json"
|
|
# 9. GPU Monitor self-check
|
|
echo -n " GPU Monitor... "
|
|
if curl -s -o /dev/null -w "%{http_code}" --connect-timeout 5 "$GPU_DASHBOARD/health" 2>/dev/null | grep -q 200; then
|
|
add_detail "gpu-monitor" "pass" "Monitor server health OK"
|
|
echo "pass"
|
|
else
|
|
add_detail "gpu-monitor" "warn" "GPU monitor health endpoint unreachable"
|
|
echo "warn"
|
|
fi
|
|
|
|
echo ""
|
|
echo " Report saved: $LOG_DIR/${RUN_ID}.json"
|
|
|
|
# ═══════════════════════════════════════════════════════
|
|
# PHASE 4: FEED TO KNOWLEDGE GRAPH
|
|
# PHASE 4: LOG TO GITEA (hard rule: health logs NEVER go to knowledge graph)
|
|
# Logged to SyslogSolution/health-logs/litellm/{run_id}.json — versioned, searchable.
|
|
echo -n " Gitea... "
|
|
/opt/inference-harness/scripts/gitea-logger.sh litellm "${RUN_ID}.json" "${LOG_DIR}/${RUN_ID}.json"
|
|
|
|
# PHASE 5: NOTIFY ON ISSUES
|
|
# ═══════════════════════════════════════════════════════
|
|
|
|
if [ "$ISSUES" -gt 0 ]; then
|
|
echo ""
|
|
echo "⚠️ $ISSUES issue(s) detected. Creating relay alert..."
|
|
ALERT_TITLE="⚠ LiteLLM Health Alert — $OVERALL ($ISSUES issues)"
|
|
ALERT_BODY="$SUMMARY
|
|
|
|
Details:
|
|
$DETAILS"
|
|
mcp_call "tools/call" "createRelayNode" "{\"name\":\"createRelayNode\",\"arguments\":{\"title\":\"$ALERT_TITLE\",\"source\":\"$ALERT_BODY\",\"description\":\"LiteLLM health check failure alert\"}}" > /dev/null 2>&1
|
|
fi
|
|
|
|
echo ""
|
|
echo "[$(date)] Health check complete — $RUN_ID"
|
|
exit 0
|