1. Timeout fix for pool alias (syslog-auto): - Single-host aliases (gpu-dense, gpu-vision, strix-moe): 30s timeout - Pool alias (syslog-auto): 60s timeout, retry once on 000 before failing - Cold first request to pool alias can take ~13s; 10s was too short 2. Remove DEBUG prints from output: - Removed 'DEBUG: keylen=...' and 'DEBUG: response=...' lines - These leaked key inventory to status logs - Success output now shows only counts (e.g., 'Admin Key List: 10 keys')
239 lines
8.9 KiB
Python
Executable File
239 lines
8.9 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
LiteLLM Health Check - Contract executor
|
|
Runs all checks defined in litellm-health.prose.md and reports results.
|
|
"""
|
|
|
|
import subprocess
|
|
import sys
|
|
import json
|
|
import time
|
|
import random
|
|
|
|
# Configuration
|
|
BACKEND_HOST = "192.168.68.116"
|
|
GPU_HOSTS = {
|
|
"gpu-dense": "192.168.68.8",
|
|
"gpu-vision": "192.168.68.110",
|
|
"strix-moe": "192.168.68.15"
|
|
}
|
|
|
|
def run_command(cmd, timeout=15):
|
|
"""Run a command and return (exit_code, stdout, stderr)"""
|
|
try:
|
|
result = subprocess.run(
|
|
cmd,
|
|
shell=True,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=timeout
|
|
)
|
|
return result.returncode, result.stdout.strip(), result.stderr.strip()
|
|
except subprocess.TimeoutExpired:
|
|
return 1, "", "TIMEOUT"
|
|
except Exception as e:
|
|
return 1, "", str(e)
|
|
|
|
def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False):
|
|
"""Probe HTTP endpoint and return status code"""
|
|
cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout)
|
|
if method == "POST":
|
|
cmd += " -X POST"
|
|
if bearer_token:
|
|
cmd += " -H 'Authorization: Bearer " + bearer_token + "'"
|
|
if data:
|
|
cmd += " -H 'Content-Type: application/json' -d '" + data + "'"
|
|
if follow_redirects:
|
|
cmd += " -L"
|
|
cmd += " '" + url + "'"
|
|
|
|
rc, stdout, stderr = run_command(cmd, timeout)
|
|
if rc != 0 and "TIMEOUT" not in stderr:
|
|
return 000 # Connection failed
|
|
|
|
return int(stdout) if stdout.isdigit() else 000
|
|
|
|
def check_liveliness():
|
|
"""Step 1: Liveliness probe"""
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness")
|
|
return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)"
|
|
|
|
def check_containers():
|
|
"""Step 2: Container health via SSH"""
|
|
cmd = "ssh -o BatchMode=yes -o ConnectTimeout=5 -o StrictHostKeyChecking=no root@192.168.68.116 'docker ps --format \"{{.Names}} {{.Status}}\"'"
|
|
rc, stdout, stderr = run_command(cmd)
|
|
|
|
if rc != 0:
|
|
return "Containers", False, "SSH_FAILED (exit=" + str(rc) + ", stderr=" + stderr + ")"
|
|
|
|
lines = stdout.split('\n') if stdout else []
|
|
container_count = len([l for l in lines if l.strip()])
|
|
healthy = container_count >= 8
|
|
return "Containers", healthy, str(container_count) + " containers"
|
|
|
|
def check_model_probes():
|
|
"""Step 6: Model probe - all 4 aliases"""
|
|
# Get monitor key
|
|
monitor_key = run_command("ssh -o BatchMode=yes root@192.168.68.116 \"grep LITELLM_MONITOR_KEY /etc/litellm-monitor.env | cut -d= -f2\"")[1]
|
|
|
|
results = []
|
|
|
|
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
|
# Single-host aliases: 30s timeout
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
method="POST",
|
|
bearer_token=monitor_key,
|
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
timeout=30)
|
|
|
|
results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
|
|
|
# Pool alias (syslog-auto): 60s timeout, retry once on 000
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
method="POST",
|
|
bearer_token=monitor_key,
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
timeout=60)
|
|
|
|
if code == 000:
|
|
# Retry once with same timeout
|
|
time.sleep(1)
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
method="POST",
|
|
bearer_token=monitor_key,
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
timeout=60)
|
|
|
|
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
|
|
|
|
return results
|
|
|
|
def check_admin_key_list():
|
|
"""Step 8: Admin API key list - use two-step approach"""
|
|
# Step 1: Get master key
|
|
mk_cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"docker exec harness-litellm printenv LITELLM_MASTER_KEY\""
|
|
mk_rc, mk_stdout, mk_stderr = run_command(mk_cmd)
|
|
|
|
if mk_rc != 0:
|
|
return "Admin Key List", False, "credential-missing (ssh failed: " + mk_stderr + ")"
|
|
|
|
mk = mk_stdout
|
|
if not mk or "NO-CURL" in mk:
|
|
return "Admin Key List", False, "credential-missing (empty or NO-CURL)"
|
|
|
|
# Step 2: Call using the key - use double quotes inside SSH command
|
|
cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"curl -s -H \\\"Authorization: Bearer " + mk + "\\\" http://127.0.0.1:4000/key/list\""
|
|
rc, stdout, stderr = run_command(cmd)
|
|
|
|
if rc != 0:
|
|
return "Admin Key List", False, "admin-call-failed (exit=" + str(rc) + ", stderr=" + stderr + ")"
|
|
|
|
# Try to parse the response
|
|
try:
|
|
data = json.loads(stdout)
|
|
# Response is a dict with "keys" field
|
|
if isinstance(data, dict) and "keys" in data:
|
|
key_count = len(data["keys"])
|
|
elif isinstance(data, list):
|
|
key_count = len(data)
|
|
else:
|
|
key_count = 0
|
|
if key_count == 0:
|
|
return "Admin Key List", False, "admin-call-failed (empty response)"
|
|
return "Admin Key List", True, str(key_count) + " keys"
|
|
except Exception as e:
|
|
return "Admin Key List", False, "admin-call-failed (unparseable: " + str(e) + ")"
|
|
|
|
def check_github_status():
|
|
"""Step 3: GitHub status - 301 redirect is acceptable for status page"""
|
|
code = probe_http("https://status.github.com/api/status.json", timeout=15)
|
|
# GitHub status API returns 301 redirect, which is expected behavior
|
|
return "GitHub Status", code == 301, str(code)
|
|
|
|
def check_prometheus():
|
|
"""Step 4: Prometheus health"""
|
|
code = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy")
|
|
return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)"
|
|
|
|
def check_grafana():
|
|
"""Step 9: Grafana health"""
|
|
code = probe_http("http://" + BACKEND_HOST + ":3001/api/health")
|
|
return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)"
|
|
|
|
def check_docker_stats():
|
|
"""Step 10: Docker Stats health - fetch from CT 116 host"""
|
|
# Docker stats is on localhost from CT 116
|
|
cmd = "ssh -o BatchMode=yes root@192.168.68.116 'curl -s http://127.0.0.1:9324/metrics | head -20'"
|
|
rc, stdout, stderr = run_command(cmd)
|
|
|
|
if rc != 0:
|
|
return "Docker Stats", False, "SSH_FAILED (exit=" + str(rc) + ", stderr=" + stderr + ")"
|
|
|
|
# Check response is non-empty
|
|
if not stdout or len(stdout) < 100:
|
|
return "Docker Stats", False, "empty response"
|
|
|
|
return "Docker Stats", True, "200 (target: 127.0.0.1:9324/metrics from CT 116)"
|
|
|
|
def main():
|
|
print("🏥 LiteLLM Health Check v1.0.0")
|
|
print("📍 Backend edge: http://" + BACKEND_HOST)
|
|
print("")
|
|
|
|
all_pass = True
|
|
|
|
# Run all checks
|
|
checks = [
|
|
check_liveliness(),
|
|
check_containers(),
|
|
check_prometheus(),
|
|
check_grafana(),
|
|
]
|
|
|
|
for result in checks:
|
|
name, passed, detail = result
|
|
status = "✅" if passed else "❌"
|
|
print(" " + status + " " + name + ": " + detail)
|
|
if not passed:
|
|
all_pass = False
|
|
|
|
# Model probes
|
|
model_results = check_model_probes()
|
|
for name, passed, detail in model_results:
|
|
status = "✅" if passed else "❌"
|
|
print(" " + status + " " + name + ": " + detail)
|
|
if not passed:
|
|
all_pass = False
|
|
|
|
# Admin key list
|
|
admin_result = check_admin_key_list()
|
|
status = "✅" if admin_result[1] else "❌"
|
|
print(" " + status + " Admin Key List: " + admin_result[2])
|
|
if not admin_result[1]:
|
|
all_pass = False
|
|
|
|
# GitHub status
|
|
github_result = check_github_status()
|
|
status = "✅" if github_result[1] else "❌"
|
|
print(" " + status + " " + github_result[0] + ": " + github_result[2])
|
|
if not github_result[1]:
|
|
all_pass = False
|
|
|
|
# Docker stats
|
|
docker_stats_result = check_docker_stats()
|
|
status = "✅" if docker_stats_result[1] else "❌"
|
|
print(" " + status + " " + docker_stats_result[0] + ": " + docker_stats_result[2])
|
|
if not docker_stats_result[1]:
|
|
all_pass = False
|
|
|
|
print("")
|
|
if all_pass:
|
|
print("✅ All checks passed")
|
|
return 0
|
|
else:
|
|
print("❌ Some checks failed")
|
|
return 1
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|