Files
prose-contracts/scripts/litellm-health-check.py
T
root 7f62f19c24
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 7s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 0s
fix: litellm-key-count-self-describing-20260916
The key-count line in litellm-health-check now reports self-describing
output: '18 total (10 on page 1)' instead of bare '18' or '4'. Uses
total_count from the paginated API response and names what was
counted. Previous bare numbers could not reconcile changes between
runs; now a reader sees both the total and the page 1 sample.
2026-09-16 15:20:50 +00:00

305 lines
13 KiB
Python
Executable File

#!/usr/bin/env python3
"""
LiteLLM Health Check - Contract executor
Runs all checks defined in litellm-health.prose.md and reports results.
"""
import subprocess
import sys
import json
import time
import random
# Configuration
BACKEND_HOST = "192.168.68.116"
GPU_HOSTS = {
"gpu-dense": "192.168.68.8",
"gpu-vision": "192.168.68.110",
"strix-moe": "192.168.68.15"
}
def run_command(cmd, timeout=15):
"""Run a command and return (exit_code, stdout, stderr)"""
try:
result = subprocess.run(
cmd,
shell=True,
capture_output=True,
text=True,
timeout=timeout
)
return result.returncode, result.stdout.strip(), result.stderr.strip()
except subprocess.TimeoutExpired:
return 1, "", "TIMEOUT"
except Exception as e:
return 1, "", str(e)
def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False):
"""Probe HTTP endpoint and return (status_code, failure_kind)
Returns:
(code, None) if successful or HTTP response received
(000, kind) if connection failed, where kind is 'timeout', 'refused', 'dns', etc.
"""
cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout)
if method == "POST":
cmd += " -X POST"
if bearer_token:
cmd += " -H 'Authorization: Bearer " + bearer_token + "'"
if data:
cmd += " -H 'Content-Type: application/json' -d '" + data + "'"
if follow_redirects:
cmd += " -L"
cmd += " '" + url + "'"
try:
rc, stdout, stderr = run_command(cmd, timeout)
if rc != 0:
# Determine failure kind from curl exit code
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
if rc == 28:
return (000, "timeout after " + str(timeout) + "s")
elif rc == 7:
return (000, "connection refused")
elif rc == 6:
return (000, "dns failure")
elif rc == 35:
return (000, "ssl error")
elif rc == 52:
return (000, "empty response")
else:
return (000, "curl exit " + str(rc))
return (int(stdout), None) if stdout.isdigit() else (000, "unparseable response")
except subprocess.TimeoutExpired:
return (000, "timeout after " + str(timeout) + "s")
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
cmd = "curl -s -m " + str(timeout)
if method == "POST":
cmd += " -X POST"
if bearer_token:
cmd += " -H 'Authorization: Bearer " + bearer_token + "'"
if data:
cmd += " -H 'Content-Type: application/json' -d '" + data + "'"
cmd += " '" + url + "'"
rc, stdout, stderr = run_command(cmd, timeout)
# Return first 200 chars, single line
body = stdout.replace('\n', ' ').replace('\t', ' ')[:200] if stdout else ""
return body
def check_liveliness():
"""Step 1: Liveliness probe"""
code, _ = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness")
return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)"
def check_containers():
"""Step 2: Container health via SSH"""
cmd = "ssh -o BatchMode=yes -o ConnectTimeout=5 -o StrictHostKeyChecking=no root@192.168.68.116 'docker ps --format \"{{.Names}} {{.Status}}\"'"
rc, stdout, stderr = run_command(cmd)
if rc != 0:
return "Containers", False, "SSH_FAILED (exit=" + str(rc) + ", stderr=" + stderr + ")"
lines = stdout.split('\n') if stdout else []
container_count = len([l for l in lines if l.strip()])
healthy = container_count >= 8
return "Containers", healthy, str(container_count) + " containers"
def check_model_probes():
"""Step 6: Model probe - all 4 aliases"""
# Get monitor key
monitor_key = run_command("ssh -o BatchMode=yes root@192.168.68.116 \"grep LITELLM_MONITOR_KEY /etc/litellm-monitor.env | cut -d= -f2\"")[1]
results = []
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
# Single-host aliases: 30s timeout each
# gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=30)
if code == 000 and failure_kind:
# Report probe failure with kind, do not assert a service verdict
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)"))
elif code == 200:
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
elif code in (401, 403):
# Credential fault - capture body and key alias
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
timeout=10)
# Resolve key alias
alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
else:
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
# Pool alias (syslog-auto): 60s timeout, retry once on 000
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=60)
if code == 000 and failure_kind:
# Retry once with same timeout
time.sleep(1)
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=60)
if code == 000 and failure_kind:
results.append(("syslog-auto", False, "probe-failed: syslog-auto " + failure_kind + " (60s timeout, retry)"))
elif code in (401, 403):
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
timeout=10)
alias = "monitor-20260813"
results.append(("syslog-auto", False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
else:
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
else:
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
return results
def check_admin_key_list():
"""Step 8: Admin API key list - use two-step approach"""
# Step 1: Get master key
mk_cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"docker exec harness-litellm printenv LITELLM_MASTER_KEY\""
mk_rc, mk_stdout, mk_stderr = run_command(mk_cmd)
if mk_rc != 0:
return "Admin Key List", False, "credential-missing (ssh failed: " + mk_stderr + ")"
mk = mk_stdout
if not mk or "NO-CURL" in mk:
return "Admin Key List", False, "credential-missing (empty or NO-CURL)"
# Step 2: Call using the key - use double quotes inside SSH command
cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"curl -s -H \\\"Authorization: Bearer " + mk + "\\\" http://127.0.0.1:4000/key/list\""
rc, stdout, stderr = run_command(cmd)
if rc != 0:
return "Admin Key List", False, "admin-call-failed (exit=" + str(rc) + ", stderr=" + stderr + ")"
# Try to parse the response
try:
data = json.loads(stdout)
# Response is a dict with "keys" (paginated list) and "total_count" fields
if isinstance(data, dict) and "keys" in data:
key_count = data.get("total_count", len(data["keys"]))
elif isinstance(data, list):
key_count = len(data)
else:
key_count = 0
if key_count == 0:
return "Admin Key List", False, "admin-call-failed (empty response)"
return "Admin Key List", True, str(key_count) + " total (" + str(len(data.get("keys", []) if isinstance(data, dict) else data)) + " on page 1)" if isinstance(data, dict) else str(key_count) + " total"
except Exception as e:
return "Admin Key List", False, "admin-call-failed (unparseable: " + str(e) + ")"
def check_github_status():
"""Step 3: GitHub status - 301 redirect is acceptable for status page"""
code, _ = probe_http("https://status.github.com/api/status.json", timeout=15)
# GitHub status API returns 301 redirect, which is expected behavior
return "GitHub Status", code == 301, str(code)
def check_prometheus():
"""Step 4: Prometheus health"""
code, _ = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy")
return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)"
def check_grafana():
"""Step 9: Grafana health"""
code, _ = probe_http("http://" + BACKEND_HOST + ":3001/api/health")
return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)"
def check_docker_stats():
"""Step 10: Docker Stats health - fetch from CT 116 host"""
# Docker stats is on localhost from CT 116
cmd = "ssh -o BatchMode=yes root@192.168.68.116 'curl -s http://127.0.0.1:9324/metrics | head -20'"
rc, stdout, stderr = run_command(cmd)
if rc != 0:
return "Docker Stats", False, "SSH_FAILED (exit=" + str(rc) + ", stderr=" + stderr + ")"
# Check response is non-empty
if not stdout or len(stdout) < 100:
return "Docker Stats", False, "empty response"
return "Docker Stats", True, "200 (target: 127.0.0.1:9324/metrics from CT 116)"
def main():
print("🏥 LiteLLM Health Check v1.0.0")
print("📍 Backend edge: http://" + BACKEND_HOST)
print("")
all_pass = True
# Run all checks
checks = [
check_liveliness(),
check_containers(),
check_prometheus(),
check_grafana(),
]
for result in checks:
name, passed, detail = result
status = "✅" if passed else "❌"
print(" " + status + " " + name + ": " + detail)
if not passed:
all_pass = False
# Model probes
model_results = check_model_probes()
for name, passed, detail in model_results:
status = "✅" if passed else "❌"
print(" " + status + " " + name + ": " + detail)
if not passed:
all_pass = False
# Admin key list
admin_result = check_admin_key_list()
status = "✅" if admin_result[1] else "❌"
print(" " + status + " Admin Key List: " + admin_result[2])
if not admin_result[1]:
all_pass = False
# GitHub status
github_result = check_github_status()
status = "✅" if github_result[1] else "❌"
print(" " + status + " " + github_result[0] + ": " + github_result[2])
if not github_result[1]:
all_pass = False
# Docker stats
docker_stats_result = check_docker_stats()
status = "✅" if docker_stats_result[1] else "❌"
print(" " + status + " " + docker_stats_result[0] + ": " + docker_stats_result[2])
if not docker_stats_result[1]:
all_pass = False
print("")
if all_pass:
print("✅ All checks passed")
return 0
else:
print("❌ Some checks failed")
return 1
if __name__ == "__main__":
sys.exit(main())