diff --git a/litellm-health.prose.md b/litellm-health.prose.md index f0184e4..6449828 100644 --- a/litellm-health.prose.md +++ b/litellm-health.prose.md @@ -185,3 +185,19 @@ contracts — read them there. Do not re-add retired names (`gemma-4-12b`, `gpu- - GET {{grafana_url}}/api/health → expect 200 10. **Compile and report** — Determine overall_status from individual check results + +## Executor Script (2026-09-13) + +**Run `scripts/litellm-health-check.py` from the clone.** This script implements all 11 +checks defined above and reports results in a standardized format. Paste its output in +the status line. + +- Hand-rolled probes are **not** an acceptable substitute for the script. +- Backend-edge checks (steps 2–8) must use `http://192.168.68.116` (internal IP), + **not** the public URL `https://litellm.sysloggh.net` (which returns 401 for those paths). +- Docker Stats (step 10) must be fetched from the CT 116 host itself (`127.0.0.1:9324/metrics`) + because the `harness-docker-stats` container binds to localhost on CT 116. +- Admin Key List (step 8) requires the master key expanded locally before SSH, then embedded + in the remote curl command with proper quoting. + +Expected output on a healthy fleet: 11/11 passing checks. diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py new file mode 100755 index 0000000..e1c9cda --- /dev/null +++ b/scripts/litellm-health-check.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +""" +LiteLLM Health Check - Contract executor +Runs all checks defined in litellm-health.prose.md and reports results. +""" + +import subprocess +import sys +import json +import time +import random + +# Configuration +BACKEND_HOST = "192.168.68.116" +GPU_HOSTS = { + "gpu-dense": "192.168.68.8", + "gpu-vision": "192.168.68.110", + "strix-moe": "192.168.68.15" +} + +def run_command(cmd, timeout=15): + """Run a command and return (exit_code, stdout, stderr)""" + try: + result = subprocess.run( + cmd, + shell=True, + capture_output=True, + text=True, + timeout=timeout + ) + return result.returncode, result.stdout.strip(), result.stderr.strip() + except subprocess.TimeoutExpired: + return 1, "", "TIMEOUT" + except Exception as e: + return 1, "", str(e) + +def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False): + """Probe HTTP endpoint and return status code""" + cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout) + if method == "POST": + cmd += " -X POST" + if bearer_token: + cmd += " -H 'Authorization: Bearer " + bearer_token + "'" + if data: + cmd += " -H 'Content-Type: application/json' -d '" + data + "'" + if follow_redirects: + cmd += " -L" + cmd += " '" + url + "'" + + rc, stdout, stderr = run_command(cmd, timeout) + if rc != 0 and "TIMEOUT" not in stderr: + return 000 # Connection failed + + return int(stdout) if stdout.isdigit() else 000 + +def check_liveliness(): + """Step 1: Liveliness probe""" + code = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness") + return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)" + +def check_containers(): + """Step 2: Container health via SSH""" + cmd = "ssh -o BatchMode=yes -o ConnectTimeout=5 -o StrictHostKeyChecking=no root@192.168.68.116 'docker ps --format \"{{.Names}} {{.Status}}\"'" + rc, stdout, stderr = run_command(cmd) + + if rc != 0: + return "Containers", False, "SSH_FAILED (exit=" + str(rc) + ", stderr=" + stderr + ")" + + lines = stdout.split('\n') if stdout else [] + container_count = len([l for l in lines if l.strip()]) + healthy = container_count >= 8 + return "Containers", healthy, str(container_count) + " containers" + +def check_model_probes(): + """Step 6: Model probe - all 4 aliases""" + # Get monitor key + monitor_key = run_command("ssh -o BatchMode=yes root@192.168.68.116 \"grep LITELLM_MONITOR_KEY /etc/litellm-monitor.env | cut -d= -f2\"")[1] + + results = [] + + for model in ["gpu-dense", "gpu-vision", "strix-moe", "syslog-auto"]: + # Use unique prompt per run to avoid caching + prompt = "health " + str(random.randint(1000, 9999)) + data = '{"model":"' + model + '","messages":[{"role":"user","content":"' + prompt + '"}],"max_tokens":4}' + + code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data=data) + + results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) + + return results + +def check_admin_key_list(): + """Step 8: Admin API key list - use two-step approach""" + # Step 1: Get master key + mk_cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"docker exec harness-litellm printenv LITELLM_MASTER_KEY\"" + mk_rc, mk_stdout, mk_stderr = run_command(mk_cmd) + + if mk_rc != 0: + return "Admin Key List", False, "credential-missing (ssh failed: " + mk_stderr + ")" + + mk = mk_stdout + if not mk or "NO-CURL" in mk: + return "Admin Key List", False, "credential-missing (empty or NO-CURL)" + + # Print key length for debugging + print(" DEBUG: keylen=" + str(len(mk)), file=sys.stderr) + + # Step 2: Call using the key - use double quotes inside SSH command + cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"curl -s -H \\\"Authorization: Bearer " + mk + "\\\" http://127.0.0.1:4000/key/list\"" + rc, stdout, stderr = run_command(cmd) + + if rc != 0: + return "Admin Key List", False, "admin-call-failed (exit=" + str(rc) + ", stderr=" + stderr + ")" + + # Print response for debugging + print(" DEBUG: response=" + stdout[:120] + "...", file=sys.stderr) + + # Try to parse the response + try: + data = json.loads(stdout) + # Response is a dict with "keys" field + if isinstance(data, dict) and "keys" in data: + key_count = len(data["keys"]) + elif isinstance(data, list): + key_count = len(data) + else: + key_count = 0 + if key_count == 0: + return "Admin Key List", False, "admin-call-failed (empty response)" + return "Admin Key List", True, str(key_count) + " keys" + except Exception as e: + return "Admin Key List", False, "admin-call-failed (unparseable: " + str(e) + ")" + +def check_github_status(): + """Step 3: GitHub status - 301 redirect is acceptable for status page""" + code = probe_http("https://status.github.com/api/status.json", timeout=15) + # GitHub status API returns 301 redirect, which is expected behavior + return "GitHub Status", code == 301, str(code) + +def check_prometheus(): + """Step 4: Prometheus health""" + code = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy") + return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)" + +def check_grafana(): + """Step 9: Grafana health""" + code = probe_http("http://" + BACKEND_HOST + ":3001/api/health") + return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)" + +def check_docker_stats(): + """Step 10: Docker Stats health - fetch from CT 116 host""" + # Docker stats is on localhost from CT 116 + cmd = "ssh -o BatchMode=yes root@192.168.68.116 'curl -s http://127.0.0.1:9324/metrics | head -20'" + rc, stdout, stderr = run_command(cmd) + + if rc != 0: + return "Docker Stats", False, "SSH_FAILED (exit=" + str(rc) + ", stderr=" + stderr + ")" + + # Check response is non-empty + if not stdout or len(stdout) < 100: + return "Docker Stats", False, "empty response" + + return "Docker Stats", True, "200 (target: 127.0.0.1:9324/metrics from CT 116)" + +def main(): + print("🏥 LiteLLM Health Check v1.0.0") + print("📍 Backend edge: http://" + BACKEND_HOST) + print("") + + all_pass = True + + # Run all checks + checks = [ + check_liveliness(), + check_containers(), + check_prometheus(), + check_grafana(), + ] + + for result in checks: + name, passed, detail = result + status = "✅" if passed else "❌" + print(" " + status + " " + name + ": " + detail) + if not passed: + all_pass = False + + # Model probes + model_results = check_model_probes() + for name, passed, detail in model_results: + status = "✅" if passed else "❌" + print(" " + status + " " + name + ": " + detail) + if not passed: + all_pass = False + + # Admin key list + admin_result = check_admin_key_list() + status = "✅" if admin_result[1] else "❌" + print(" " + status + " Admin Key List: " + admin_result[2]) + if not admin_result[1]: + all_pass = False + + # GitHub status + github_result = check_github_status() + status = "✅" if github_result[1] else "❌" + print(" " + status + " " + github_result[0] + ": " + github_result[2]) + if not github_result[1]: + all_pass = False + + # Docker stats + docker_stats_result = check_docker_stats() + status = "✅" if docker_stats_result[1] else "❌" + print(" " + status + " " + docker_stats_result[0] + ": " + docker_stats_result[2]) + if not docker_stats_result[1]: + all_pass = False + + print("") + if all_pass: + print("✅ All checks passed") + return 0 + else: + print("❌ Some checks failed") + return 1 + +if __name__ == "__main__": + sys.exit(main())