diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 50ec076..c94617c 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -55,9 +55,12 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll try: rc, stdout, stderr = run_command(cmd, timeout) if rc != 0: - # Determine failure kind from curl exit code + # Check if this is a timeout from run_command (rc=1, stderr="TIMEOUT") + if rc == 1 and stderr == "TIMEOUT": + return (000, "timeout after " + str(timeout) + "s") + # Otherwise, determine failure kind from curl exit code # curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty - if rc == 28: + elif rc == 28: return (000, "timeout after " + str(timeout) + "s") elif rc == 7: return (000, "connection refused") @@ -73,6 +76,20 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll except subprocess.TimeoutExpired: return (000, "timeout after " + str(timeout) + "s") +def check_host_health(host_ip): + """Check if the GPU host's llama-chat-api health endpoint is reachable + + Returns: (healthy: bool, detail: str) + """ + code, _ = probe_http("http://" + host_ip + ":8080/health", timeout=10) + if code == 200: + return True, "host healthy (200)" + elif code == 000: + return False, "host unreachable (timeout or refused)" + else: + return False, "host unhealthy (HTTP " + str(code) + ")" + + def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30): """Get response body for 401/403 credential faults (truncated to 200 chars)""" cmd = "curl -s -m " + str(timeout) @@ -115,43 +132,53 @@ def check_model_probes(): results = [] + # Host health mapping: model -> host IP + model_hosts = { + "gpu-dense": "192.168.68.8", # RTX 3090 + "gpu-vision": "192.168.68.110", # RTX 5070 + "strix-moe": "192.168.68.15" # Strix Halo + } + for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s initial timeout, retry once at 45s on failure - # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold + host_ip = model_hosts[model] + # Single-host aliases: 30s initial timeout, retry once at 90s on failure + # Worst-case prefill ~76s, so 90s retry ensures we cover it code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', timeout=30) - first_kind = None # Track first attempt's failure kind + first_kind = None if code == 000 and failure_kind: - # Retry once with longer timeout (45s) before declaring failure first_kind = failure_kind time.sleep(1) code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', - timeout=45) + timeout=90) if code == 000 and failure_kind: - # Both attempts failed - report both kinds - if first_kind: - results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)")) + # Both attempts failed - check host health to distinguish busy from down + host_healthy, host_detail = check_host_health(host_ip) + if host_healthy: + results.append((model, False, "busy (completion timed out after retry; " + host_detail + ")")) else: - results.append((model, False, "probe-failed: " + model + " " + failure_kind)) + # Host unreachable - report both kinds + if first_kind: + results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)")) + else: + results.append((model, False, "probe-failed: " + model + " " + failure_kind)) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403): - # Credential fault - capture body and key alias body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}', timeout=10) - # Resolve key alias - alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116 + alias = "monitor-20260813" results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias)) else: results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) @@ -261,6 +288,7 @@ def main(): print("") all_pass = True + degraded = [] # Track degraded (busy) checks # Run all checks checks = [ @@ -280,9 +308,15 @@ def main(): # Model probes model_results = check_model_probes() for name, passed, detail in model_results: - status = "✅" if passed else "❌" + # Check if this is a busy (degraded) verdict + if not passed and detail.startswith("busy "): + status = "⚠️" + degraded.append(name) + else: + status = "✅" if passed else "❌" print(" " + status + " " + name + ": " + detail) - if not passed: + # Only set all_pass=False for real failures (not busy) + if not passed and not detail.startswith("busy "): all_pass = False # Admin key list @@ -308,10 +342,16 @@ def main(): print("") if all_pass: - print("✅ All checks passed") + if degraded: + print("✅ All checks passed (" + str(len(degraded)) + " degraded: " + ", ".join(degraded) + ")") + else: + print("✅ All checks passed") return 0 else: - print("❌ Some checks failed") + if degraded: + print("❌ Some checks failed (" + str(len(degraded)) + " degraded: " + ", ".join(degraded) + ")") + else: + print("❌ Some checks failed") return 1 if __name__ == "__main__":