From 7bf9f78fc68e8b099c889bedf41a54a20bcc4cd8 Mon Sep 17 00:00:00 2001 From: root Date: Tue, 15 Sep 2026 04:22:46 +0000 Subject: [PATCH] fix: gpu-dense probe timeout handling - report probe-failed with kind, not service verdict - probe_http now returns (code, failure_kind) tuple - Model probes report 'probe-failed: (Ns timeout)' on 000 - Do not assert a service verdict from a failed probe - 30s timeout for single-host aliases (RTX 3090 needs long warmup/prefill) - 60s timeout for syslog-auto pool alias with retry on 000 Signed-off-by: Abiba --- scripts/litellm-health-check.py | 91 ++++++++++++++++++++++----------- 1 file changed, 61 insertions(+), 30 deletions(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 2d2f8e5..a5f2911 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -35,7 +35,12 @@ def run_command(cmd, timeout=15): return 1, "", str(e) def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False): - """Probe HTTP endpoint and return status code""" + """Probe HTTP endpoint and return (status_code, failure_kind) + + Returns: + (code, None) if successful or HTTP response received + (000, kind) if connection failed, where kind is 'timeout', 'refused', 'dns', etc. + """ cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout) if method == "POST": cmd += " -X POST" @@ -47,15 +52,30 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll cmd += " -L" cmd += " '" + url + "'" - rc, stdout, stderr = run_command(cmd, timeout) - if rc != 0 and "TIMEOUT" not in stderr: - return 000 # Connection failed - - return int(stdout) if stdout.isdigit() else 000 + try: + rc, stdout, stderr = run_command(cmd, timeout) + if rc != 0: + # Determine failure kind from curl exit code + # curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty + if rc == 28: + return (000, "timeout after " + str(timeout) + "s") + elif rc == 7: + return (000, "connection refused") + elif rc == 6: + return (000, "dns failure") + elif rc == 35: + return (000, "ssl error") + elif rc == 52: + return (000, "empty response") + else: + return (000, "curl exit " + str(rc)) + return (int(stdout), None) if stdout.isdigit() else (000, "unparseable response") + except subprocess.TimeoutExpired: + return (000, "timeout after " + str(timeout) + "s") def check_liveliness(): """Step 1: Liveliness probe""" - code = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness") + code, _ = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness") return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)" def check_containers(): @@ -79,32 +99,43 @@ def check_model_probes(): results = [] for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s timeout - code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", - method="POST", - bearer_token=monitor_key, - data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', - timeout=30) + # Single-host aliases: 30s timeout each + # gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=30) - results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) + if code == 000 and failure_kind: + # Report probe failure with kind, do not assert a service verdict + results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) + elif code == 200: + results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) + else: + results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) # Pool alias (syslog-auto): 60s timeout, retry once on 000 - code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", - method="POST", - bearer_token=monitor_key, - data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', - timeout=60) + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=60) - if code == 000: + if code == 000 and failure_kind: # Retry once with same timeout time.sleep(1) - code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", - method="POST", - bearer_token=monitor_key, - data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', - timeout=60) - - results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)")) + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=60) + if code == 000 and failure_kind: + results.append(("syslog-auto", False, "probe-failed: syslog-auto " + failure_kind + " (60s timeout, retry)")) + else: + results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)")) + else: + results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)")) return results @@ -146,18 +177,18 @@ def check_admin_key_list(): def check_github_status(): """Step 3: GitHub status - 301 redirect is acceptable for status page""" - code = probe_http("https://status.github.com/api/status.json", timeout=15) + code, _ = probe_http("https://status.github.com/api/status.json", timeout=15) # GitHub status API returns 301 redirect, which is expected behavior return "GitHub Status", code == 301, str(code) def check_prometheus(): """Step 4: Prometheus health""" - code = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy") + code, _ = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy") return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)" def check_grafana(): """Step 9: Grafana health""" - code = probe_http("http://" + BACKEND_HOST + ":3001/api/health") + code, _ = probe_http("http://" + BACKEND_HOST + ":3001/api/health") return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)" def check_docker_stats():