diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 057da20..25ec8f2 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -116,8 +116,8 @@ def check_model_probes(): results = [] for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s timeout each - # gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start + # Single-host aliases: 30s initial timeout, retry once at 45s on failure + # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, @@ -125,8 +125,17 @@ def check_model_probes(): timeout=30) if code == 000 and failure_kind: - # Report probe failure with kind, do not assert a service verdict - results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) + # Retry once with longer timeout (45s) before declaring failure + time.sleep(1) + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=45) + + if code == 000 and failure_kind: + # Both attempts failed - report with duration + results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (timeout after retry, 45s)")) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403):