diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 057da20..50ec076 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -116,17 +116,31 @@ def check_model_probes(): results = [] for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s timeout each - # gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start + # Single-host aliases: 30s initial timeout, retry once at 45s on failure + # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', timeout=30) + first_kind = None # Track first attempt's failure kind if code == 000 and failure_kind: - # Report probe failure with kind, do not assert a service verdict - results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) + # Retry once with longer timeout (45s) before declaring failure + first_kind = failure_kind + time.sleep(1) + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=45) + + if code == 000 and failure_kind: + # Both attempts failed - report both kinds + if first_kind: + results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)")) + else: + results.append((model, False, "probe-failed: " + model + " " + failure_kind)) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403):