From 76653381ece6f5d8e688ee41fea5765e8a12a9e4 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 10:43:54 +0000 Subject: [PATCH] fix(litellm): Add retry with longer timeout for single-host model probes Single-host models (gpu-dense, gpu-vision, strix-moe) now retry once at 45s on initial 30s timeout failure before declaring probe-failed. This prevents a single transient timeout (cold prefill ~13s or concurrent generation hold) from failing the entire health digest. Evidence: 2026-09-19 ~06:55Z digest failed gpu-dense at 30s; 06:56Z direct probe 200 in 1.04s. The failed-probe-fails-the-run property is preserved: if both attempts fail, the script still exits non-zero with the target and duration named. Closes: daily-health-digest false negative on single transient timeout --- scripts/litellm-health-check.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 057da20..25ec8f2 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -116,8 +116,8 @@ def check_model_probes(): results = [] for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s timeout each - # gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start + # Single-host aliases: 30s initial timeout, retry once at 45s on failure + # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, @@ -125,8 +125,17 @@ def check_model_probes(): timeout=30) if code == 000 and failure_kind: - # Report probe failure with kind, do not assert a service verdict - results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) + # Retry once with longer timeout (45s) before declaring failure + time.sleep(1) + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=45) + + if code == 000 and failure_kind: + # Both attempts failed - report with duration + results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (timeout after retry, 45s)")) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403):