From 76653381ece6f5d8e688ee41fea5765e8a12a9e4 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 10:43:54 +0000 Subject: [PATCH 1/2] fix(litellm): Add retry with longer timeout for single-host model probes Single-host models (gpu-dense, gpu-vision, strix-moe) now retry once at 45s on initial 30s timeout failure before declaring probe-failed. This prevents a single transient timeout (cold prefill ~13s or concurrent generation hold) from failing the entire health digest. Evidence: 2026-09-19 ~06:55Z digest failed gpu-dense at 30s; 06:56Z direct probe 200 in 1.04s. The failed-probe-fails-the-run property is preserved: if both attempts fail, the script still exits non-zero with the target and duration named. Closes: daily-health-digest false negative on single transient timeout --- scripts/litellm-health-check.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 057da20..25ec8f2 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -116,8 +116,8 @@ def check_model_probes(): results = [] for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s timeout each - # gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start + # Single-host aliases: 30s initial timeout, retry once at 45s on failure + # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, @@ -125,8 +125,17 @@ def check_model_probes(): timeout=30) if code == 000 and failure_kind: - # Report probe failure with kind, do not assert a service verdict - results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) + # Retry once with longer timeout (45s) before declaring failure + time.sleep(1) + code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', + timeout=45) + + if code == 000 and failure_kind: + # Both attempts failed - report with duration + results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (timeout after retry, 45s)")) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403): -- 2.54.0 From aee2de25ac16ad024c7945025a9ca86cebd9fbaa Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 10:53:44 +0000 Subject: [PATCH 2/2] fix(litellm): Report both attempts' failure kinds in probe-failed The failure line now preserves both attempts' failure kinds instead of hardcoding 'timeout after retry, 45s'. If both attempts fail, the report shows: 'probe-failed: then (2 attempts)'. This fixes the self-contradictory output when the first attempt timed out but the retry failed with connection refused, and prevents the duration from appearing twice when both attempts were timeouts. Example outputs: - timeout then timeout: 'probe-failed: gpu-dense timeout after 30s then timeout after 45s (2 attempts)' - timeout then refused: 'probe-failed: gpu-dense timeout after 30s then connection refused (2 attempts)' - refused then refused: 'probe-failed: gpu-dense connection refused then connection refused (2 attempts)' --- scripts/litellm-health-check.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 25ec8f2..50ec076 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -124,8 +124,10 @@ def check_model_probes(): data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', timeout=30) + first_kind = None # Track first attempt's failure kind if code == 000 and failure_kind: # Retry once with longer timeout (45s) before declaring failure + first_kind = failure_kind time.sleep(1) code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", @@ -134,8 +136,11 @@ def check_model_probes(): timeout=45) if code == 000 and failure_kind: - # Both attempts failed - report with duration - results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (timeout after retry, 45s)")) + # Both attempts failed - report both kinds + if first_kind: + results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)")) + else: + results.append((model, False, "probe-failed: " + model + " " + failure_kind)) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403): -- 2.54.0