fix(litellm-health): retry single-host model probes once at a longer timeout #121

Merged
abiba-bot merged 2 commits from fix/litellm-health-retry-timeout-20260919 into master 2026-09-19 10:57:40 +00:00
+18 -4
View File
@@ -116,17 +116,31 @@ def check_model_probes():
results = [] results = []
for model in ["gpu-dense", "gpu-vision", "strix-moe"]: for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
# Single-host aliases: 30s timeout each # Single-host aliases: 30s initial timeout, retry once at 45s on failure
# gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST", method="POST",
bearer_token=monitor_key, bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=30) timeout=30)
first_kind = None # Track first attempt's failure kind
if code == 000 and failure_kind: if code == 000 and failure_kind:
# Report probe failure with kind, do not assert a service verdict # Retry once with longer timeout (45s) before declaring failure
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) first_kind = failure_kind
time.sleep(1)
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=45)
if code == 000 and failure_kind:
# Both attempts failed - report both kinds
if first_kind:
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
else:
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
elif code == 200: elif code == 200:
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
elif code in (401, 403): elif code in (401, 403):