fix(litellm-health): retry single-host model probes once at a longer timeout #121
@@ -116,17 +116,31 @@ def check_model_probes():
|
|||||||
results = []
|
results = []
|
||||||
|
|
||||||
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
||||||
# Single-host aliases: 30s timeout each
|
# Single-host aliases: 30s initial timeout, retry once at 45s on failure
|
||||||
# gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start
|
# gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
|
||||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||||
timeout=30)
|
timeout=30)
|
||||||
|
|
||||||
|
first_kind = None # Track first attempt's failure kind
|
||||||
if code == 000 and failure_kind:
|
if code == 000 and failure_kind:
|
||||||
# Report probe failure with kind, do not assert a service verdict
|
# Retry once with longer timeout (45s) before declaring failure
|
||||||
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)"))
|
first_kind = failure_kind
|
||||||
|
time.sleep(1)
|
||||||
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
|
method="POST",
|
||||||
|
bearer_token=monitor_key,
|
||||||
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||||
|
timeout=45)
|
||||||
|
|
||||||
|
if code == 000 and failure_kind:
|
||||||
|
# Both attempts failed - report both kinds
|
||||||
|
if first_kind:
|
||||||
|
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
|
||||||
|
else:
|
||||||
|
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
|
||||||
elif code == 200:
|
elif code == 200:
|
||||||
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
||||||
elif code in (401, 403):
|
elif code in (401, 403):
|
||||||
|
|||||||
Reference in New Issue
Block a user