Merge pull request 'fix(litellm-health): retry single-host model probes once at a longer timeout' (#121) from fix/litellm-health-retry-timeout-20260919 into master
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 9s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 0s
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 9s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 0s
This commit was merged in pull request #121.
This commit is contained in:
@@ -116,17 +116,31 @@ def check_model_probes():
|
||||
results = []
|
||||
|
||||
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
||||
# Single-host aliases: 30s timeout each
|
||||
# gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start
|
||||
# Single-host aliases: 30s initial timeout, retry once at 45s on failure
|
||||
# gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
|
||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||
method="POST",
|
||||
bearer_token=monitor_key,
|
||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||
timeout=30)
|
||||
|
||||
first_kind = None # Track first attempt's failure kind
|
||||
if code == 000 and failure_kind:
|
||||
# Report probe failure with kind, do not assert a service verdict
|
||||
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)"))
|
||||
# Retry once with longer timeout (45s) before declaring failure
|
||||
first_kind = failure_kind
|
||||
time.sleep(1)
|
||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||
method="POST",
|
||||
bearer_token=monitor_key,
|
||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||
timeout=45)
|
||||
|
||||
if code == 000 and failure_kind:
|
||||
# Both attempts failed - report both kinds
|
||||
if first_kind:
|
||||
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
|
||||
else:
|
||||
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
|
||||
elif code == 200:
|
||||
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
||||
elif code in (401, 403):
|
||||
|
||||
Reference in New Issue
Block a user