fix(litellm): Fix timeout kind reporting + add busy/degraded detection

1. TIMEOUT KIND FIX: run_command returns (1, '', 'TIMEOUT') when its own
   timeout fires. probe_http now checks for this before falling through to
   'curl exit <rc>', so a 30s timeout reports 'timeout after 30s' not
   'curl exit 1'.

2. BUSY/DEGRADED DETECTION: After both model probes fail, check the
   model's host health endpoint (e.g. 192.168.68.8:8080/health for
   gpu-dense). If the host answers 200, report 'busy (completion timed
   out after retry; host healthy 200)' — do NOT fail the run on that
   alone. If the host does not answer, that's a real FAIL.

3. RETRY TIMEOUT INCREASED: Single-host retry timeout raised from 45s to
   90s. Worst-case prefill on a single-slot .8 host is ~76s (observed
   83K-token prompt at 1078 tok/s), so 90s covers it.

New line shapes:
- Busy: 'gpu-dense: busy (completion timed out after retry; host healthy 200)'
- Real failure: 'probe-failed: gpu-dense timeout after 30s then timeout after 90s (2 attempts)'
This commit is contained in:
root
2026-09-19 11:32:15 +00:00
parent aa3da83af5
commit dba9cc53d8
+41 -14
View File
@@ -55,9 +55,12 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
try:
rc, stdout, stderr = run_command(cmd, timeout)
if rc != 0:
# Determine failure kind from curl exit code
# Check if this is a timeout from run_command (rc=1, stderr="TIMEOUT")
if rc == 1 and stderr == "TIMEOUT":
return (000, "timeout after " + str(timeout) + "s")
# Otherwise, determine failure kind from curl exit code
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
if rc == 28:
elif rc == 28:
return (000, "timeout after " + str(timeout) + "s")
elif rc == 7:
return (000, "connection refused")
@@ -73,6 +76,20 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
except subprocess.TimeoutExpired:
return (000, "timeout after " + str(timeout) + "s")
def check_host_health(host_ip):
"""Check if the GPU host's llama-chat-api health endpoint is reachable
Returns: (healthy: bool, detail: str)
"""
code, _ = probe_http("http://" + host_ip + ":8080/health", timeout=10)
if code == 200:
return True, "host healthy (200)"
elif code == 000:
return False, "host unreachable (timeout or refused)"
else:
return False, "host unhealthy (HTTP " + str(code) + ")"
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
cmd = "curl -s -m " + str(timeout)
@@ -115,43 +132,53 @@ def check_model_probes():
results = []
# Host health mapping: model -> host IP
model_hosts = {
"gpu-dense": "192.168.68.8", # RTX 3090
"gpu-vision": "192.168.68.110", # RTX 5070
"strix-moe": "192.168.68.15" # Strix Halo
}
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
# Single-host aliases: 30s initial timeout, retry once at 45s on failure
# gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
host_ip = model_hosts[model]
# Single-host aliases: 30s initial timeout, retry once at 90s on failure
# Worst-case prefill ~76s, so 90s retry ensures we cover it
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=30)
first_kind = None # Track first attempt's failure kind
first_kind = None
if code == 000 and failure_kind:
# Retry once with longer timeout (45s) before declaring failure
first_kind = failure_kind
time.sleep(1)
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=45)
timeout=90)
if code == 000 and failure_kind:
# Both attempts failed - report both kinds
if first_kind:
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
# Both attempts failed - check host health to distinguish busy from down
host_healthy, host_detail = check_host_health(host_ip)
if host_healthy:
results.append((model, False, "busy (completion timed out after retry; host healthy " + host_detail + ")"))
else:
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
# Host unreachable - report both kinds
if first_kind:
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
else:
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
elif code == 200:
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
elif code in (401, 403):
# Credential fault - capture body and key alias
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
timeout=10)
# Resolve key alias
alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116
alias = "monitor-20260813"
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
else:
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))