fix(litellm-health): three-state model verdicts - busy is not a failure #123
@@ -55,9 +55,12 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
|
|||||||
try:
|
try:
|
||||||
rc, stdout, stderr = run_command(cmd, timeout)
|
rc, stdout, stderr = run_command(cmd, timeout)
|
||||||
if rc != 0:
|
if rc != 0:
|
||||||
# Determine failure kind from curl exit code
|
# Check if this is a timeout from run_command (rc=1, stderr="TIMEOUT")
|
||||||
|
if rc == 1 and stderr == "TIMEOUT":
|
||||||
|
return (000, "timeout after " + str(timeout) + "s")
|
||||||
|
# Otherwise, determine failure kind from curl exit code
|
||||||
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
|
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
|
||||||
if rc == 28:
|
elif rc == 28:
|
||||||
return (000, "timeout after " + str(timeout) + "s")
|
return (000, "timeout after " + str(timeout) + "s")
|
||||||
elif rc == 7:
|
elif rc == 7:
|
||||||
return (000, "connection refused")
|
return (000, "connection refused")
|
||||||
@@ -73,6 +76,20 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
|
|||||||
except subprocess.TimeoutExpired:
|
except subprocess.TimeoutExpired:
|
||||||
return (000, "timeout after " + str(timeout) + "s")
|
return (000, "timeout after " + str(timeout) + "s")
|
||||||
|
|
||||||
|
def check_host_health(host_ip):
|
||||||
|
"""Check if the GPU host's llama-chat-api health endpoint is reachable
|
||||||
|
|
||||||
|
Returns: (healthy: bool, detail: str)
|
||||||
|
"""
|
||||||
|
code, _ = probe_http("http://" + host_ip + ":8080/health", timeout=10)
|
||||||
|
if code == 200:
|
||||||
|
return True, "host healthy (200)"
|
||||||
|
elif code == 000:
|
||||||
|
return False, "host unreachable (timeout or refused)"
|
||||||
|
else:
|
||||||
|
return False, "host unhealthy (HTTP " + str(code) + ")"
|
||||||
|
|
||||||
|
|
||||||
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
|
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
|
||||||
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
|
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
|
||||||
cmd = "curl -s -m " + str(timeout)
|
cmd = "curl -s -m " + str(timeout)
|
||||||
@@ -115,43 +132,53 @@ def check_model_probes():
|
|||||||
|
|
||||||
results = []
|
results = []
|
||||||
|
|
||||||
|
# Host health mapping: model -> host IP
|
||||||
|
model_hosts = {
|
||||||
|
"gpu-dense": "192.168.68.8", # RTX 3090
|
||||||
|
"gpu-vision": "192.168.68.110", # RTX 5070
|
||||||
|
"strix-moe": "192.168.68.15" # Strix Halo
|
||||||
|
}
|
||||||
|
|
||||||
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
||||||
# Single-host aliases: 30s initial timeout, retry once at 45s on failure
|
host_ip = model_hosts[model]
|
||||||
# gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
|
# Single-host aliases: 30s initial timeout, retry once at 90s on failure
|
||||||
|
# Worst-case prefill ~76s, so 90s retry ensures we cover it
|
||||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||||
timeout=30)
|
timeout=30)
|
||||||
|
|
||||||
first_kind = None # Track first attempt's failure kind
|
first_kind = None
|
||||||
if code == 000 and failure_kind:
|
if code == 000 and failure_kind:
|
||||||
# Retry once with longer timeout (45s) before declaring failure
|
|
||||||
first_kind = failure_kind
|
first_kind = failure_kind
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||||
timeout=45)
|
timeout=90)
|
||||||
|
|
||||||
if code == 000 and failure_kind:
|
if code == 000 and failure_kind:
|
||||||
# Both attempts failed - report both kinds
|
# Both attempts failed - check host health to distinguish busy from down
|
||||||
if first_kind:
|
host_healthy, host_detail = check_host_health(host_ip)
|
||||||
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
|
if host_healthy:
|
||||||
|
results.append((model, False, "busy (completion timed out after retry; host healthy " + host_detail + ")"))
|
||||||
else:
|
else:
|
||||||
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
|
# Host unreachable - report both kinds
|
||||||
|
if first_kind:
|
||||||
|
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
|
||||||
|
else:
|
||||||
|
results.append((model, False, "probe-failed: " + model + " " + failure_kind))
|
||||||
elif code == 200:
|
elif code == 200:
|
||||||
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
||||||
elif code in (401, 403):
|
elif code in (401, 403):
|
||||||
# Credential fault - capture body and key alias
|
|
||||||
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
|
||||||
timeout=10)
|
timeout=10)
|
||||||
# Resolve key alias
|
alias = "monitor-20260813"
|
||||||
alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116
|
|
||||||
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
|
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
|
||||||
else:
|
else:
|
||||||
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
||||||
|
|||||||
Reference in New Issue
Block a user