Merge pull request 'fix(litellm-health): three-state model verdicts - busy is not a failure' (#123) from fix/litellm-health-busy-vs-down-20260919 into master
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 7s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 1s
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 7s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 1s
This commit was merged in pull request #123.
This commit is contained in:
@@ -55,9 +55,12 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
|
|||||||
try:
|
try:
|
||||||
rc, stdout, stderr = run_command(cmd, timeout)
|
rc, stdout, stderr = run_command(cmd, timeout)
|
||||||
if rc != 0:
|
if rc != 0:
|
||||||
# Determine failure kind from curl exit code
|
# Check if this is a timeout from run_command (rc=1, stderr="TIMEOUT")
|
||||||
|
if rc == 1 and stderr == "TIMEOUT":
|
||||||
|
return (000, "timeout after " + str(timeout) + "s")
|
||||||
|
# Otherwise, determine failure kind from curl exit code
|
||||||
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
|
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
|
||||||
if rc == 28:
|
elif rc == 28:
|
||||||
return (000, "timeout after " + str(timeout) + "s")
|
return (000, "timeout after " + str(timeout) + "s")
|
||||||
elif rc == 7:
|
elif rc == 7:
|
||||||
return (000, "connection refused")
|
return (000, "connection refused")
|
||||||
@@ -73,6 +76,20 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
|
|||||||
except subprocess.TimeoutExpired:
|
except subprocess.TimeoutExpired:
|
||||||
return (000, "timeout after " + str(timeout) + "s")
|
return (000, "timeout after " + str(timeout) + "s")
|
||||||
|
|
||||||
|
def check_host_health(host_ip):
|
||||||
|
"""Check if the GPU host's llama-chat-api health endpoint is reachable
|
||||||
|
|
||||||
|
Returns: (healthy: bool, detail: str)
|
||||||
|
"""
|
||||||
|
code, _ = probe_http("http://" + host_ip + ":8080/health", timeout=10)
|
||||||
|
if code == 200:
|
||||||
|
return True, "host healthy (200)"
|
||||||
|
elif code == 000:
|
||||||
|
return False, "host unreachable (timeout or refused)"
|
||||||
|
else:
|
||||||
|
return False, "host unhealthy (HTTP " + str(code) + ")"
|
||||||
|
|
||||||
|
|
||||||
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
|
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
|
||||||
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
|
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
|
||||||
cmd = "curl -s -m " + str(timeout)
|
cmd = "curl -s -m " + str(timeout)
|
||||||
@@ -115,28 +132,40 @@ def check_model_probes():
|
|||||||
|
|
||||||
results = []
|
results = []
|
||||||
|
|
||||||
|
# Host health mapping: model -> host IP
|
||||||
|
model_hosts = {
|
||||||
|
"gpu-dense": "192.168.68.8", # RTX 3090
|
||||||
|
"gpu-vision": "192.168.68.110", # RTX 5070
|
||||||
|
"strix-moe": "192.168.68.15" # Strix Halo
|
||||||
|
}
|
||||||
|
|
||||||
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
||||||
# Single-host aliases: 30s initial timeout, retry once at 45s on failure
|
host_ip = model_hosts[model]
|
||||||
# gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold
|
# Single-host aliases: 30s initial timeout, retry once at 90s on failure
|
||||||
|
# Worst-case prefill ~76s, so 90s retry ensures we cover it
|
||||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||||
timeout=30)
|
timeout=30)
|
||||||
|
|
||||||
first_kind = None # Track first attempt's failure kind
|
first_kind = None
|
||||||
if code == 000 and failure_kind:
|
if code == 000 and failure_kind:
|
||||||
# Retry once with longer timeout (45s) before declaring failure
|
|
||||||
first_kind = failure_kind
|
first_kind = failure_kind
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
||||||
timeout=45)
|
timeout=90)
|
||||||
|
|
||||||
if code == 000 and failure_kind:
|
if code == 000 and failure_kind:
|
||||||
# Both attempts failed - report both kinds
|
# Both attempts failed - check host health to distinguish busy from down
|
||||||
|
host_healthy, host_detail = check_host_health(host_ip)
|
||||||
|
if host_healthy:
|
||||||
|
results.append((model, False, "busy (completion timed out after retry; " + host_detail + ")"))
|
||||||
|
else:
|
||||||
|
# Host unreachable - report both kinds
|
||||||
if first_kind:
|
if first_kind:
|
||||||
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
|
results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)"))
|
||||||
else:
|
else:
|
||||||
@@ -144,14 +173,12 @@ def check_model_probes():
|
|||||||
elif code == 200:
|
elif code == 200:
|
||||||
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
||||||
elif code in (401, 403):
|
elif code in (401, 403):
|
||||||
# Credential fault - capture body and key alias
|
|
||||||
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
||||||
method="POST",
|
method="POST",
|
||||||
bearer_token=monitor_key,
|
bearer_token=monitor_key,
|
||||||
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
|
||||||
timeout=10)
|
timeout=10)
|
||||||
# Resolve key alias
|
alias = "monitor-20260813"
|
||||||
alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116
|
|
||||||
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
|
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
|
||||||
else:
|
else:
|
||||||
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
||||||
@@ -261,6 +288,7 @@ def main():
|
|||||||
print("")
|
print("")
|
||||||
|
|
||||||
all_pass = True
|
all_pass = True
|
||||||
|
degraded = [] # Track degraded (busy) checks
|
||||||
|
|
||||||
# Run all checks
|
# Run all checks
|
||||||
checks = [
|
checks = [
|
||||||
@@ -280,9 +308,15 @@ def main():
|
|||||||
# Model probes
|
# Model probes
|
||||||
model_results = check_model_probes()
|
model_results = check_model_probes()
|
||||||
for name, passed, detail in model_results:
|
for name, passed, detail in model_results:
|
||||||
|
# Check if this is a busy (degraded) verdict
|
||||||
|
if not passed and detail.startswith("busy "):
|
||||||
|
status = "⚠️"
|
||||||
|
degraded.append(name)
|
||||||
|
else:
|
||||||
status = "✅" if passed else "❌"
|
status = "✅" if passed else "❌"
|
||||||
print(" " + status + " " + name + ": " + detail)
|
print(" " + status + " " + name + ": " + detail)
|
||||||
if not passed:
|
# Only set all_pass=False for real failures (not busy)
|
||||||
|
if not passed and not detail.startswith("busy "):
|
||||||
all_pass = False
|
all_pass = False
|
||||||
|
|
||||||
# Admin key list
|
# Admin key list
|
||||||
@@ -308,8 +342,14 @@ def main():
|
|||||||
|
|
||||||
print("")
|
print("")
|
||||||
if all_pass:
|
if all_pass:
|
||||||
|
if degraded:
|
||||||
|
print("✅ All checks passed (" + str(len(degraded)) + " degraded: " + ", ".join(degraded) + ")")
|
||||||
|
else:
|
||||||
print("✅ All checks passed")
|
print("✅ All checks passed")
|
||||||
return 0
|
return 0
|
||||||
|
else:
|
||||||
|
if degraded:
|
||||||
|
print("❌ Some checks failed (" + str(len(degraded)) + " degraded: " + ", ".join(degraded) + ")")
|
||||||
else:
|
else:
|
||||||
print("❌ Some checks failed")
|
print("❌ Some checks failed")
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
Reference in New Issue
Block a user