From 0b9aebca37795eb9311ff21372189911d71f2959 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 11:32:15 +0000 Subject: [PATCH 1/3] fix(litellm): Fix timeout kind reporting + add busy/degraded detection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1. TIMEOUT KIND FIX: run_command returns (1, '', 'TIMEOUT') when its own timeout fires. probe_http now checks for this before falling through to 'curl exit ', so a 30s timeout reports 'timeout after 30s' not 'curl exit 1'. 2. BUSY/DEGRADED DETECTION: After both model probes fail, check the model's host health endpoint (e.g. 192.168.68.8:8080/health for gpu-dense). If the host answers 200, report 'busy (completion timed out after retry; host healthy 200)' — do NOT fail the run on that alone. If the host does not answer, that's a real FAIL. 3. RETRY TIMEOUT INCREASED: Single-host retry timeout raised from 45s to 90s. Worst-case prefill on a single-slot .8 host is ~76s (observed 83K-token prompt at 1078 tok/s), so 90s covers it. New line shapes: - Busy: 'gpu-dense: busy (completion timed out after retry; host healthy 200)' - Real failure: 'probe-failed: gpu-dense timeout after 30s then timeout after 90s (2 attempts)' --- scripts/litellm-health-check.py | 55 ++++++++++++++++++++++++--------- 1 file changed, 41 insertions(+), 14 deletions(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 50ec076..89a9785 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -55,9 +55,12 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll try: rc, stdout, stderr = run_command(cmd, timeout) if rc != 0: - # Determine failure kind from curl exit code + # Check if this is a timeout from run_command (rc=1, stderr="TIMEOUT") + if rc == 1 and stderr == "TIMEOUT": + return (000, "timeout after " + str(timeout) + "s") + # Otherwise, determine failure kind from curl exit code # curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty - if rc == 28: + elif rc == 28: return (000, "timeout after " + str(timeout) + "s") elif rc == 7: return (000, "connection refused") @@ -73,6 +76,20 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll except subprocess.TimeoutExpired: return (000, "timeout after " + str(timeout) + "s") +def check_host_health(host_ip): + """Check if the GPU host's llama-chat-api health endpoint is reachable + + Returns: (healthy: bool, detail: str) + """ + code, _ = probe_http("http://" + host_ip + ":8080/health", timeout=10) + if code == 200: + return True, "host healthy (200)" + elif code == 000: + return False, "host unreachable (timeout or refused)" + else: + return False, "host unhealthy (HTTP " + str(code) + ")" + + def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30): """Get response body for 401/403 credential faults (truncated to 200 chars)""" cmd = "curl -s -m " + str(timeout) @@ -115,43 +132,53 @@ def check_model_probes(): results = [] + # Host health mapping: model -> host IP + model_hosts = { + "gpu-dense": "192.168.68.8", # RTX 3090 + "gpu-vision": "192.168.68.110", # RTX 5070 + "strix-moe": "192.168.68.15" # Strix Halo + } + for model in ["gpu-dense", "gpu-vision", "strix-moe"]: - # Single-host aliases: 30s initial timeout, retry once at 45s on failure - # gpu-dense (RTX 3090) may need long warmup/prefill or concurrent generation hold + host_ip = model_hosts[model] + # Single-host aliases: 30s initial timeout, retry once at 90s on failure + # Worst-case prefill ~76s, so 90s retry ensures we cover it code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', timeout=30) - first_kind = None # Track first attempt's failure kind + first_kind = None if code == 000 and failure_kind: - # Retry once with longer timeout (45s) before declaring failure first_kind = failure_kind time.sleep(1) code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', - timeout=45) + timeout=90) if code == 000 and failure_kind: - # Both attempts failed - report both kinds - if first_kind: - results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)")) + # Both attempts failed - check host health to distinguish busy from down + host_healthy, host_detail = check_host_health(host_ip) + if host_healthy: + results.append((model, False, "busy (completion timed out after retry; host healthy " + host_detail + ")")) else: - results.append((model, False, "probe-failed: " + model + " " + failure_kind)) + # Host unreachable - report both kinds + if first_kind: + results.append((model, False, "probe-failed: " + model + " " + first_kind + " then " + failure_kind + " (2 attempts)")) + else: + results.append((model, False, "probe-failed: " + model + " " + failure_kind)) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) elif code in (401, 403): - # Credential fault - capture body and key alias body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", method="POST", bearer_token=monitor_key, data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}', timeout=10) - # Resolve key alias - alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116 + alias = "monitor-20260813" results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias)) else: results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) -- 2.54.0 From a12abbeb14c7f8265875ed503222743e7365a626 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 11:43:39 +0000 Subject: [PATCH 2/3] fix(litellm): Implement busy vs down with proper degraded state MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three states: - healthy: passed, exit 0 (unchanged) - busy (completion timed out after retry AND host /health answered): ⚠️ DEGRADED line, does NOT fail the run, exit 0 - host unreachable or real fault: ❌, exit 1 (unchanged) Summary now reports degraded count: - All pass, no degraded: '✅ All checks passed' - All pass, 1+ degraded: '✅ All checks passed (1 degraded: gpu-dense)' - Some failed: '❌ Some checks failed' or '❌ Some checks failed (1 degraded: ...)' Host health mapping verified: - gpu-dense -> 192.168.68.8:8080/health - gpu-vision -> 192.168.68.110:8080/health - strix-moe -> 192.168.68.15:8080/health --- scripts/litellm-health-check.py | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 89a9785..19170b9 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -288,6 +288,7 @@ def main(): print("") all_pass = True + degraded = [] # Track degraded (busy) checks # Run all checks checks = [ @@ -307,9 +308,15 @@ def main(): # Model probes model_results = check_model_probes() for name, passed, detail in model_results: - status = "✅" if passed else "❌" + # Check if this is a busy (degraded) verdict + if not passed and detail.startswith("busy "): + status = "⚠️" + degraded.append(name) + else: + status = "✅" if passed else "❌" print(" " + status + " " + name + ": " + detail) - if not passed: + # Only set all_pass=False for real failures (not busy) + if not passed and not detail.startswith("busy "): all_pass = False # Admin key list @@ -335,10 +342,16 @@ def main(): print("") if all_pass: - print("✅ All checks passed") + if degraded: + print("✅ All checks passed (" + str(len(degraded)) + " degraded: " + ", ".join(degraded) + ")") + else: + print("✅ All checks passed") return 0 else: - print("❌ Some checks failed") + if degraded: + print("❌ Some checks failed (" + str(len(degraded)) + " degraded: " + ", ".join(degraded) + ")") + else: + print("❌ Some checks failed") return 1 if __name__ == "__main__": -- 2.54.0 From 9a2ee6faecaefb6e1c3bc073c44dd351f3a8b59a Mon Sep 17 00:00:00 2001 From: root Date: Sat, 19 Sep 2026 11:53:32 +0000 Subject: [PATCH 3/3] fix(litellm): Remove duplicate 'host healthy' from busy line The busy line was rendering as: 'busy (completion timed out after retry; host healthy host healthy (200))' because host_detail already contains 'host healthy (200)' and the prefix also said 'host healthy'. Fixed to: 'busy (completion timed out after retry; host healthy (200))' F1 cosmetic fix from PR #123 verify. --- scripts/litellm-health-check.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index 19170b9..c94617c 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -163,7 +163,7 @@ def check_model_probes(): # Both attempts failed - check host health to distinguish busy from down host_healthy, host_detail = check_host_health(host_ip) if host_healthy: - results.append((model, False, "busy (completion timed out after retry; host healthy " + host_detail + ")")) + results.append((model, False, "busy (completion timed out after retry; " + host_detail + ")")) else: # Host unreachable - report both kinds if first_kind: -- 2.54.0