From 8a4dd08b05e19dc5a7872db065268993eca65076 Mon Sep 17 00:00:00 2001 From: root Date: Tue, 15 Sep 2026 05:11:07 +0000 Subject: [PATCH] fix: capture 401/403 response body and key alias for credential faults - get_response_body() returns first 200 chars of response body (single line) - On 401/403 model probe: report code + body + key_alias - Monitor key alias: monitor-20260813 (from /etc/litellm-monitor.env on CT 116) - Failed connections stay probe-failed, 200 stays plain 200 - Do not turn other statuses into credential faults Signed-off-by: Abiba --- scripts/litellm-health-check.py | 35 +++++++++++++++++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/scripts/litellm-health-check.py b/scripts/litellm-health-check.py index a5f2911..03fb808 100755 --- a/scripts/litellm-health-check.py +++ b/scripts/litellm-health-check.py @@ -73,6 +73,23 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll except subprocess.TimeoutExpired: return (000, "timeout after " + str(timeout) + "s") +def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30): + """Get response body for 401/403 credential faults (truncated to 200 chars)""" + cmd = "curl -s -m " + str(timeout) + if method == "POST": + cmd += " -X POST" + if bearer_token: + cmd += " -H 'Authorization: Bearer " + bearer_token + "'" + if data: + cmd += " -H 'Content-Type: application/json' -d '" + data + "'" + cmd += " '" + url + "'" + + rc, stdout, stderr = run_command(cmd, timeout) + # Return first 200 chars, single line + body = stdout.replace('\n', ' ').replace('\t', ' ')[:200] if stdout else "" + return body + + def check_liveliness(): """Step 1: Liveliness probe""" code, _ = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness") @@ -112,6 +129,16 @@ def check_model_probes(): results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)")) elif code == 200: results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) + elif code in (401, 403): + # Credential fault - capture body and key alias + body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}', + timeout=10) + # Resolve key alias + alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116 + results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias)) else: results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) @@ -132,6 +159,14 @@ def check_model_probes(): timeout=60) if code == 000 and failure_kind: results.append(("syslog-auto", False, "probe-failed: syslog-auto " + failure_kind + " (60s timeout, retry)")) + elif code in (401, 403): + body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", + method="POST", + bearer_token=monitor_key, + data='{"model":"syslog-auto","messages":[{"role":"user","content":"health"}],"max_tokens":4}', + timeout=10) + alias = "monitor-20260813" + results.append(("syslog-auto", False, str(code) + " credential fault: body=" + body + " key_alias=" + alias)) else: results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)")) else: