Merge pull request 'fix(monitoring): litellm-health script - pool-alias timeout and leaked key-list debug output' (#88) from fix/litellm-health-timeout-and-debug-20260914 into master

This commit was merged in pull request #88.
This commit is contained in:
2026-09-14 03:34:58 +00:00
+22 -12
View File
@@ -78,18 +78,34 @@ def check_model_probes():
results = [] results = []
for model in ["gpu-dense", "gpu-vision", "strix-moe", "syslog-auto"]: for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
# Use unique prompt per run to avoid caching # Single-host aliases: 30s timeout
prompt = "health " + str(random.randint(1000, 9999))
data = '{"model":"' + model + '","messages":[{"role":"user","content":"' + prompt + '"}],"max_tokens":4}'
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST", method="POST",
bearer_token=monitor_key, bearer_token=monitor_key,
data=data) data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=30)
results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
# Pool alias (syslog-auto): 60s timeout, retry once on 000
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=60)
if code == 000:
# Retry once with same timeout
time.sleep(1)
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=60)
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
return results return results
def check_admin_key_list(): def check_admin_key_list():
@@ -105,9 +121,6 @@ def check_admin_key_list():
if not mk or "NO-CURL" in mk: if not mk or "NO-CURL" in mk:
return "Admin Key List", False, "credential-missing (empty or NO-CURL)" return "Admin Key List", False, "credential-missing (empty or NO-CURL)"
# Print key length for debugging
print(" DEBUG: keylen=" + str(len(mk)), file=sys.stderr)
# Step 2: Call using the key - use double quotes inside SSH command # Step 2: Call using the key - use double quotes inside SSH command
cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"curl -s -H \\\"Authorization: Bearer " + mk + "\\\" http://127.0.0.1:4000/key/list\"" cmd = "ssh -o BatchMode=yes root@192.168.68.116 \"curl -s -H \\\"Authorization: Bearer " + mk + "\\\" http://127.0.0.1:4000/key/list\""
rc, stdout, stderr = run_command(cmd) rc, stdout, stderr = run_command(cmd)
@@ -115,9 +128,6 @@ def check_admin_key_list():
if rc != 0: if rc != 0:
return "Admin Key List", False, "admin-call-failed (exit=" + str(rc) + ", stderr=" + stderr + ")" return "Admin Key List", False, "admin-call-failed (exit=" + str(rc) + ", stderr=" + stderr + ")"
# Print response for debugging
print(" DEBUG: response=" + stdout[:120] + "...", file=sys.stderr)
# Try to parse the response # Try to parse the response
try: try:
data = json.loads(stdout) data = json.loads(stdout)