|
|
|
@@ -35,12 +35,7 @@ def run_command(cmd, timeout=15):
|
|
|
|
|
return 1, "", str(e)
|
|
|
|
|
|
|
|
|
|
def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False):
|
|
|
|
|
"""Probe HTTP endpoint and return (status_code, failure_kind)
|
|
|
|
|
|
|
|
|
|
Returns:
|
|
|
|
|
(code, None) if successful or HTTP response received
|
|
|
|
|
(000, kind) if connection failed, where kind is 'timeout', 'refused', 'dns', etc.
|
|
|
|
|
"""
|
|
|
|
|
"""Probe HTTP endpoint and return status code"""
|
|
|
|
|
cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout)
|
|
|
|
|
if method == "POST":
|
|
|
|
|
cmd += " -X POST"
|
|
|
|
@@ -52,47 +47,15 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
|
|
|
|
|
cmd += " -L"
|
|
|
|
|
cmd += " '" + url + "'"
|
|
|
|
|
|
|
|
|
|
try:
|
|
|
|
|
rc, stdout, stderr = run_command(cmd, timeout)
|
|
|
|
|
if rc != 0:
|
|
|
|
|
# Determine failure kind from curl exit code
|
|
|
|
|
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
|
|
|
|
|
if rc == 28:
|
|
|
|
|
return (000, "timeout after " + str(timeout) + "s")
|
|
|
|
|
elif rc == 7:
|
|
|
|
|
return (000, "connection refused")
|
|
|
|
|
elif rc == 6:
|
|
|
|
|
return (000, "dns failure")
|
|
|
|
|
elif rc == 35:
|
|
|
|
|
return (000, "ssl error")
|
|
|
|
|
elif rc == 52:
|
|
|
|
|
return (000, "empty response")
|
|
|
|
|
else:
|
|
|
|
|
return (000, "curl exit " + str(rc))
|
|
|
|
|
return (int(stdout), None) if stdout.isdigit() else (000, "unparseable response")
|
|
|
|
|
except subprocess.TimeoutExpired:
|
|
|
|
|
return (000, "timeout after " + str(timeout) + "s")
|
|
|
|
|
|
|
|
|
|
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
|
|
|
|
|
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
|
|
|
|
|
cmd = "curl -s -m " + str(timeout)
|
|
|
|
|
if method == "POST":
|
|
|
|
|
cmd += " -X POST"
|
|
|
|
|
if bearer_token:
|
|
|
|
|
cmd += " -H 'Authorization: Bearer " + bearer_token + "'"
|
|
|
|
|
if data:
|
|
|
|
|
cmd += " -H 'Content-Type: application/json' -d '" + data + "'"
|
|
|
|
|
cmd += " '" + url + "'"
|
|
|
|
|
|
|
|
|
|
rc, stdout, stderr = run_command(cmd, timeout)
|
|
|
|
|
# Return first 200 chars, single line
|
|
|
|
|
body = stdout.replace('\n', ' ').replace('\t', ' ')[:200] if stdout else ""
|
|
|
|
|
return body
|
|
|
|
|
|
|
|
|
|
if rc != 0 and "TIMEOUT" not in stderr:
|
|
|
|
|
return 000 # Connection failed
|
|
|
|
|
|
|
|
|
|
return int(stdout) if stdout.isdigit() else 000
|
|
|
|
|
|
|
|
|
|
def check_liveliness():
|
|
|
|
|
"""Step 1: Liveliness probe"""
|
|
|
|
|
code, _ = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness")
|
|
|
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness")
|
|
|
|
|
return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)"
|
|
|
|
|
|
|
|
|
|
def check_containers():
|
|
|
|
@@ -116,61 +79,32 @@ def check_model_probes():
|
|
|
|
|
results = []
|
|
|
|
|
|
|
|
|
|
for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
|
|
|
|
|
# Single-host aliases: 30s timeout each
|
|
|
|
|
# gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start
|
|
|
|
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
|
|
|
timeout=30)
|
|
|
|
|
# Single-host aliases: 30s timeout
|
|
|
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
|
|
|
timeout=30)
|
|
|
|
|
|
|
|
|
|
if code == 000 and failure_kind:
|
|
|
|
|
# Report probe failure with kind, do not assert a service verdict
|
|
|
|
|
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)"))
|
|
|
|
|
elif code == 200:
|
|
|
|
|
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
|
|
|
|
elif code in (401, 403):
|
|
|
|
|
# Credential fault - capture body and key alias
|
|
|
|
|
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
|
|
|
|
|
timeout=10)
|
|
|
|
|
# Resolve key alias
|
|
|
|
|
alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116
|
|
|
|
|
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
|
|
|
|
|
else:
|
|
|
|
|
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
|
|
|
|
results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
|
|
|
|
|
|
|
|
|
|
# Pool alias (syslog-auto): 60s timeout, retry once on 000
|
|
|
|
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
|
|
|
timeout=60)
|
|
|
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
|
|
|
timeout=60)
|
|
|
|
|
|
|
|
|
|
if code == 000 and failure_kind:
|
|
|
|
|
if code == 000:
|
|
|
|
|
# Retry once with same timeout
|
|
|
|
|
time.sleep(1)
|
|
|
|
|
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
|
|
|
timeout=60)
|
|
|
|
|
if code == 000 and failure_kind:
|
|
|
|
|
results.append(("syslog-auto", False, "probe-failed: syslog-auto " + failure_kind + " (60s timeout, retry)"))
|
|
|
|
|
elif code in (401, 403):
|
|
|
|
|
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
|
|
|
|
|
timeout=10)
|
|
|
|
|
alias = "monitor-20260813"
|
|
|
|
|
results.append(("syslog-auto", False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
|
|
|
|
|
else:
|
|
|
|
|
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
|
|
|
|
|
else:
|
|
|
|
|
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
|
|
|
|
|
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
|
|
|
|
|
method="POST",
|
|
|
|
|
bearer_token=monitor_key,
|
|
|
|
|
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
|
|
|
|
|
timeout=60)
|
|
|
|
|
|
|
|
|
|
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
|
|
|
|
|
|
|
|
|
|
return results
|
|
|
|
|
|
|
|
|
@@ -212,18 +146,18 @@ def check_admin_key_list():
|
|
|
|
|
|
|
|
|
|
def check_github_status():
|
|
|
|
|
"""Step 3: GitHub status - 301 redirect is acceptable for status page"""
|
|
|
|
|
code, _ = probe_http("https://status.github.com/api/status.json", timeout=15)
|
|
|
|
|
code = probe_http("https://status.github.com/api/status.json", timeout=15)
|
|
|
|
|
# GitHub status API returns 301 redirect, which is expected behavior
|
|
|
|
|
return "GitHub Status", code == 301, str(code)
|
|
|
|
|
|
|
|
|
|
def check_prometheus():
|
|
|
|
|
"""Step 4: Prometheus health"""
|
|
|
|
|
code, _ = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy")
|
|
|
|
|
code = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy")
|
|
|
|
|
return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)"
|
|
|
|
|
|
|
|
|
|
def check_grafana():
|
|
|
|
|
"""Step 9: Grafana health"""
|
|
|
|
|
code, _ = probe_http("http://" + BACKEND_HOST + ":3001/api/health")
|
|
|
|
|
code = probe_http("http://" + BACKEND_HOST + ":3001/api/health")
|
|
|
|
|
return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)"
|
|
|
|
|
|
|
|
|
|
def check_docker_stats():
|
|
|
|
|