fix(keys): correct the key-lifecycle contracts to the measured truth #95

Merged
abiba-bot merged 5 commits from fix/key-expiry-enforcement-20260915 into master 2026-09-15 05:37:46 +00:00
3 changed files with 108 additions and 32 deletions
+4 -2
View File
@@ -207,9 +207,11 @@ The agent picks up the new key via `infisical run --` at gateway startup.
**Keys are permanent and use bare agent name aliases.** **Keys are permanent and use bare agent name aliases.**
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.) - **Duration**: `null` — keys never expire by default. **Expiry must be set EXPLICITLY at creation** with the `duration` parameter (e.g., `90d` for 90 days). The 90-day default is the standard; however, the config default is **NOT honoured** by LiteLLM 1.99.1 (verified on CT 116: a key generated with no explicit duration returns `expires=null`). This has been recorded in `/opt/inference-harness/litellm_config.yaml` to prevent re-filing as a bug.
- **Daily Audit**: A daily audit job runs at 00:00 UTC (`/usr/local/bin/litellm-key-renewal-ct116.sh`, cron 00:00). It is **AUDIT-ONLY** and does not perform renewal. It lists every key, reports those with no expiry and those inside a 14-day warning window, explicitly EXCLUDES `abiba-pi` and `koby` (report-only, and .129 must never be touched), and logs `RENEWAL-REQUIRED-BUT-NOT-PERFORMED + NO KEY WAS CHANGED` when renewal is skipped. **Renewal is NOT implemented** — keys must not be rotated until delivery (vault injection + consumer verification) exists and is proven end-to-end.
- **Exclusions**: `abiba-pi` and every firstmate/secondmate/crewmate key stay **WITHOUT an expiry** until a proven renewal path exists. `koby` is **report-only** (never touched). These exclusions are enforced by the audit job.
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity. - **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven. - **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven. Manual rotation is permitted only when the renewal delivery path is proven and verified on a throwaway consumer before production use.
- **Max budget**: $100 per key (config default). - **Max budget**: $100 per key (config default).
```yaml ```yaml
+9 -1
View File
@@ -320,7 +320,15 @@ directly call OpenRouter via Python's requests library. Converting would require
## LiteLLM Master Key (use sparingly — agents should NOT use it directly) ## LiteLLM Master Key (use sparingly — agents should NOT use it directly)
- Master key: `sk-litellm-7f96080dd99b15c36bd4b333b58a6796` (in /opt/inference-harness/.env on CT116, Infisical project=infrastructure env=production secret=LITELLM_MASTER_KEY) - Master key: **Retrieval path (do not trust a literal value in this file — the key rotates)**:
```bash
# PRIMARY (proven, runs on CT 116 with no extra tooling):
docker exec harness-litellm printenv LITELLM_MASTER_KEY
# Note: the same value is stored in /opt/inference-harness/.env on CT 116 (verified matching)
# The master key is NOT in the Infisical vault (project=infrastructure env=production does not contain it)
# Prove a key is live with a 200 from /key/list on the CT 116 host (the container has no curl):
curl -s -H "Authorization: Bearer <key>" http://127.0.0.1:4000/key/list | jq length
```
- Used for /key/generate, /key/delete, /key/list (GET), DB queries - Used for /key/generate, /key/delete, /key/list (GET), DB queries
- **Known violation (RESOLVED 2026-07-16):** Abiba's LITELLM_API_KEY was previously the master key. - **Known violation (RESOLVED 2026-07-16):** Abiba's LITELLM_API_KEY was previously the master key.
It is now a dedicated agent key `sk-sxbphLvk1OU…` (vault secret `ABIBA_LITELLM_API_KEY`, alias `abiba-pi`). It is now a dedicated agent key `sk-sxbphLvk1OU…` (vault secret `ABIBA_LITELLM_API_KEY`, alias `abiba-pi`).
+81 -15
View File
@@ -35,7 +35,12 @@ def run_command(cmd, timeout=15):
return 1, "", str(e) return 1, "", str(e)
def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False): def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, follow_redirects=False):
"""Probe HTTP endpoint and return status code""" """Probe HTTP endpoint and return (status_code, failure_kind)
Returns:
(code, None) if successful or HTTP response received
(000, kind) if connection failed, where kind is 'timeout', 'refused', 'dns', etc.
"""
cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout) cmd = "curl -s -o /dev/null -w '%{http_code}' -m " + str(timeout)
if method == "POST": if method == "POST":
cmd += " -X POST" cmd += " -X POST"
@@ -47,15 +52,47 @@ def probe_http(url, method="GET", bearer_token=None, data=None, timeout=10, foll
cmd += " -L" cmd += " -L"
cmd += " '" + url + "'" cmd += " '" + url + "'"
try:
rc, stdout, stderr = run_command(cmd, timeout) rc, stdout, stderr = run_command(cmd, timeout)
if rc != 0 and "TIMEOUT" not in stderr: if rc != 0:
return 000 # Connection failed # Determine failure kind from curl exit code
# curl exit codes: 28=timeout, 7=refused, 6=dns, 35=ssl, 52=empty
if rc == 28:
return (000, "timeout after " + str(timeout) + "s")
elif rc == 7:
return (000, "connection refused")
elif rc == 6:
return (000, "dns failure")
elif rc == 35:
return (000, "ssl error")
elif rc == 52:
return (000, "empty response")
else:
return (000, "curl exit " + str(rc))
return (int(stdout), None) if stdout.isdigit() else (000, "unparseable response")
except subprocess.TimeoutExpired:
return (000, "timeout after " + str(timeout) + "s")
def get_response_body(url, method="POST", bearer_token=None, data=None, timeout=30):
"""Get response body for 401/403 credential faults (truncated to 200 chars)"""
cmd = "curl -s -m " + str(timeout)
if method == "POST":
cmd += " -X POST"
if bearer_token:
cmd += " -H 'Authorization: Bearer " + bearer_token + "'"
if data:
cmd += " -H 'Content-Type: application/json' -d '" + data + "'"
cmd += " '" + url + "'"
rc, stdout, stderr = run_command(cmd, timeout)
# Return first 200 chars, single line
body = stdout.replace('\n', ' ').replace('\t', ' ')[:200] if stdout else ""
return body
return int(stdout) if stdout.isdigit() else 000
def check_liveliness(): def check_liveliness():
"""Step 1: Liveliness probe""" """Step 1: Liveliness probe"""
code = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness") code, _ = probe_http("http://" + BACKEND_HOST + "/litellm/health/liveliness")
return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)" return "Liveliness", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/health/liveliness)"
def check_containers(): def check_containers():
@@ -79,31 +116,60 @@ def check_model_probes():
results = [] results = []
for model in ["gpu-dense", "gpu-vision", "strix-moe"]: for model in ["gpu-dense", "gpu-vision", "strix-moe"]:
# Single-host aliases: 30s timeout # Single-host aliases: 30s timeout each
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", # gpu-dense (RTX 3090) may need long warmup/prefill - timeout is acceptable on cold-start
code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST", method="POST",
bearer_token=monitor_key, bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', data='{"model":"' + model + '","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=30) timeout=30)
results.append((model, code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")")) if code == 000 and failure_kind:
# Report probe failure with kind, do not assert a service verdict
results.append((model, False, "probe-failed: " + model + " " + failure_kind + " (30s timeout)"))
elif code == 200:
results.append((model, True, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
elif code in (401, 403):
# Credential fault - capture body and key alias
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"' + model + '","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
timeout=10)
# Resolve key alias
alias = "monitor-20260813" # Known from /etc/litellm-monitor.env on CT 116
results.append((model, False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
else:
results.append((model, False, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=" + model + ")"))
# Pool alias (syslog-auto): 60s timeout, retry once on 000 # Pool alias (syslog-auto): 60s timeout, retry once on 000
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST", method="POST",
bearer_token=monitor_key, bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=60) timeout=60)
if code == 000: if code == 000 and failure_kind:
# Retry once with same timeout # Retry once with same timeout
time.sleep(1) time.sleep(1)
code = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions", code, failure_kind = probe_http("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST", method="POST",
bearer_token=monitor_key, bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}', data='{"model":"syslog-auto","messages":[{"role":"user","content":"health ' + str(random.randint(1000, 9999)) + '"}],"max_tokens":4}',
timeout=60) timeout=60)
if code == 000 and failure_kind:
results.append(("syslog-auto", False, "probe-failed: syslog-auto " + failure_kind + " (60s timeout, retry)"))
elif code in (401, 403):
body = get_response_body("http://" + BACKEND_HOST + "/litellm/v1/chat/completions",
method="POST",
bearer_token=monitor_key,
data='{"model":"syslog-auto","messages":[{"role":"user","content":"health"}],"max_tokens":4}',
timeout=10)
alias = "monitor-20260813"
results.append(("syslog-auto", False, str(code) + " credential fault: body=" + body + " key_alias=" + alias))
else:
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
else:
results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)")) results.append(("syslog-auto", code == 200, str(code) + " (target: " + BACKEND_HOST + "/litellm/v1/chat/completions, model=syslog-auto)"))
return results return results
@@ -146,18 +212,18 @@ def check_admin_key_list():
def check_github_status(): def check_github_status():
"""Step 3: GitHub status - 301 redirect is acceptable for status page""" """Step 3: GitHub status - 301 redirect is acceptable for status page"""
code = probe_http("https://status.github.com/api/status.json", timeout=15) code, _ = probe_http("https://status.github.com/api/status.json", timeout=15)
# GitHub status API returns 301 redirect, which is expected behavior # GitHub status API returns 301 redirect, which is expected behavior
return "GitHub Status", code == 301, str(code) return "GitHub Status", code == 301, str(code)
def check_prometheus(): def check_prometheus():
"""Step 4: Prometheus health""" """Step 4: Prometheus health"""
code = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy") code, _ = probe_http("http://" + BACKEND_HOST + ":9090/-/healthy")
return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)" return "Prometheus", code == 200, str(code) + " (target: " + BACKEND_HOST + ":9090/-/healthy)"
def check_grafana(): def check_grafana():
"""Step 9: Grafana health""" """Step 9: Grafana health"""
code = probe_http("http://" + BACKEND_HOST + ":3001/api/health") code, _ = probe_http("http://" + BACKEND_HOST + ":3001/api/health")
return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)" return "Grafana", code == 200, str(code) + " (target: " + BACKEND_HOST + ":3001/api/health)"
def check_docker_stats(): def check_docker_stats():