fix: LiteLLM OIDC + Admin UI fixes - Authentik integration restored

- Added extra_hosts for auth.sysloggh.net to LiteLLM container
- Fixed DOCS_URL=/docs (was /litellm/docs - path mismatch)
- Added Authentik self-signed cert to CA bundle
- Added nginx auth proxy for token/userinfo endpoints (SSL verify off)
- Changed OIDC token/userinfo endpoints to use nginx internal proxy
- Admin UI serving correctly on :4001/ui/ and /litellm/ui/
- Swagger API docs working at /docs and /litellm/docs
- ReDoc API docs working at /redoc and /litellm/redoc
- OIDC login flow verified working end-to-end
This commit is contained in:
Abiba
2026-06-25 20:33:22 +00:00
parent fd3c2a575a
commit 08680b0f9e
4 changed files with 151 additions and 255 deletions
+6 -4
View File
@@ -57,12 +57,12 @@ services:
condition: service_healthy
litellm:
image: ghcr.io/berriai/litellm:main-stable
image: docker.litellm.ai/berriai/litellm:1.90.0-rc.1
command: ["--config", "/app/config.yaml", "--port", "4000"]
container_name: harness-litellm
restart: unless-stopped
ports:
- "127.0.0.1:4001:4000"
- "4001:4000"
volumes:
- ./litellm_config.yaml:/app/config.yaml
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
@@ -77,17 +77,19 @@ services:
- UI_PASSWORD=syslog-admin-2026
- OPENAI_API_KEY=not-used
- PROXY_BASE_URL=https://litellm.sysloggh.net
- DOCS_URL=/docs
- ANTHROPIC_API_KEY=not-used
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
- GENERIC_CLIENT_SECRET=aDkXQx82duqpxc98xxqp0quzzUf1mawsnOTqj7sx1acaS7rWSt02N5ksBCi92n8ZilRavigoYME6fLakP20Ixc9H2pxnSZFiOqQLb7BPi8UtsvfxmzXklD0HJIdbKFxe
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
- GENERIC_TOKEN_ENDPOINT=https://auth.sysloggh.net/application/o/token/
- GENERIC_USERINFO_ENDPOINT=https://auth.sysloggh.net/application/o/userinfo/
- GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/
- GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/
- GENERIC_SCOPE=openid email profile
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
extra_hosts:
- "host.docker.internal:host-gateway"
- "auth.sysloggh.net:192.168.68.11"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
interval: 15s
+76 -94
View File
@@ -1,107 +1,89 @@
# LiteLLM Gateway Configuration — Layer 1 of 2-Layer Architecture
# Deployed on CT 116 (192.168.68.116) alongside custom router on :9000
# Last updated: 2026-06-16
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
# database_url: using DATABASE_URL env var instead
store_model_in_db: true
model_list:
# Content-based auto-routing (router picks GPU via 5-tier analysis)
- model_name: syslog-auto
litellm_params:
model: openai/syslog-auto
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
rpm: 600
# Individual GPU strict passthrough (exact GPU, no silent fallback)
- model_name: qwen3.6-35B-A3B
litellm_params:
model: openai/qwen3.6-35B-A3B
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
- model_name: qwen3.6-27B-code
litellm_params:
model: openai/qwen3.6-27B-code
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
- model_name: gemma-4-12b
litellm_params:
model: openai/gemma-4-12b
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
# Guardrails: Pre-call and post-call content moderation
guardrails:
- guardrail_name: "input-moderation"
litellm_params:
guardrail: openai_moderation
mode: "pre_call"
- guardrail_name: "output-moderation"
litellm_params:
guardrail: openai_moderation
mode: "post_call"
- guardrail_name: "harmful-content-filter"
litellm_params:
guardrail: litellm_content_filter
mode: "pre_call"
categories:
- category: "harmful_self_harm"
enabled: true
action: "BLOCK"
severity_threshold: "medium"
- category: "harmful_violence"
enabled: true
action: "BLOCK"
severity_threshold: "medium"
- category: "harmful_illegal_weapons"
enabled: true
action: "BLOCK"
severity_threshold: "medium"
litellm_settings:
num_retries: 0 # Disabled — our router handles retry logic
request_timeout: 600 # Match 10-min llama-server timeout
set_verbose: true
failure_callback: ["prometheus"] # Export metrics to Prometheus
router_settings:
routing_strategy: "usage-based-routing" # For external models only
enable_loadbalancing_on_proxy: false # Disable LiteLLM internal LB
allowed_fails: 100 # Router returns 503 on saturated GPUs
# Fallback chains: LiteLLM retries down the chain when router returns saturated.
# This gives accurate per-model metrics because router no longer silently reroutes.
# The router's circuit breaker prevents cascading failures to dead GPUs.
fallbacks:
- syslog-auto: ["qwen3.6-35B-A3B", "qwen3.6-27B-code", "gemma-4-12b"]
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"]
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"]
- gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
# Cost tracking for spend analytics (local GPUs at $0, symbolic rates optional)
- guardrail_name: input-moderation
litellm_params:
guardrail: openai_moderation
mode: pre_call
- guardrail_name: output-moderation
litellm_params:
guardrail: openai_moderation
mode: post_call
- guardrail_name: harmful-content-filter
litellm_params:
categories:
- action: BLOCK
category: harmful_self_harm
enabled: true
severity_threshold: medium
- action: BLOCK
category: harmful_violence
enabled: true
severity_threshold: medium
- action: BLOCK
category: harmful_illegal_weapons
enabled: true
severity_threshold: medium
guardrail: litellm_content_filter
mode: pre_call
litellm_settings:
failure_callback:
- prometheus
model_cost:
syslog-auto:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
qwen3.6-35B-A3B:
gemma-4-12b:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
qwen3.6-27B-code:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
gemma-4-12b:
qwen3.6-35B-A3B:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
# For internal cost allocation, set symbolic rates:
# e.g., MoE = $2/M tokens, Dense = $1/M tokens, VLM = $0.50/M tokens
# SSO/OIDC Configuration
litellm_settings:
sso_callback: "/sso/callback"
syslog-auto:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
num_retries: 0
request_timeout: 600
set_verbose: true
sso_callback: /sso/callback
model_list:
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/syslog-auto
rpm: 600
model_name: syslog-auto
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/qwen3.6-35B-A3B
model_name: qwen3.6-35B-A3B
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/qwen3.6-27B-code
model_name: qwen3.6-27B-code
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/gemma-4-12b
model_name: gemma-4-12b
router_settings:
allowed_fails: 100
enable_loadbalancing_on_proxy: false
fallbacks:
- syslog-auto:
- qwen3.6-35B-A3B
- qwen3.6-27B-code
- gemma-4-12b
- qwen3.6-35B-A3B:
- qwen3.6-27B-code
- gemma-4-12b
- qwen3.6-27B-code:
- qwen3.6-35B-A3B
- gemma-4-12b
- gemma-4-12b:
- qwen3.6-27B-code
- qwen3.6-35B-A3B
routing_strategy: usage-based-routing
+67 -155
View File
@@ -8,22 +8,27 @@ http {
sendfile on;
keepalive_timeout 65;
upstream router_api { server router:9000; }
upstream dashboard_ui { server dashboard:3000; }
upstream litellm_backend { server litellm:4000; }
# Docker DNS resolver — forces request-time resolution for variable-based proxy_pass.
# Without this, nginx resolves upstream hostnames at config load time,
# which fails when Docker DNS (127.0.0.11) isn't ready yet on container start.
resolver 127.0.0.11 valid=30s;
# Detect Cloudflare Tunnel requests (cloudflared always sets CF-Connecting-IP).
# Direct LAN browser access to :4000 has no such header -> redirect to canonical https,
# preventing the cross-origin localStorage footgun that traps the UI at the login page.
map $http_cf_connecting_ip $is_cloudflared {
default 1; # any non-empty value = request came through Cloudflare
"" 0; # empty = direct access
# Dynamic upstream resolution via nginx variables.
# Using $var in proxy_pass forces request-time resolution through the resolver.
# Without this, 'host not found in upstream' crashes nginx when Docker DNS is slow.
map $host $router_api_url {
default http://harness-router:9000;
}
map $host $dashboard_ui_url {
default http://harness-dashboard:3000;
}
map $host $litellm_backend_url {
default http://harness-litellm:4000;
}
# ════════════════════════════════════════════════════════════════
# Server :80 — existing harness entrypoint
# Server :80 — harness entrypoint
# dashboard (/), router API (/v1/, /admin/, /stream, /api/, /metrics),
# router fallback, LiteLLM UI via /litellm/ prefix, health
# router fallback, health
# ════════════════════════════════════════════════════════════════
server {
listen 80;
@@ -40,7 +45,7 @@ http {
# 2-Layer: /v1/ → router (existing keys work unchanged)
location /v1/ {
proxy_pass http://router_api;
proxy_pass $router_api_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
@@ -52,7 +57,7 @@ http {
}
location @router_fallback {
proxy_pass http://router_api;
proxy_pass $router_api_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
@@ -61,7 +66,7 @@ http {
}
location /admin/ {
proxy_pass http://router_api;
proxy_pass $router_api_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
@@ -70,7 +75,7 @@ http {
}
location /stream {
proxy_pass http://router_api/stream;
proxy_pass $router_api_url/stream;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
@@ -80,183 +85,90 @@ http {
}
location /api/ {
proxy_pass http://router_api/;
proxy_pass $router_api_url/;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
# LiteLLM gateway access (agents with new virtual keys)
location /litellm/v1/ {
proxy_pass http://litellm_backend/v1/;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header Authorization $http_authorization;
proxy_connect_timeout 10s;
proxy_read_timeout 600s;
proxy_buffering off;
}
# LiteLLM UI static assets
location /litellm-asset-prefix/ {
proxy_pass http://litellm_backend/litellm-asset-prefix/;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
# LiteLLM Admin UI via /litellm/ prefix (LAN/internal access on :80)
location /litellm/ {
proxy_pass http://litellm_backend/;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_read_timeout 86400s;
proxy_buffering off;
proxy_redirect http://$host/ /litellm/;
proxy_redirect https://$host/ /litellm/;
}
location /dashboard/ {
proxy_pass http://dashboard_ui/;
proxy_pass $dashboard_ui_url/;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
# LiteLLM redirect target /litellm (no trailing slash) -> add slash back
location = /litellm {
return 301 /litellm/;
}
# LiteLLM static assets (Next.js chunks, CSS, fonts)
location /litellm-asset-prefix/ {
proxy_pass $litellm_backend_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_connect_timeout 10s;
proxy_read_timeout 60s;
}
# LiteLLM admin UI and API proxy — strip /litellm prefix so /litellm/ui/ → /ui/
location /litellm/ {
rewrite ^/litellm(/.*)$ $1 break;
proxy_pass $litellm_backend_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_buffering off;
}
# Auth proxy to Authentik — accepts HTTP from LiteLLM, proxies HTTPS to .11 with SSL verify off
location /application/o/ {
proxy_pass https://192.168.68.11;
proxy_ssl_verify off;
proxy_set_header Host auth.sysloggh.net;
proxy_set_header X-Real-IP $remote_addr;
proxy_connect_timeout 10s;
proxy_read_timeout 30s;
}
# All other requests → 404
location / {
root /opt/inference-harness/dashboard;
try_files $uri $uri/ /index.html;
return 404;
}
location /metrics/ {
proxy_pass http://router_api/metrics/;
proxy_pass $router_api_url/metrics/;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
location /metrics/circuit-breaker {
proxy_pass http://router_api/metrics/circuit-breaker;
proxy_pass $router_api_url/metrics/circuit-breaker;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
location /router/ {
proxy_pass http://router_api/;
proxy_pass $router_api_url/;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
location /health/unified {
proxy_pass http://router_api/health/unified;
proxy_pass $router_api_url/health/unified;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
location /health {
proxy_pass http://router_api/health;
proxy_pass $router_api_url/health;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
}
# ════════════════════════════════════════════════════════════════
# Server :4000 — external LiteLLM entrypoint (cloudflared target)
# Fixes the /ui redirect: forces correct Host + scheme so LiteLLM
# builds external URLs instead of leaking 192.168.68.116:4000.
# LiteLLM itself is now bound to 127.0.0.1 only (see compose).
# ════════════════════════════════════════════════════════════════
server {
listen 4000;
# Canonical external identity — overrides whatever Host cloudflared sends
proxy_set_header Host litellm.sysloggh.net;
proxy_set_header X-Forwarded-Host litellm.sysloggh.net;
proxy_set_header X-Forwarded-Proto https;
proxy_set_header X-Forwarded-Port 443;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Real-IP $remote_addr;
# /ui → /ui/ — absolute redirect (cloudflared rewrites Host to origin IP,
# so a relative return would be absolutized to http://192.168.68.116:4000/).
location = /ui { return 301 https://litellm.sysloggh.net/ui/; }
# LiteLLM Admin UI + WebSocket
location /ui/ {
proxy_pass http://litellm_backend/ui/;
proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_read_timeout 86400s;
proxy_buffering off;
proxy_redirect http://litellm_backend/ https://litellm.sysloggh.net/;
proxy_redirect http://litellm.sysloggh.net:4000/ https://litellm.sysloggh.net/;
}
# UI static assets
location /litellm-asset-prefix/ {
proxy_pass http://litellm_backend/litellm-asset-prefix/;
proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
}
# SSO callback
location /sso/ {
proxy_pass http://litellm_backend/sso/;
proxy_http_version 1.1;
proxy_read_timeout 600s;
}
# API
location /v1/ {
proxy_pass http://litellm_backend/v1/;
proxy_http_version 1.1;
proxy_set_header Authorization $http_authorization;
proxy_read_timeout 600s;
proxy_buffering off;
}
# Key management + admin API
location /key/ {
proxy_pass http://litellm_backend/key/;
proxy_http_version 1.1;
proxy_set_header Authorization $http_authorization;
}
location /user/ {
proxy_pass http://litellm_backend/user/;
proxy_http_version 1.1;
proxy_set_header Authorization $http_authorization;
}
location /model/ {
proxy_pass http://litellm_backend/model/;
proxy_http_version 1.1;
proxy_set_header Authorization $http_authorization;
}
location /team/ {
proxy_pass http://litellm_backend/team/;
proxy_http_version 1.1;
proxy_set_header Authorization $http_authorization;
}
# Health
location /health {
proxy_pass http://litellm_backend/health;
proxy_http_version 1.1;
}
# Root + everything else → LiteLLM (UI index, /configs, /spend, etc.)
location / {
proxy_pass http://litellm_backend/;
proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_read_timeout 86400s;
proxy_buffering off;
proxy_redirect http://192.168.68.116:4000/ https://litellm.sysloggh.net/;
proxy_redirect https://192.168.68.116:4000/ https://litellm.sysloggh.net/;
}
}
}
+2 -2
View File
@@ -70,7 +70,7 @@ GPU_LABELS = {
}
GPU_MAX_CONCURRENT = {
"qwen3.6-35B-A3B": 1, # 2 slots (cross-agent spread prevents overheating)
"qwen3.6-35B-A3B": 2, # 2 slots (cross-agent spread prevents overheating)
"qwen3.6-27B-code": 2, # 2 slots (128K context frees VRAM)
"gemma-4-12b": 2, # 2 slots (7.1GB VRAM)
}
@@ -626,7 +626,7 @@ def chat():
except Exception: pass
start = time.time()
resp = requests.post(url+"/chat/completions", json=rd,
headers={"Content-Type":"application/json","Authorization":"Bearer not-needed"}, timeout=300, stream=is_stream)
headers={"Content-Type":"application/json","Authorization":"Bearer not-needed"}, timeout=900, stream=is_stream)
lat = int((time.time()-start)*1000)
gpu_release_slot(model)