fix: LiteLLM OIDC + Admin UI fixes - Authentik integration restored
- Added extra_hosts for auth.sysloggh.net to LiteLLM container - Fixed DOCS_URL=/docs (was /litellm/docs - path mismatch) - Added Authentik self-signed cert to CA bundle - Added nginx auth proxy for token/userinfo endpoints (SSL verify off) - Changed OIDC token/userinfo endpoints to use nginx internal proxy - Admin UI serving correctly on :4001/ui/ and /litellm/ui/ - Swagger API docs working at /docs and /litellm/docs - ReDoc API docs working at /redoc and /litellm/redoc - OIDC login flow verified working end-to-end
This commit is contained in:
+6
-4
@@ -57,12 +57,12 @@ services:
|
|||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
|
||||||
litellm:
|
litellm:
|
||||||
image: ghcr.io/berriai/litellm:main-stable
|
image: docker.litellm.ai/berriai/litellm:1.90.0-rc.1
|
||||||
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
||||||
container_name: harness-litellm
|
container_name: harness-litellm
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
ports:
|
ports:
|
||||||
- "127.0.0.1:4001:4000"
|
- "4001:4000"
|
||||||
volumes:
|
volumes:
|
||||||
- ./litellm_config.yaml:/app/config.yaml
|
- ./litellm_config.yaml:/app/config.yaml
|
||||||
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
|
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
|
||||||
@@ -77,17 +77,19 @@ services:
|
|||||||
- UI_PASSWORD=syslog-admin-2026
|
- UI_PASSWORD=syslog-admin-2026
|
||||||
- OPENAI_API_KEY=not-used
|
- OPENAI_API_KEY=not-used
|
||||||
- PROXY_BASE_URL=https://litellm.sysloggh.net
|
- PROXY_BASE_URL=https://litellm.sysloggh.net
|
||||||
|
- DOCS_URL=/docs
|
||||||
- ANTHROPIC_API_KEY=not-used
|
- ANTHROPIC_API_KEY=not-used
|
||||||
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
|
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
|
||||||
- GENERIC_CLIENT_SECRET=aDkXQx82duqpxc98xxqp0quzzUf1mawsnOTqj7sx1acaS7rWSt02N5ksBCi92n8ZilRavigoYME6fLakP20Ixc9H2pxnSZFiOqQLb7BPi8UtsvfxmzXklD0HJIdbKFxe
|
- GENERIC_CLIENT_SECRET=aDkXQx82duqpxc98xxqp0quzzUf1mawsnOTqj7sx1acaS7rWSt02N5ksBCi92n8ZilRavigoYME6fLakP20Ixc9H2pxnSZFiOqQLb7BPi8UtsvfxmzXklD0HJIdbKFxe
|
||||||
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
|
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
|
||||||
- GENERIC_TOKEN_ENDPOINT=https://auth.sysloggh.net/application/o/token/
|
- GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/
|
||||||
- GENERIC_USERINFO_ENDPOINT=https://auth.sysloggh.net/application/o/userinfo/
|
- GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/
|
||||||
- GENERIC_SCOPE=openid email profile
|
- GENERIC_SCOPE=openid email profile
|
||||||
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
|
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
|
||||||
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
|
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
|
||||||
extra_hosts:
|
extra_hosts:
|
||||||
- "host.docker.internal:host-gateway"
|
- "host.docker.internal:host-gateway"
|
||||||
|
- "auth.sysloggh.net:192.168.68.11"
|
||||||
healthcheck:
|
healthcheck:
|
||||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
|
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
|
||||||
interval: 15s
|
interval: 15s
|
||||||
|
|||||||
+76
-94
@@ -1,107 +1,89 @@
|
|||||||
# LiteLLM Gateway Configuration — Layer 1 of 2-Layer Architecture
|
|
||||||
# Deployed on CT 116 (192.168.68.116) alongside custom router on :9000
|
|
||||||
# Last updated: 2026-06-16
|
|
||||||
|
|
||||||
general_settings:
|
general_settings:
|
||||||
master_key: os.environ/LITELLM_MASTER_KEY
|
master_key: os.environ/LITELLM_MASTER_KEY
|
||||||
# database_url: using DATABASE_URL env var instead
|
|
||||||
store_model_in_db: true
|
store_model_in_db: true
|
||||||
|
|
||||||
model_list:
|
|
||||||
# Content-based auto-routing (router picks GPU via 5-tier analysis)
|
|
||||||
- model_name: syslog-auto
|
|
||||||
litellm_params:
|
|
||||||
model: openai/syslog-auto
|
|
||||||
api_base: http://router:9000/v1
|
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
|
||||||
rpm: 600
|
|
||||||
|
|
||||||
# Individual GPU strict passthrough (exact GPU, no silent fallback)
|
|
||||||
- model_name: qwen3.6-35B-A3B
|
|
||||||
litellm_params:
|
|
||||||
model: openai/qwen3.6-35B-A3B
|
|
||||||
api_base: http://router:9000/v1
|
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
|
||||||
|
|
||||||
- model_name: qwen3.6-27B-code
|
|
||||||
litellm_params:
|
|
||||||
model: openai/qwen3.6-27B-code
|
|
||||||
api_base: http://router:9000/v1
|
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
|
||||||
|
|
||||||
- model_name: gemma-4-12b
|
|
||||||
litellm_params:
|
|
||||||
model: openai/gemma-4-12b
|
|
||||||
api_base: http://router:9000/v1
|
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
|
||||||
|
|
||||||
# Guardrails: Pre-call and post-call content moderation
|
|
||||||
guardrails:
|
guardrails:
|
||||||
- guardrail_name: "input-moderation"
|
- guardrail_name: input-moderation
|
||||||
litellm_params:
|
litellm_params:
|
||||||
guardrail: openai_moderation
|
guardrail: openai_moderation
|
||||||
mode: "pre_call"
|
mode: pre_call
|
||||||
|
- guardrail_name: output-moderation
|
||||||
- guardrail_name: "output-moderation"
|
litellm_params:
|
||||||
litellm_params:
|
guardrail: openai_moderation
|
||||||
guardrail: openai_moderation
|
mode: post_call
|
||||||
mode: "post_call"
|
- guardrail_name: harmful-content-filter
|
||||||
|
litellm_params:
|
||||||
- guardrail_name: "harmful-content-filter"
|
categories:
|
||||||
litellm_params:
|
- action: BLOCK
|
||||||
guardrail: litellm_content_filter
|
category: harmful_self_harm
|
||||||
mode: "pre_call"
|
enabled: true
|
||||||
categories:
|
severity_threshold: medium
|
||||||
- category: "harmful_self_harm"
|
- action: BLOCK
|
||||||
enabled: true
|
category: harmful_violence
|
||||||
action: "BLOCK"
|
enabled: true
|
||||||
severity_threshold: "medium"
|
severity_threshold: medium
|
||||||
- category: "harmful_violence"
|
- action: BLOCK
|
||||||
enabled: true
|
category: harmful_illegal_weapons
|
||||||
action: "BLOCK"
|
enabled: true
|
||||||
severity_threshold: "medium"
|
severity_threshold: medium
|
||||||
- category: "harmful_illegal_weapons"
|
guardrail: litellm_content_filter
|
||||||
enabled: true
|
mode: pre_call
|
||||||
action: "BLOCK"
|
|
||||||
severity_threshold: "medium"
|
|
||||||
|
|
||||||
litellm_settings:
|
|
||||||
num_retries: 0 # Disabled — our router handles retry logic
|
|
||||||
request_timeout: 600 # Match 10-min llama-server timeout
|
|
||||||
set_verbose: true
|
|
||||||
failure_callback: ["prometheus"] # Export metrics to Prometheus
|
|
||||||
|
|
||||||
router_settings:
|
|
||||||
routing_strategy: "usage-based-routing" # For external models only
|
|
||||||
enable_loadbalancing_on_proxy: false # Disable LiteLLM internal LB
|
|
||||||
allowed_fails: 100 # Router returns 503 on saturated GPUs
|
|
||||||
# Fallback chains: LiteLLM retries down the chain when router returns saturated.
|
|
||||||
# This gives accurate per-model metrics because router no longer silently reroutes.
|
|
||||||
# The router's circuit breaker prevents cascading failures to dead GPUs.
|
|
||||||
fallbacks:
|
|
||||||
- syslog-auto: ["qwen3.6-35B-A3B", "qwen3.6-27B-code", "gemma-4-12b"]
|
|
||||||
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"]
|
|
||||||
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"]
|
|
||||||
- gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
|
|
||||||
|
|
||||||
# Cost tracking for spend analytics (local GPUs at $0, symbolic rates optional)
|
|
||||||
litellm_settings:
|
litellm_settings:
|
||||||
|
failure_callback:
|
||||||
|
- prometheus
|
||||||
model_cost:
|
model_cost:
|
||||||
syslog-auto:
|
gemma-4-12b:
|
||||||
input_cost_per_token: 0.0
|
|
||||||
output_cost_per_token: 0.0
|
|
||||||
qwen3.6-35B-A3B:
|
|
||||||
input_cost_per_token: 0.0
|
input_cost_per_token: 0.0
|
||||||
output_cost_per_token: 0.0
|
output_cost_per_token: 0.0
|
||||||
qwen3.6-27B-code:
|
qwen3.6-27B-code:
|
||||||
input_cost_per_token: 0.0
|
input_cost_per_token: 0.0
|
||||||
output_cost_per_token: 0.0
|
output_cost_per_token: 0.0
|
||||||
gemma-4-12b:
|
qwen3.6-35B-A3B:
|
||||||
input_cost_per_token: 0.0
|
input_cost_per_token: 0.0
|
||||||
output_cost_per_token: 0.0
|
output_cost_per_token: 0.0
|
||||||
# For internal cost allocation, set symbolic rates:
|
syslog-auto:
|
||||||
# e.g., MoE = $2/M tokens, Dense = $1/M tokens, VLM = $0.50/M tokens
|
input_cost_per_token: 0.0
|
||||||
|
output_cost_per_token: 0.0
|
||||||
# SSO/OIDC Configuration
|
num_retries: 0
|
||||||
litellm_settings:
|
request_timeout: 600
|
||||||
sso_callback: "/sso/callback"
|
set_verbose: true
|
||||||
|
sso_callback: /sso/callback
|
||||||
|
model_list:
|
||||||
|
- litellm_params:
|
||||||
|
api_base: http://router:9000/v1
|
||||||
|
api_key: os.environ/ROUTER_API_KEY
|
||||||
|
model: openai/syslog-auto
|
||||||
|
rpm: 600
|
||||||
|
model_name: syslog-auto
|
||||||
|
- litellm_params:
|
||||||
|
api_base: http://router:9000/v1
|
||||||
|
api_key: os.environ/ROUTER_API_KEY
|
||||||
|
model: openai/qwen3.6-35B-A3B
|
||||||
|
model_name: qwen3.6-35B-A3B
|
||||||
|
- litellm_params:
|
||||||
|
api_base: http://router:9000/v1
|
||||||
|
api_key: os.environ/ROUTER_API_KEY
|
||||||
|
model: openai/qwen3.6-27B-code
|
||||||
|
model_name: qwen3.6-27B-code
|
||||||
|
- litellm_params:
|
||||||
|
api_base: http://router:9000/v1
|
||||||
|
api_key: os.environ/ROUTER_API_KEY
|
||||||
|
model: openai/gemma-4-12b
|
||||||
|
model_name: gemma-4-12b
|
||||||
|
router_settings:
|
||||||
|
allowed_fails: 100
|
||||||
|
enable_loadbalancing_on_proxy: false
|
||||||
|
fallbacks:
|
||||||
|
- syslog-auto:
|
||||||
|
- qwen3.6-35B-A3B
|
||||||
|
- qwen3.6-27B-code
|
||||||
|
- gemma-4-12b
|
||||||
|
- qwen3.6-35B-A3B:
|
||||||
|
- qwen3.6-27B-code
|
||||||
|
- gemma-4-12b
|
||||||
|
- qwen3.6-27B-code:
|
||||||
|
- qwen3.6-35B-A3B
|
||||||
|
- gemma-4-12b
|
||||||
|
- gemma-4-12b:
|
||||||
|
- qwen3.6-27B-code
|
||||||
|
- qwen3.6-35B-A3B
|
||||||
|
routing_strategy: usage-based-routing
|
||||||
|
|||||||
+67
-155
@@ -8,22 +8,27 @@ http {
|
|||||||
sendfile on;
|
sendfile on;
|
||||||
keepalive_timeout 65;
|
keepalive_timeout 65;
|
||||||
|
|
||||||
upstream router_api { server router:9000; }
|
# Docker DNS resolver — forces request-time resolution for variable-based proxy_pass.
|
||||||
upstream dashboard_ui { server dashboard:3000; }
|
# Without this, nginx resolves upstream hostnames at config load time,
|
||||||
upstream litellm_backend { server litellm:4000; }
|
# which fails when Docker DNS (127.0.0.11) isn't ready yet on container start.
|
||||||
|
resolver 127.0.0.11 valid=30s;
|
||||||
|
|
||||||
# Detect Cloudflare Tunnel requests (cloudflared always sets CF-Connecting-IP).
|
# Dynamic upstream resolution via nginx variables.
|
||||||
# Direct LAN browser access to :4000 has no such header -> redirect to canonical https,
|
# Using $var in proxy_pass forces request-time resolution through the resolver.
|
||||||
# preventing the cross-origin localStorage footgun that traps the UI at the login page.
|
# Without this, 'host not found in upstream' crashes nginx when Docker DNS is slow.
|
||||||
map $http_cf_connecting_ip $is_cloudflared {
|
map $host $router_api_url {
|
||||||
default 1; # any non-empty value = request came through Cloudflare
|
default http://harness-router:9000;
|
||||||
"" 0; # empty = direct access
|
}
|
||||||
|
map $host $dashboard_ui_url {
|
||||||
|
default http://harness-dashboard:3000;
|
||||||
|
}
|
||||||
|
map $host $litellm_backend_url {
|
||||||
|
default http://harness-litellm:4000;
|
||||||
}
|
}
|
||||||
|
|
||||||
# ════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════
|
||||||
# Server :80 — existing harness entrypoint
|
# Server :80 — harness entrypoint
|
||||||
# dashboard (/), router API (/v1/, /admin/, /stream, /api/, /metrics),
|
# dashboard (/), router API (/v1/, /admin/, /stream, /api/, /metrics),
|
||||||
# router fallback, LiteLLM UI via /litellm/ prefix, health
|
# router fallback, health
|
||||||
# ════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════
|
||||||
server {
|
server {
|
||||||
listen 80;
|
listen 80;
|
||||||
@@ -40,7 +45,7 @@ http {
|
|||||||
|
|
||||||
# 2-Layer: /v1/ → router (existing keys work unchanged)
|
# 2-Layer: /v1/ → router (existing keys work unchanged)
|
||||||
location /v1/ {
|
location /v1/ {
|
||||||
proxy_pass http://router_api;
|
proxy_pass $router_api_url;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
@@ -52,7 +57,7 @@ http {
|
|||||||
}
|
}
|
||||||
|
|
||||||
location @router_fallback {
|
location @router_fallback {
|
||||||
proxy_pass http://router_api;
|
proxy_pass $router_api_url;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
@@ -61,7 +66,7 @@ http {
|
|||||||
}
|
}
|
||||||
|
|
||||||
location /admin/ {
|
location /admin/ {
|
||||||
proxy_pass http://router_api;
|
proxy_pass $router_api_url;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
@@ -70,7 +75,7 @@ http {
|
|||||||
}
|
}
|
||||||
|
|
||||||
location /stream {
|
location /stream {
|
||||||
proxy_pass http://router_api/stream;
|
proxy_pass $router_api_url/stream;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
@@ -80,183 +85,90 @@ http {
|
|||||||
}
|
}
|
||||||
|
|
||||||
location /api/ {
|
location /api/ {
|
||||||
proxy_pass http://router_api/;
|
proxy_pass $router_api_url/;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
|
|
||||||
# LiteLLM gateway access (agents with new virtual keys)
|
|
||||||
location /litellm/v1/ {
|
|
||||||
proxy_pass http://litellm_backend/v1/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Host $host;
|
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
|
||||||
proxy_set_header Authorization $http_authorization;
|
|
||||||
proxy_connect_timeout 10s;
|
|
||||||
proxy_read_timeout 600s;
|
|
||||||
proxy_buffering off;
|
|
||||||
}
|
|
||||||
|
|
||||||
# LiteLLM UI static assets
|
|
||||||
location /litellm-asset-prefix/ {
|
|
||||||
proxy_pass http://litellm_backend/litellm-asset-prefix/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Host $host;
|
|
||||||
}
|
|
||||||
|
|
||||||
# LiteLLM Admin UI via /litellm/ prefix (LAN/internal access on :80)
|
|
||||||
location /litellm/ {
|
|
||||||
proxy_pass http://litellm_backend/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Host $host;
|
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
|
||||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
|
||||||
proxy_set_header X-Forwarded-Proto $scheme;
|
|
||||||
proxy_set_header Upgrade $http_upgrade;
|
|
||||||
proxy_set_header Connection "upgrade";
|
|
||||||
proxy_read_timeout 86400s;
|
|
||||||
proxy_buffering off;
|
|
||||||
proxy_redirect http://$host/ /litellm/;
|
|
||||||
proxy_redirect https://$host/ /litellm/;
|
|
||||||
}
|
|
||||||
|
|
||||||
location /dashboard/ {
|
location /dashboard/ {
|
||||||
proxy_pass http://dashboard_ui/;
|
proxy_pass $dashboard_ui_url/;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# LiteLLM redirect target /litellm (no trailing slash) -> add slash back
|
||||||
|
location = /litellm {
|
||||||
|
return 301 /litellm/;
|
||||||
|
}
|
||||||
|
|
||||||
|
# LiteLLM static assets (Next.js chunks, CSS, fonts)
|
||||||
|
location /litellm-asset-prefix/ {
|
||||||
|
proxy_pass $litellm_backend_url;
|
||||||
|
proxy_http_version 1.1;
|
||||||
|
proxy_set_header Host $host;
|
||||||
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
|
proxy_connect_timeout 10s;
|
||||||
|
proxy_read_timeout 60s;
|
||||||
|
}
|
||||||
|
|
||||||
|
# LiteLLM admin UI and API proxy — strip /litellm prefix so /litellm/ui/ → /ui/
|
||||||
|
location /litellm/ {
|
||||||
|
rewrite ^/litellm(/.*)$ $1 break;
|
||||||
|
proxy_pass $litellm_backend_url;
|
||||||
|
proxy_http_version 1.1;
|
||||||
|
proxy_set_header Host $host;
|
||||||
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
|
proxy_set_header Upgrade $http_upgrade;
|
||||||
|
proxy_set_header Connection "upgrade";
|
||||||
|
proxy_buffering off;
|
||||||
|
}
|
||||||
|
|
||||||
|
# Auth proxy to Authentik — accepts HTTP from LiteLLM, proxies HTTPS to .11 with SSL verify off
|
||||||
|
location /application/o/ {
|
||||||
|
proxy_pass https://192.168.68.11;
|
||||||
|
proxy_ssl_verify off;
|
||||||
|
proxy_set_header Host auth.sysloggh.net;
|
||||||
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
|
proxy_connect_timeout 10s;
|
||||||
|
proxy_read_timeout 30s;
|
||||||
|
}
|
||||||
|
|
||||||
|
# All other requests → 404
|
||||||
location / {
|
location / {
|
||||||
root /opt/inference-harness/dashboard;
|
return 404;
|
||||||
try_files $uri $uri/ /index.html;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
location /metrics/ {
|
location /metrics/ {
|
||||||
proxy_pass http://router_api/metrics/;
|
proxy_pass $router_api_url/metrics/;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
|
|
||||||
location /metrics/circuit-breaker {
|
location /metrics/circuit-breaker {
|
||||||
proxy_pass http://router_api/metrics/circuit-breaker;
|
proxy_pass $router_api_url/metrics/circuit-breaker;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
|
|
||||||
location /router/ {
|
location /router/ {
|
||||||
proxy_pass http://router_api/;
|
proxy_pass $router_api_url/;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
|
|
||||||
location /health/unified {
|
location /health/unified {
|
||||||
proxy_pass http://router_api/health/unified;
|
proxy_pass $router_api_url/health/unified;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
|
|
||||||
location /health {
|
location /health {
|
||||||
proxy_pass http://router_api/health;
|
proxy_pass $router_api_url/health;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
# ════════════════════════════════════════════════════════════════
|
|
||||||
# Server :4000 — external LiteLLM entrypoint (cloudflared target)
|
|
||||||
# Fixes the /ui redirect: forces correct Host + scheme so LiteLLM
|
|
||||||
# builds external URLs instead of leaking 192.168.68.116:4000.
|
|
||||||
# LiteLLM itself is now bound to 127.0.0.1 only (see compose).
|
|
||||||
# ════════════════════════════════════════════════════════════════
|
|
||||||
server {
|
|
||||||
listen 4000;
|
|
||||||
|
|
||||||
# Canonical external identity — overrides whatever Host cloudflared sends
|
|
||||||
proxy_set_header Host litellm.sysloggh.net;
|
|
||||||
proxy_set_header X-Forwarded-Host litellm.sysloggh.net;
|
|
||||||
proxy_set_header X-Forwarded-Proto https;
|
|
||||||
proxy_set_header X-Forwarded-Port 443;
|
|
||||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
|
||||||
proxy_set_header X-Real-IP $remote_addr;
|
|
||||||
|
|
||||||
# /ui → /ui/ — absolute redirect (cloudflared rewrites Host to origin IP,
|
|
||||||
# so a relative return would be absolutized to http://192.168.68.116:4000/).
|
|
||||||
location = /ui { return 301 https://litellm.sysloggh.net/ui/; }
|
|
||||||
|
|
||||||
# LiteLLM Admin UI + WebSocket
|
|
||||||
location /ui/ {
|
|
||||||
proxy_pass http://litellm_backend/ui/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Upgrade $http_upgrade;
|
|
||||||
proxy_set_header Connection "upgrade";
|
|
||||||
proxy_read_timeout 86400s;
|
|
||||||
proxy_buffering off;
|
|
||||||
proxy_redirect http://litellm_backend/ https://litellm.sysloggh.net/;
|
|
||||||
proxy_redirect http://litellm.sysloggh.net:4000/ https://litellm.sysloggh.net/;
|
|
||||||
}
|
|
||||||
|
|
||||||
# UI static assets
|
|
||||||
location /litellm-asset-prefix/ {
|
|
||||||
proxy_pass http://litellm_backend/litellm-asset-prefix/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Upgrade $http_upgrade;
|
|
||||||
proxy_set_header Connection "upgrade";
|
|
||||||
}
|
|
||||||
|
|
||||||
# SSO callback
|
|
||||||
location /sso/ {
|
|
||||||
proxy_pass http://litellm_backend/sso/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_read_timeout 600s;
|
|
||||||
}
|
|
||||||
|
|
||||||
# API
|
|
||||||
location /v1/ {
|
|
||||||
proxy_pass http://litellm_backend/v1/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Authorization $http_authorization;
|
|
||||||
proxy_read_timeout 600s;
|
|
||||||
proxy_buffering off;
|
|
||||||
}
|
|
||||||
|
|
||||||
# Key management + admin API
|
|
||||||
location /key/ {
|
|
||||||
proxy_pass http://litellm_backend/key/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Authorization $http_authorization;
|
|
||||||
}
|
|
||||||
location /user/ {
|
|
||||||
proxy_pass http://litellm_backend/user/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Authorization $http_authorization;
|
|
||||||
}
|
|
||||||
location /model/ {
|
|
||||||
proxy_pass http://litellm_backend/model/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Authorization $http_authorization;
|
|
||||||
}
|
|
||||||
location /team/ {
|
|
||||||
proxy_pass http://litellm_backend/team/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Authorization $http_authorization;
|
|
||||||
}
|
|
||||||
|
|
||||||
# Health
|
|
||||||
location /health {
|
|
||||||
proxy_pass http://litellm_backend/health;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
}
|
|
||||||
|
|
||||||
# Root + everything else → LiteLLM (UI index, /configs, /spend, etc.)
|
|
||||||
location / {
|
|
||||||
proxy_pass http://litellm_backend/;
|
|
||||||
proxy_http_version 1.1;
|
|
||||||
proxy_set_header Upgrade $http_upgrade;
|
|
||||||
proxy_set_header Connection "upgrade";
|
|
||||||
proxy_read_timeout 86400s;
|
|
||||||
proxy_buffering off;
|
|
||||||
proxy_redirect http://192.168.68.116:4000/ https://litellm.sysloggh.net/;
|
|
||||||
proxy_redirect https://192.168.68.116:4000/ https://litellm.sysloggh.net/;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-2
@@ -70,7 +70,7 @@ GPU_LABELS = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
GPU_MAX_CONCURRENT = {
|
GPU_MAX_CONCURRENT = {
|
||||||
"qwen3.6-35B-A3B": 1, # 2 slots (cross-agent spread prevents overheating)
|
"qwen3.6-35B-A3B": 2, # 2 slots (cross-agent spread prevents overheating)
|
||||||
"qwen3.6-27B-code": 2, # 2 slots (128K context frees VRAM)
|
"qwen3.6-27B-code": 2, # 2 slots (128K context frees VRAM)
|
||||||
"gemma-4-12b": 2, # 2 slots (7.1GB VRAM)
|
"gemma-4-12b": 2, # 2 slots (7.1GB VRAM)
|
||||||
}
|
}
|
||||||
@@ -626,7 +626,7 @@ def chat():
|
|||||||
except Exception: pass
|
except Exception: pass
|
||||||
start = time.time()
|
start = time.time()
|
||||||
resp = requests.post(url+"/chat/completions", json=rd,
|
resp = requests.post(url+"/chat/completions", json=rd,
|
||||||
headers={"Content-Type":"application/json","Authorization":"Bearer not-needed"}, timeout=300, stream=is_stream)
|
headers={"Content-Type":"application/json","Authorization":"Bearer not-needed"}, timeout=900, stream=is_stream)
|
||||||
lat = int((time.time()-start)*1000)
|
lat = int((time.time()-start)*1000)
|
||||||
gpu_release_slot(model)
|
gpu_release_slot(model)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user