diff --git a/docker-compose.yml b/docker-compose.yml index 517e3c1..96278eb 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -57,12 +57,12 @@ services: condition: service_healthy litellm: - image: ghcr.io/berriai/litellm:main-stable + image: docker.litellm.ai/berriai/litellm:1.90.0-rc.1 command: ["--config", "/app/config.yaml", "--port", "4000"] container_name: harness-litellm restart: unless-stopped ports: - - "127.0.0.1:4001:4000" + - "4001:4000" volumes: - ./litellm_config.yaml:/app/config.yaml - /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro @@ -77,17 +77,19 @@ services: - UI_PASSWORD=syslog-admin-2026 - OPENAI_API_KEY=not-used - PROXY_BASE_URL=https://litellm.sysloggh.net + - DOCS_URL=/docs - ANTHROPIC_API_KEY=not-used - GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU - GENERIC_CLIENT_SECRET=aDkXQx82duqpxc98xxqp0quzzUf1mawsnOTqj7sx1acaS7rWSt02N5ksBCi92n8ZilRavigoYME6fLakP20Ixc9H2pxnSZFiOqQLb7BPi8UtsvfxmzXklD0HJIdbKFxe - GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/ - - GENERIC_TOKEN_ENDPOINT=https://auth.sysloggh.net/application/o/token/ - - GENERIC_USERINFO_ENDPOINT=https://auth.sysloggh.net/application/o/userinfo/ + - GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/ + - GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/ - GENERIC_SCOPE=openid email profile - GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin - SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt extra_hosts: - "host.docker.internal:host-gateway" + - "auth.sysloggh.net:192.168.68.11" healthcheck: test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"] interval: 15s diff --git a/litellm_config.yaml b/litellm_config.yaml index 6cf03d8..361b9c6 100644 --- a/litellm_config.yaml +++ b/litellm_config.yaml @@ -1,107 +1,89 @@ -# LiteLLM Gateway Configuration — Layer 1 of 2-Layer Architecture -# Deployed on CT 116 (192.168.68.116) alongside custom router on :9000 -# Last updated: 2026-06-16 - general_settings: master_key: os.environ/LITELLM_MASTER_KEY - # database_url: using DATABASE_URL env var instead store_model_in_db: true - -model_list: - # Content-based auto-routing (router picks GPU via 5-tier analysis) - - model_name: syslog-auto - litellm_params: - model: openai/syslog-auto - api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY - rpm: 600 - - # Individual GPU strict passthrough (exact GPU, no silent fallback) - - model_name: qwen3.6-35B-A3B - litellm_params: - model: openai/qwen3.6-35B-A3B - api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY - - - model_name: qwen3.6-27B-code - litellm_params: - model: openai/qwen3.6-27B-code - api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY - - - model_name: gemma-4-12b - litellm_params: - model: openai/gemma-4-12b - api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY - -# Guardrails: Pre-call and post-call content moderation guardrails: - - guardrail_name: "input-moderation" - litellm_params: - guardrail: openai_moderation - mode: "pre_call" - - - guardrail_name: "output-moderation" - litellm_params: - guardrail: openai_moderation - mode: "post_call" - - - guardrail_name: "harmful-content-filter" - litellm_params: - guardrail: litellm_content_filter - mode: "pre_call" - categories: - - category: "harmful_self_harm" - enabled: true - action: "BLOCK" - severity_threshold: "medium" - - category: "harmful_violence" - enabled: true - action: "BLOCK" - severity_threshold: "medium" - - category: "harmful_illegal_weapons" - enabled: true - action: "BLOCK" - severity_threshold: "medium" - -litellm_settings: - num_retries: 0 # Disabled — our router handles retry logic - request_timeout: 600 # Match 10-min llama-server timeout - set_verbose: true - failure_callback: ["prometheus"] # Export metrics to Prometheus - -router_settings: - routing_strategy: "usage-based-routing" # For external models only - enable_loadbalancing_on_proxy: false # Disable LiteLLM internal LB - allowed_fails: 100 # Router returns 503 on saturated GPUs - # Fallback chains: LiteLLM retries down the chain when router returns saturated. - # This gives accurate per-model metrics because router no longer silently reroutes. - # The router's circuit breaker prevents cascading failures to dead GPUs. - fallbacks: - - syslog-auto: ["qwen3.6-35B-A3B", "qwen3.6-27B-code", "gemma-4-12b"] - - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"] - - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"] - - gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] - -# Cost tracking for spend analytics (local GPUs at $0, symbolic rates optional) +- guardrail_name: input-moderation + litellm_params: + guardrail: openai_moderation + mode: pre_call +- guardrail_name: output-moderation + litellm_params: + guardrail: openai_moderation + mode: post_call +- guardrail_name: harmful-content-filter + litellm_params: + categories: + - action: BLOCK + category: harmful_self_harm + enabled: true + severity_threshold: medium + - action: BLOCK + category: harmful_violence + enabled: true + severity_threshold: medium + - action: BLOCK + category: harmful_illegal_weapons + enabled: true + severity_threshold: medium + guardrail: litellm_content_filter + mode: pre_call litellm_settings: + failure_callback: + - prometheus model_cost: - syslog-auto: - input_cost_per_token: 0.0 - output_cost_per_token: 0.0 - qwen3.6-35B-A3B: + gemma-4-12b: input_cost_per_token: 0.0 output_cost_per_token: 0.0 qwen3.6-27B-code: input_cost_per_token: 0.0 output_cost_per_token: 0.0 - gemma-4-12b: + qwen3.6-35B-A3B: input_cost_per_token: 0.0 output_cost_per_token: 0.0 - # For internal cost allocation, set symbolic rates: - # e.g., MoE = $2/M tokens, Dense = $1/M tokens, VLM = $0.50/M tokens - -# SSO/OIDC Configuration -litellm_settings: - sso_callback: "/sso/callback" + syslog-auto: + input_cost_per_token: 0.0 + output_cost_per_token: 0.0 + num_retries: 0 + request_timeout: 600 + set_verbose: true + sso_callback: /sso/callback +model_list: +- litellm_params: + api_base: http://router:9000/v1 + api_key: os.environ/ROUTER_API_KEY + model: openai/syslog-auto + rpm: 600 + model_name: syslog-auto +- litellm_params: + api_base: http://router:9000/v1 + api_key: os.environ/ROUTER_API_KEY + model: openai/qwen3.6-35B-A3B + model_name: qwen3.6-35B-A3B +- litellm_params: + api_base: http://router:9000/v1 + api_key: os.environ/ROUTER_API_KEY + model: openai/qwen3.6-27B-code + model_name: qwen3.6-27B-code +- litellm_params: + api_base: http://router:9000/v1 + api_key: os.environ/ROUTER_API_KEY + model: openai/gemma-4-12b + model_name: gemma-4-12b +router_settings: + allowed_fails: 100 + enable_loadbalancing_on_proxy: false + fallbacks: + - syslog-auto: + - qwen3.6-35B-A3B + - qwen3.6-27B-code + - gemma-4-12b + - qwen3.6-35B-A3B: + - qwen3.6-27B-code + - gemma-4-12b + - qwen3.6-27B-code: + - qwen3.6-35B-A3B + - gemma-4-12b + - gemma-4-12b: + - qwen3.6-27B-code + - qwen3.6-35B-A3B + routing_strategy: usage-based-routing diff --git a/nginx/nginx.conf b/nginx/nginx.conf index 1338f75..a6256e1 100644 --- a/nginx/nginx.conf +++ b/nginx/nginx.conf @@ -8,22 +8,27 @@ http { sendfile on; keepalive_timeout 65; - upstream router_api { server router:9000; } - upstream dashboard_ui { server dashboard:3000; } - upstream litellm_backend { server litellm:4000; } + # Docker DNS resolver — forces request-time resolution for variable-based proxy_pass. + # Without this, nginx resolves upstream hostnames at config load time, + # which fails when Docker DNS (127.0.0.11) isn't ready yet on container start. + resolver 127.0.0.11 valid=30s; - # Detect Cloudflare Tunnel requests (cloudflared always sets CF-Connecting-IP). - # Direct LAN browser access to :4000 has no such header -> redirect to canonical https, - # preventing the cross-origin localStorage footgun that traps the UI at the login page. - map $http_cf_connecting_ip $is_cloudflared { - default 1; # any non-empty value = request came through Cloudflare - "" 0; # empty = direct access + # Dynamic upstream resolution via nginx variables. + # Using $var in proxy_pass forces request-time resolution through the resolver. + # Without this, 'host not found in upstream' crashes nginx when Docker DNS is slow. + map $host $router_api_url { + default http://harness-router:9000; + } + map $host $dashboard_ui_url { + default http://harness-dashboard:3000; + } + map $host $litellm_backend_url { + default http://harness-litellm:4000; } - # ════════════════════════════════════════════════════════════════ - # Server :80 — existing harness entrypoint + # Server :80 — harness entrypoint # dashboard (/), router API (/v1/, /admin/, /stream, /api/, /metrics), - # router fallback, LiteLLM UI via /litellm/ prefix, health + # router fallback, health # ════════════════════════════════════════════════════════════════ server { listen 80; @@ -40,7 +45,7 @@ http { # 2-Layer: /v1/ → router (existing keys work unchanged) location /v1/ { - proxy_pass http://router_api; + proxy_pass $router_api_url; proxy_http_version 1.1; proxy_set_header Host $host; proxy_set_header X-Real-IP $remote_addr; @@ -52,7 +57,7 @@ http { } location @router_fallback { - proxy_pass http://router_api; + proxy_pass $router_api_url; proxy_http_version 1.1; proxy_set_header Host $host; proxy_set_header X-Real-IP $remote_addr; @@ -61,7 +66,7 @@ http { } location /admin/ { - proxy_pass http://router_api; + proxy_pass $router_api_url; proxy_http_version 1.1; proxy_set_header Host $host; proxy_set_header X-Real-IP $remote_addr; @@ -70,7 +75,7 @@ http { } location /stream { - proxy_pass http://router_api/stream; + proxy_pass $router_api_url/stream; proxy_http_version 1.1; proxy_set_header Host $host; proxy_set_header X-Real-IP $remote_addr; @@ -80,183 +85,90 @@ http { } location /api/ { - proxy_pass http://router_api/; + proxy_pass $router_api_url/; proxy_http_version 1.1; proxy_set_header Host $host; } - # LiteLLM gateway access (agents with new virtual keys) - location /litellm/v1/ { - proxy_pass http://litellm_backend/v1/; - proxy_http_version 1.1; - proxy_set_header Host $host; - proxy_set_header X-Real-IP $remote_addr; - proxy_set_header Authorization $http_authorization; - proxy_connect_timeout 10s; - proxy_read_timeout 600s; - proxy_buffering off; - } - # LiteLLM UI static assets - location /litellm-asset-prefix/ { - proxy_pass http://litellm_backend/litellm-asset-prefix/; - proxy_http_version 1.1; - proxy_set_header Host $host; - } - - # LiteLLM Admin UI via /litellm/ prefix (LAN/internal access on :80) - location /litellm/ { - proxy_pass http://litellm_backend/; - proxy_http_version 1.1; - proxy_set_header Host $host; - proxy_set_header X-Real-IP $remote_addr; - proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - proxy_set_header X-Forwarded-Proto $scheme; - proxy_set_header Upgrade $http_upgrade; - proxy_set_header Connection "upgrade"; - proxy_read_timeout 86400s; - proxy_buffering off; - proxy_redirect http://$host/ /litellm/; - proxy_redirect https://$host/ /litellm/; - } location /dashboard/ { - proxy_pass http://dashboard_ui/; + proxy_pass $dashboard_ui_url/; proxy_http_version 1.1; proxy_set_header Host $host; } + # LiteLLM redirect target /litellm (no trailing slash) -> add slash back + location = /litellm { + return 301 /litellm/; + } + + # LiteLLM static assets (Next.js chunks, CSS, fonts) + location /litellm-asset-prefix/ { + proxy_pass $litellm_backend_url; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_connect_timeout 10s; + proxy_read_timeout 60s; + } + + # LiteLLM admin UI and API proxy — strip /litellm prefix so /litellm/ui/ → /ui/ + location /litellm/ { + rewrite ^/litellm(/.*)$ $1 break; + proxy_pass $litellm_backend_url; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection "upgrade"; + proxy_buffering off; + } + + # Auth proxy to Authentik — accepts HTTP from LiteLLM, proxies HTTPS to .11 with SSL verify off + location /application/o/ { + proxy_pass https://192.168.68.11; + proxy_ssl_verify off; + proxy_set_header Host auth.sysloggh.net; + proxy_set_header X-Real-IP $remote_addr; + proxy_connect_timeout 10s; + proxy_read_timeout 30s; + } + + # All other requests → 404 location / { - root /opt/inference-harness/dashboard; - try_files $uri $uri/ /index.html; + return 404; } location /metrics/ { - proxy_pass http://router_api/metrics/; + proxy_pass $router_api_url/metrics/; proxy_http_version 1.1; proxy_set_header Host $host; } location /metrics/circuit-breaker { - proxy_pass http://router_api/metrics/circuit-breaker; + proxy_pass $router_api_url/metrics/circuit-breaker; proxy_http_version 1.1; proxy_set_header Host $host; } location /router/ { - proxy_pass http://router_api/; + proxy_pass $router_api_url/; proxy_http_version 1.1; proxy_set_header Host $host; } location /health/unified { - proxy_pass http://router_api/health/unified; + proxy_pass $router_api_url/health/unified; proxy_http_version 1.1; proxy_set_header Host $host; } location /health { - proxy_pass http://router_api/health; + proxy_pass $router_api_url/health; proxy_http_version 1.1; proxy_set_header Host $host; } } - # ════════════════════════════════════════════════════════════════ - # Server :4000 — external LiteLLM entrypoint (cloudflared target) - # Fixes the /ui redirect: forces correct Host + scheme so LiteLLM - # builds external URLs instead of leaking 192.168.68.116:4000. - # LiteLLM itself is now bound to 127.0.0.1 only (see compose). - # ════════════════════════════════════════════════════════════════ - server { - listen 4000; - - # Canonical external identity — overrides whatever Host cloudflared sends - proxy_set_header Host litellm.sysloggh.net; - proxy_set_header X-Forwarded-Host litellm.sysloggh.net; - proxy_set_header X-Forwarded-Proto https; - proxy_set_header X-Forwarded-Port 443; - proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - proxy_set_header X-Real-IP $remote_addr; - - # /ui → /ui/ — absolute redirect (cloudflared rewrites Host to origin IP, - # so a relative return would be absolutized to http://192.168.68.116:4000/). - location = /ui { return 301 https://litellm.sysloggh.net/ui/; } - - # LiteLLM Admin UI + WebSocket - location /ui/ { - proxy_pass http://litellm_backend/ui/; - proxy_http_version 1.1; - proxy_set_header Upgrade $http_upgrade; - proxy_set_header Connection "upgrade"; - proxy_read_timeout 86400s; - proxy_buffering off; - proxy_redirect http://litellm_backend/ https://litellm.sysloggh.net/; - proxy_redirect http://litellm.sysloggh.net:4000/ https://litellm.sysloggh.net/; - } - - # UI static assets - location /litellm-asset-prefix/ { - proxy_pass http://litellm_backend/litellm-asset-prefix/; - proxy_http_version 1.1; - proxy_set_header Upgrade $http_upgrade; - proxy_set_header Connection "upgrade"; - } - - # SSO callback - location /sso/ { - proxy_pass http://litellm_backend/sso/; - proxy_http_version 1.1; - proxy_read_timeout 600s; - } - - # API - location /v1/ { - proxy_pass http://litellm_backend/v1/; - proxy_http_version 1.1; - proxy_set_header Authorization $http_authorization; - proxy_read_timeout 600s; - proxy_buffering off; - } - - # Key management + admin API - location /key/ { - proxy_pass http://litellm_backend/key/; - proxy_http_version 1.1; - proxy_set_header Authorization $http_authorization; - } - location /user/ { - proxy_pass http://litellm_backend/user/; - proxy_http_version 1.1; - proxy_set_header Authorization $http_authorization; - } - location /model/ { - proxy_pass http://litellm_backend/model/; - proxy_http_version 1.1; - proxy_set_header Authorization $http_authorization; - } - location /team/ { - proxy_pass http://litellm_backend/team/; - proxy_http_version 1.1; - proxy_set_header Authorization $http_authorization; - } - - # Health - location /health { - proxy_pass http://litellm_backend/health; - proxy_http_version 1.1; - } - - # Root + everything else → LiteLLM (UI index, /configs, /spend, etc.) - location / { - proxy_pass http://litellm_backend/; - proxy_http_version 1.1; - proxy_set_header Upgrade $http_upgrade; - proxy_set_header Connection "upgrade"; - proxy_read_timeout 86400s; - proxy_buffering off; - proxy_redirect http://192.168.68.116:4000/ https://litellm.sysloggh.net/; - proxy_redirect https://192.168.68.116:4000/ https://litellm.sysloggh.net/; - } - } } diff --git a/router/router.py b/router/router.py index 94e81bb..c2e3872 100644 --- a/router/router.py +++ b/router/router.py @@ -70,7 +70,7 @@ GPU_LABELS = { } GPU_MAX_CONCURRENT = { - "qwen3.6-35B-A3B": 1, # 2 slots (cross-agent spread prevents overheating) + "qwen3.6-35B-A3B": 2, # 2 slots (cross-agent spread prevents overheating) "qwen3.6-27B-code": 2, # 2 slots (128K context frees VRAM) "gemma-4-12b": 2, # 2 slots (7.1GB VRAM) } @@ -626,7 +626,7 @@ def chat(): except Exception: pass start = time.time() resp = requests.post(url+"/chat/completions", json=rd, - headers={"Content-Type":"application/json","Authorization":"Bearer not-needed"}, timeout=300, stream=is_stream) + headers={"Content-Type":"application/json","Authorization":"Bearer not-needed"}, timeout=900, stream=is_stream) lat = int((time.time()-start)*1000) gpu_release_slot(model)