From e9aa220ee8d15e607e50c18917508d27dc21de3c Mon Sep 17 00:00:00 2001 From: Abiba Date: Sun, 28 Jun 2026 16:24:54 +0000 Subject: [PATCH 1/8] feat(gpu-fleet): roster-driven GPU config + Ornith-1.0-35B --- gpu_roster.yaml | 55 +++++++++++++++++++++++++++++++++ litellm_config.yaml | 25 ++++++--------- roster_loader.py | 75 +++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 140 insertions(+), 15 deletions(-) create mode 100644 gpu_roster.yaml create mode 100644 roster_loader.py diff --git a/gpu_roster.yaml b/gpu_roster.yaml new file mode 100644 index 0000000..4b76fc4 --- /dev/null +++ b/gpu_roster.yaml @@ -0,0 +1,55 @@ +models: + gemma-4-12b: + gpu_url: http://192.168.68.110:8080/v1 + sidecar_url: http://192.168.68.110:8090 + gpu_host: 192.168.68.110 + label: Gemma-4 12B (RTX 5070) + max_concurrent: 2 + context: 262144 + tiers: [starter, professional, enterprise] + capabilities: [completion, multimodal] + model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf + args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384 + + qwen3.6-27B-code: + gpu_url: http://192.168.68.8:8080/v1 + sidecar_url: http://192.168.68.8:8090 + gpu_host: 192.168.68.8 + label: Qwen3.6 27B Code (RTX 3090) + max_concurrent: 2 + context: 262144 + tiers: [professional, enterprise] + capabilities: [completion] + model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf + args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8 + + ornith-1.0-35b: + gpu_url: http://192.168.68.15:8080/v1 + sidecar_url: http://192.168.68.15:8090 + gpu_host: 192.168.68.15 + label: Ornith-1.0 35B (Strix Halo) + max_concurrent: 1 + context: 4096 + tiers: [professional, enterprise] + capabilities: [completion] + model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf + args: -c 262144 -ngl 99 --flash-attn on + +hosts: + gpu-light: + address: 192.168.68.110 + gpu_name: NVIDIA GeForce RTX 5070 + vram_gb: 12 + current_model: gemma-4-12b + + gpu-dense: + address: 192.168.68.8 + gpu_name: NVIDIA GeForce RTX 3090 + vram_gb: 24 + current_model: qwen3.6-27B-code + + gpu-moe: + address: 192.168.68.15 + gpu_name: AMD Strix Halo (iGPU) + vram_gb: 64 + current_model: ornith-1.0-35b diff --git a/litellm_config.yaml b/litellm_config.yaml index 361b9c6..bcab179 100644 --- a/litellm_config.yaml +++ b/litellm_config.yaml @@ -37,7 +37,7 @@ litellm_settings: qwen3.6-27B-code: input_cost_per_token: 0.0 output_cost_per_token: 0.0 - qwen3.6-35B-A3B: + ornith-1.0-35b: input_cost_per_token: 0.0 output_cost_per_token: 0.0 syslog-auto: @@ -50,40 +50,35 @@ litellm_settings: model_list: - litellm_params: api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 model: openai/syslog-auto rpm: 600 model_name: syslog-auto - litellm_params: api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY - model: openai/qwen3.6-35B-A3B - model_name: qwen3.6-35B-A3B -- litellm_params: - api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 model: openai/qwen3.6-27B-code model_name: qwen3.6-27B-code - litellm_params: api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 model: openai/gemma-4-12b model_name: gemma-4-12b + +- litellm_params: + api_base: http://router:9000/v1 + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 + model: openai/ornith-1.0-35b + model_name: ornith-1.0-35b router_settings: allowed_fails: 100 enable_loadbalancing_on_proxy: false fallbacks: - syslog-auto: - - qwen3.6-35B-A3B - - qwen3.6-27B-code - - gemma-4-12b - - qwen3.6-35B-A3B: - qwen3.6-27B-code - gemma-4-12b - qwen3.6-27B-code: - - qwen3.6-35B-A3B - gemma-4-12b - gemma-4-12b: - qwen3.6-27B-code - - qwen3.6-35B-A3B routing_strategy: usage-based-routing diff --git a/roster_loader.py b/roster_loader.py new file mode 100644 index 0000000..f500e66 --- /dev/null +++ b/roster_loader.py @@ -0,0 +1,75 @@ +# GPU Roster Loader - reads gpu_roster.yaml via PyYAML +# Hot-reloadable via /admin/roster/reload endpoint + +import os, json, threading, time + +try: + import yaml +except ImportError: + yaml = None + +ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml') +_last_mtime = 0 +_lock = threading.Lock() + +GPU_URLS = {} +GPU_SIDECARS = {} +GPU_LABELS = {} +GPU_MAX_CONCURRENT = {} +GPU_CONTEXT = {} +TIER_MODELS = {} +HOSTS = {} + +def load_roster(path=None): + global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT + global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime + if yaml is None: + return False, 'PyYAML not installed - run: pip3 install pyyaml' + path = path or ROSTER_PATH + try: + with open(path) as f: + data = yaml.safe_load(f) + with _lock: + GPU_URLS.clear() + GPU_SIDECARS.clear() + GPU_LABELS.clear() + GPU_MAX_CONCURRENT.clear() + GPU_CONTEXT.clear() + TIER_MODELS.clear() + models = data.get('models', {}) + for name, cfg in models.items(): + GPU_URLS[name] = cfg.get('gpu_url', '') + GPU_SIDECARS[name] = cfg.get('sidecar_url', '') + GPU_LABELS[name] = cfg.get('label', name) + GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1) + GPU_CONTEXT[name] = cfg.get('context', 65536) + tiers = cfg.get('tiers', ['enterprise']) + for t in tiers: + if t not in TIER_MODELS: + TIER_MODELS[t] = [] + if name not in TIER_MODELS[t]: + TIER_MODELS[t].append(name) + HOSTS.clear() + HOSTS.update(data.get('hosts', {})) + _last_mtime = os.path.getmtime(path) + return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS)) + except Exception as e: + return False, str(e) + +def check_reload(): + global _last_mtime + try: + mtime = os.path.getmtime(ROSTER_PATH) + if mtime > _last_mtime: + success, msg = load_roster() + if success: + print('[ROSTER] Auto-reloaded:', msg) + except: + pass + +def reload_thread(interval=30): + while True: + time.sleep(interval) + check_reload() + +threading.Thread(target=reload_thread, daemon=True).start() -- 2.54.0 From 765d53762c467bba8cb4ae530c05898e31875d5b Mon Sep 17 00:00:00 2001 From: Abiba Date: Sun, 28 Jun 2026 16:27:04 +0000 Subject: [PATCH 2/8] fix(nginx): forward Authorization to LiteLLM + remove dead GPU_MOE_URL --- docker-compose.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docker-compose.yml b/docker-compose.yml index 96278eb..c519b59 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -42,8 +42,8 @@ services: - "127.0.0.1:9000:9000" environment: - REDIS_URL=redis://redis:6379 - - GPU_MOE_URL=http://192.168.68.15:8080/v1 - GPU_DENSE_URL=http://192.168.68.8:8080/v1 + - GPU_MOE_URL=http://192.168.68.110:8080/v1 - GPU_LIGHT_URL=http://192.168.68.110:8080/v1 - API_KEYS={"sk-syslog-local-master-key":{"tier":"enterprise","agent":"admin","deprecated":true},"sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64":{"tier":"enterprise","agent":"admin"},"sk-syslog-abiba":{"tier":"enterprise","agent":"Abiba","deprecated":true},"sk-856ffb0bbb-e5aaf78b10054eca608f8fbcbd73a889":{"tier":"enterprise","agent":"Abiba"},"sk-syslog-mumuni":{"tier":"enterprise","agent":"Mumuni","deprecated":true},"sk-b57e6e042e-47573660114f3138c852c47f62da807e":{"tier":"enterprise","agent":"Mumuni"},"sk-syslog-tanko":{"tier":"enterprise","agent":"Tanko","deprecated":true},"sk-620a05e95a-e93d875476b650a4d1137249ead8eaa7":{"tier":"enterprise","agent":"Tanko"},"sk-syslog-koby":{"tier":"enterprise","agent":"Koby","deprecated":true},"sk-eb3e6fc1c0-de1bf2edf35a53cb3749a2400483fdee":{"tier":"enterprise","agent":"Koby"},"sk-syslog-kagenz0":{"tier":"enterprise","agent":"Kagenz0","deprecated":true},"sk-12b66b3392-b548aed9138aeb6f698e8e521650ed9b":{"tier":"enterprise","agent":"Kagenz0"},"sk-syslog-koonimo":{"tier":"enterprise","agent":"Koonimo","deprecated":true},"sk-680d06686c-00ee8bf9dc3c93b276af122d49a14dfe":{"tier":"enterprise","agent":"Koonimo"},"sk-starter-abc123":{"tier":"starter","agent":"test-starter","deprecated":true},"sk-55da55907a-1bd7ff344e26feda50e9ac2219697860":{"tier":"starter","agent":"test-starter"},"sk-professional-xyz789":{"tier":"professional","agent":"test-pro","deprecated":true},"sk-b5159863e6-8df3ae52fb958cfe76cc2888c8c8e676":{"tier":"professional","agent":"test-pro"}} - ADMIN_KEY=sk-admin-ee09fffd04978b61a1569ac670c68814 @@ -139,5 +139,7 @@ services: - redis volumes: + - /opt/inference-harness/gpu_roster.yaml:/app/gpu_roster.yaml + - /opt/inference-harness/roster_loader.py:/app/roster_loader.py redis-data: pgdata: -- 2.54.0 From a97c7542134dd6095bf532eab9439fa3a9b274fa Mon Sep 17 00:00:00 2001 From: Abiba Date: Sun, 28 Jun 2026 17:18:43 +0000 Subject: [PATCH 3/8] fix(nginx): add default Litellm UI auth key, keep agent pass-through on /v1/ --- nginx/nginx.conf | 1 + 1 file changed, 1 insertion(+) diff --git a/nginx/nginx.conf b/nginx/nginx.conf index a6256e1..b85aabb 100644 --- a/nginx/nginx.conf +++ b/nginx/nginx.conf @@ -123,6 +123,7 @@ http { proxy_set_header Upgrade $http_upgrade; proxy_set_header Connection "upgrade"; proxy_buffering off; + proxy_set_header Authorization Bearer sk-wQWpYHbJngIR8FRxqgFAVw; } # Auth proxy to Authentik — accepts HTTP from LiteLLM, proxies HTTPS to .11 with SSL verify off -- 2.54.0 From 51f08404265d4f3d3c51aaeb341e71e873783a43 Mon Sep 17 00:00:00 2001 From: Abiba Date: Sun, 28 Jun 2026 19:03:41 +0000 Subject: [PATCH 4/8] feat: per-agent API keys + model pricing + nginx pass-through --- litellm_config.yaml | 12 ++++++------ nginx/nginx.conf | 2 +- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/litellm_config.yaml b/litellm_config.yaml index bcab179..1a4caa5 100644 --- a/litellm_config.yaml +++ b/litellm_config.yaml @@ -32,14 +32,14 @@ litellm_settings: - prometheus model_cost: gemma-4-12b: - input_cost_per_token: 0.0 - output_cost_per_token: 0.0 + input_cost_per_token: 0.000075 + output_cost_per_token: 0.0003 qwen3.6-27B-code: - input_cost_per_token: 0.0 - output_cost_per_token: 0.0 + input_cost_per_token: 0.00015 + output_cost_per_token: 0.0006 ornith-1.0-35b: - input_cost_per_token: 0.0 - output_cost_per_token: 0.0 + input_cost_per_token: 0.0002 + output_cost_per_token: 0.0008 syslog-auto: input_cost_per_token: 0.0 output_cost_per_token: 0.0 diff --git a/nginx/nginx.conf b/nginx/nginx.conf index b85aabb..eb45d71 100644 --- a/nginx/nginx.conf +++ b/nginx/nginx.conf @@ -123,7 +123,7 @@ http { proxy_set_header Upgrade $http_upgrade; proxy_set_header Connection "upgrade"; proxy_buffering off; - proxy_set_header Authorization Bearer sk-wQWpYHbJngIR8FRxqgFAVw; + proxy_set_header Authorization $http_authorization; } # Auth proxy to Authentik — accepts HTTP from LiteLLM, proxies HTTPS to .11 with SSL verify off -- 2.54.0 From ce2b90ed508ecba3d0212cf3d0a725283288cee0 Mon Sep 17 00:00:00 2001 From: Abiba Date: Sun, 28 Jun 2026 19:05:32 +0000 Subject: [PATCH 5/8] feat: daily cost snapshot script with cron --- scripts/daily-cost-snapshot.sh | 56 ++++++++++++++++++++++++++++++++++ 1 file changed, 56 insertions(+) create mode 100755 scripts/daily-cost-snapshot.sh diff --git a/scripts/daily-cost-snapshot.sh b/scripts/daily-cost-snapshot.sh new file mode 100755 index 0000000..509c0f5 --- /dev/null +++ b/scripts/daily-cost-snapshot.sh @@ -0,0 +1,56 @@ +#!/bin/bash +# Daily LiteLLM cost snapshot — cron at midnight +# Writes to /var/log/litellm/costs.csv on CT 116 + +MASTER="sk-litellm-7f96080dd99b15c36bd4b333b58a6796" +BASE="http://127.0.0.1:4001" +CSV="/var/log/litellm/costs.csv" +DATE=$(date -I) +YESTERDAY=$(date -I -d yesterday) + +# Fetch yesterday's spend by key +RESP=$(curl -s "${BASE}/global/spend/logs?start_date=${YESTERDAY}T00:00:00&end_date=${YESTERDAY}T23:59:59&page_size=1000" \ + -H "Authorization: Bearer $MASTER" 2>/dev/null) + +# Parse total spend +TOTAL=$(echo "$RESP" | python3 -c " +import sys, json +data = json.load(sys.stdin) +total = sum(r.get('spend', 0) or 0 for r in data.get('data', [])) +print(f'{total:.6f}') +" 2>/dev/null) + +# Parse per-model spend +PER_MODEL=$(echo "$RESP" | python3 -c " +import sys, json +from collections import defaultdict +data = json.load(sys.stdin) +spend = defaultdict(float) +for r in data.get('data', []): + model = r.get('model', 'unknown') + spend[model] += r.get('spend', 0) or 0 +for model, cost in sorted(spend.items()): + print(f'{model}:{cost:.6f}') +" 2>/dev/null) + +# Per-key spend +PER_KEY=$(echo "$RESP" | python3 -c " +import sys, json +from collections import defaultdict +data = json.load(sys.stdin) +spend = defaultdict(float) +for r in data.get('data', []): + key = r.get('api_key', 'unknown')[:20] + spend[key] += r.get('spend', 0) or 0 +for key, cost in sorted(spend.items()): + print(f'{key}:{cost:.6f}') +" 2>/dev/null) + +# Write CSV +mkdir -p $(dirname "$CSV") +if [ ! -f "$CSV" ]; then + echo "date,total_cost,per_model,per_key" > "$CSV" +fi +echo "${DATE},${TOTAL},\"${PER_MODEL}\",\"${PER_KEY}\"" >> "$CSV" + +echo "Snapshot: ${DATE} | Total: \$${TOTAL}" -- 2.54.0 From 7cad063e27b3fbc880ab170ebd6a2d3aa687f310 Mon Sep 17 00:00:00 2001 From: Abiba Date: Fri, 28 Aug 2026 14:53:51 +0000 Subject: [PATCH 6/8] fix(routing): restore gpu-dense (.8 qwen3.6-27B-code) as syslog-auto 0.55-weight member The Aug 24 rename moved the syslog-auto top-weight member to .110/gpu-vision, bypassing gpu-dense entirely (0 requests/hr while strix-moe and .110 absorbed everything). Restore the intended 55/30/15 split: gpu-dense, strix-moe, .110. --- litellm_config.yaml | 157 ++++++++++++++++++++++++++++++++++++-------- 1 file changed, 131 insertions(+), 26 deletions(-) diff --git a/litellm_config.yaml b/litellm_config.yaml index 1a4caa5..da85602 100644 --- a/litellm_config.yaml +++ b/litellm_config.yaml @@ -1,6 +1,9 @@ general_settings: master_key: os.environ/LITELLM_MASTER_KEY - store_model_in_db: true + store_model_in_db: false + user_url_allowed_hosts: + - 192.168.68.14 + - 192.168.68.14:5080 guardrails: - guardrail_name: input-moderation litellm_params: @@ -28,57 +31,159 @@ guardrails: guardrail: litellm_content_filter mode: pre_call litellm_settings: + user_url_allowed_hosts: + - 192.168.68.14 + - 192.168.68.14:5080 + cache: true + cache_params: + host: harness-redis + namespace: litellm + port: 6379 + ttl: 600 + type: redis + drop_params: true + success_callback: + - prometheus failure_callback: - prometheus model_cost: gemma-4-12b: - input_cost_per_token: 0.000075 - output_cost_per_token: 0.0003 + input_cost_per_token: 1.5e-07 + output_cost_per_token: 6.0e-07 qwen3.6-27B-code: - input_cost_per_token: 0.00015 - output_cost_per_token: 0.0006 - ornith-1.0-35b: - input_cost_per_token: 0.0002 - output_cost_per_token: 0.0008 + input_cost_per_token: 1.5e-07 + output_cost_per_token: 6.0e-07 + strix-moe: + input_cost_per_token: 1.5e-07 + output_cost_per_token: 6.0e-07 syslog-auto: - input_cost_per_token: 0.0 - output_cost_per_token: 0.0 - num_retries: 0 + input_cost_per_token: 1.5e-07 + output_cost_per_token: 6.0e-07 + num_retries: 2 request_timeout: 600 set_verbose: true sso_callback: /sso/callback model_list: - litellm_params: - api_base: http://router:9000/v1 - api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 - model: openai/syslog-auto - rpm: 600 - model_name: syslog-auto -- litellm_params: - api_base: http://router:9000/v1 - api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 + api_base: http://192.168.68.8:8080/v1 + api_key: not-needed model: openai/qwen3.6-27B-code + timeout: 300 + model_info: + max_input_tokens: 131072 model_name: qwen3.6-27B-code - litellm_params: - api_base: http://router:9000/v1 - api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 + api_base: http://192.168.68.110:8080/v1 + api_key: not-needed model: openai/gemma-4-12b + timeout: 120 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 model_name: gemma-4-12b - - litellm_params: - api_base: http://router:9000/v1 - api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 - model: openai/ornith-1.0-35b - model_name: ornith-1.0-35b + api_base: http://192.168.68.15:8080/v1 + api_key: not-needed + model: openai/strix-moe + timeout: 300 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 + model_name: qwen3.6-35B-udq4 +- litellm_params: + api_base: http://192.168.68.15:8080/v1 + api_key: not-needed + model: openai/strix-moe + rpm: 40 + timeout: 300 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 + model_name: strix-moe +- litellm_params: + api_base: http://192.168.68.8:8080/v1 + api_key: not-needed + model: openai/qwen3.6-27B-code + rpm: 500 + timeout: 300 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 + model_name: gpu-dense +- litellm_params: + api_base: http://192.168.68.8:8080/v1 + api_key: not-needed + model: openai/qwen3.6-27B-code + rpm: 500 + timeout: 300 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 + model_name: gpu-vision +- litellm_params: + api_base: http://192.168.68.8:8080/v1 + api_key: not-needed + model: openai/qwen3.6-27B-code + rpm: 500 + timeout: 300 + model_info: + max_input_tokens: 131072 + weight: 0.55 + model_info: + max_input_tokens: 131072 + max_model_tokens: 131072 + max_tokens: 131072 + model_name: syslog-auto +- litellm_params: + api_base: http://192.168.68.15:8080/v1 + api_key: not-needed + model: openai/strix-moe + rpm: 60 + timeout: 300 + model_info: + max_input_tokens: 131072 + weight: 0.3 + model_name: syslog-auto +- litellm_params: + api_base: http://192.168.68.110:8080/v1 + api_key: not-needed + model: openai/gemma-4-12b + rpm: 200 + timeout: 300 + model_info: + max_input_tokens: 131072 + weight: 0.15 + model_name: syslog-auto router_settings: allowed_fails: 100 enable_loadbalancing_on_proxy: false fallbacks: - syslog-auto: - qwen3.6-27B-code + - strix-moe - gemma-4-12b - qwen3.6-27B-code: - gemma-4-12b - gemma-4-12b: - qwen3.6-27B-code + - strix-moe: + - qwen3.6-27B-code + - gemma-4-12b + request_timeout: 300 routing_strategy: usage-based-routing + +agents: +- agent_name: agent-zero-homelab + agent_card_params: + name: Agent Zero HomeLab + url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab + protocolVersion: '1.0' -- 2.54.0 From 93e7605aa2fe0de1b6234549502767060907f084 Mon Sep 17 00:00:00 2001 From: agent-zero Date: Fri, 11 Sep 2026 19:34:30 +0000 Subject: [PATCH 7/8] chore(ct116): capture production state (LiteLLM 1.99.1, nginx /ui //docs, router decommission, trove agent) --- Dockerfile.dashboard | 0 Dockerfile.queue | 0 LITELLM-MIGRATION-PLAN.md | 22 +- README.md | 4 +- dashboard/dashboard.html | 2 +- dashboard/dashboard.py | 22 +- dashboard/harness.html | 8 +- docker-compose.yml | 34 +- docker-compose.yml.bak | 76 ++- gpu-router-docker.conf | 106 --- gpu-router.conf | 106 --- gpu_roster.yaml | 29 +- litellm_config.yaml | 53 +- nginx/nginx.conf | 122 ++-- queue-service/queue-service.py | 121 ---- router/Dockerfile | 9 - router/http_patch.py | 90 --- router/requirements.txt | 3 - router/route_v2.py | 77 --- router/route_v3.py | 92 --- router/router.py | 1149 -------------------------------- 21 files changed, 196 insertions(+), 1929 deletions(-) delete mode 100644 Dockerfile.dashboard delete mode 100644 Dockerfile.queue delete mode 100644 gpu-router-docker.conf delete mode 100644 gpu-router.conf delete mode 100644 queue-service/queue-service.py delete mode 100644 router/Dockerfile delete mode 100644 router/http_patch.py delete mode 100644 router/requirements.txt delete mode 100644 router/route_v2.py delete mode 100644 router/route_v3.py delete mode 100644 router/router.py diff --git a/Dockerfile.dashboard b/Dockerfile.dashboard deleted file mode 100644 index e69de29..0000000 diff --git a/Dockerfile.queue b/Dockerfile.queue deleted file mode 100644 index e69de29..0000000 diff --git a/LITELLM-MIGRATION-PLAN.md b/LITELLM-MIGRATION-PLAN.md index 3c0d05e..6108ce1 100644 --- a/LITELLM-MIGRATION-PLAN.md +++ b/LITELLM-MIGRATION-PLAN.md @@ -49,7 +49,7 @@ - qwen3.6-35B qwen3.6-27B gemma-4-12b + qwen3.6-35B qwen3.6-27B gpu-vision MoE/Strix Dense/RTX3090 VLM/RTX 5070 :8080 (llama) :8080 (llama) :8080 (llama) :8090 (side) :8090 (side) :8090 (sidecar) @@ -84,7 +84,7 @@ |-----|------|-----------|---------|------|---------| | qwen3.6-35B-A3B (MoE) | 192.168.68.15 | :8080 | :8090 | Strix Halo | 262K | | qwen3.6-27B-code (Dense) | 192.168.68.8 | :8080 | :8090 | RTX 3090 | 262K | -| gemma-4-12b (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K | +| gpu-vision (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K | ### 1.3 Existing LiteLLM POC on CT 116 @@ -249,9 +249,9 @@ model_list: api_base: http://router:9000/v1 api_key: os.environ/ROUTER_API_KEY - - model_name: gemma-4-12b + - model_name: gpu-vision litellm_params: - model: openai/gemma-4-12b + model: openai/gpu-vision api_base: http://router:9000/v1 api_key: os.environ/ROUTER_API_KEY @@ -298,9 +298,9 @@ router_settings: # Fallback chains: LiteLLM retries down the chain when router returns saturated # This gives accurate per-model metrics because router no longer silently reroutes fallbacks: - - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"] - - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"] - - gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] + - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"] + - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"] + - gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] # Cost tracking: map model names to per-token pricing for spend tracking litellm_settings: @@ -311,7 +311,7 @@ litellm_settings: qwen3.6-27B-code: input_cost_per_token: 0.0 output_cost_per_token: 0.0 - gemma-4-12b: + gpu-vision: input_cost_per_token: 0.0 output_cost_per_token: 0.0 # For internal cost allocation, set symbolic rates: @@ -705,9 +705,9 @@ if req != "auto": router_settings: allowed_fails: 100 fallbacks: - - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"] - - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"] - - gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] + - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"] + - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"] + - gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] ``` ### Result diff --git a/README.md b/README.md index c5f42d9..5e48169 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ CT 116 Docker stack for routing local GPU models through a unified OpenAI-compat nginx :80 → router :9000 → GPU backends ├─ qwen3.6-35B-A3B (MoE) @ 192.168.68.15:8080 [2 slots, 262K ctx] ├─ qwen3.6-27B-code (Dense) @ 192.168.68.8:8080 [2 slots, 262K ctx] - └─ gemma-4-12b (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx] + └─ gpu-vision (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx] Total: 6 concurrent slots LiteLLM :8081 (fallback) | Dashboard :3000 | Redis :6379 (local) @@ -63,7 +63,7 @@ When all GPUs are saturated, requests enter a polling queue (500ms intervals) in |-----|-------|------|-------| | Strix Halo | qwen3.6-35B-A3B (MoE) | 65GB | 2 | 262K | General quality | | RTX 3090 | qwen3.6-27B-code (Dense) | 24GB | 2 | 262K | Code, reasoning | -| RTX 5070 | gemma-4-12b (VLM) | 12GB | 2 | 262K | Speed, vision | +| RTX 5070 | gpu-vision (VLM) | 12GB | 2 | 262K | Speed, vision | ## Maintenance diff --git a/dashboard/dashboard.html b/dashboard/dashboard.html index f743140..fdee4c7 100644 --- a/dashboard/dashboard.html +++ b/dashboard/dashboard.html @@ -70,7 +70,7 @@ body{background:var(--bg);color:var(--text);font-family:-apple-system,BlinkMacSy