diff --git a/Dockerfile.dashboard b/Dockerfile.dashboard deleted file mode 100644 index e69de29..0000000 diff --git a/Dockerfile.queue b/Dockerfile.queue deleted file mode 100644 index e69de29..0000000 diff --git a/LITELLM-MIGRATION-PLAN.md b/LITELLM-MIGRATION-PLAN.md index 3c0d05e..6108ce1 100644 --- a/LITELLM-MIGRATION-PLAN.md +++ b/LITELLM-MIGRATION-PLAN.md @@ -49,7 +49,7 @@ - qwen3.6-35B qwen3.6-27B gemma-4-12b + qwen3.6-35B qwen3.6-27B gpu-vision MoE/Strix Dense/RTX3090 VLM/RTX 5070 :8080 (llama) :8080 (llama) :8080 (llama) :8090 (side) :8090 (side) :8090 (sidecar) @@ -84,7 +84,7 @@ |-----|------|-----------|---------|------|---------| | qwen3.6-35B-A3B (MoE) | 192.168.68.15 | :8080 | :8090 | Strix Halo | 262K | | qwen3.6-27B-code (Dense) | 192.168.68.8 | :8080 | :8090 | RTX 3090 | 262K | -| gemma-4-12b (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K | +| gpu-vision (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K | ### 1.3 Existing LiteLLM POC on CT 116 @@ -249,9 +249,9 @@ model_list: api_base: http://router:9000/v1 api_key: os.environ/ROUTER_API_KEY - - model_name: gemma-4-12b + - model_name: gpu-vision litellm_params: - model: openai/gemma-4-12b + model: openai/gpu-vision api_base: http://router:9000/v1 api_key: os.environ/ROUTER_API_KEY @@ -298,9 +298,9 @@ router_settings: # Fallback chains: LiteLLM retries down the chain when router returns saturated # This gives accurate per-model metrics because router no longer silently reroutes fallbacks: - - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"] - - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"] - - gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] + - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"] + - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"] + - gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] # Cost tracking: map model names to per-token pricing for spend tracking litellm_settings: @@ -311,7 +311,7 @@ litellm_settings: qwen3.6-27B-code: input_cost_per_token: 0.0 output_cost_per_token: 0.0 - gemma-4-12b: + gpu-vision: input_cost_per_token: 0.0 output_cost_per_token: 0.0 # For internal cost allocation, set symbolic rates: @@ -705,9 +705,9 @@ if req != "auto": router_settings: allowed_fails: 100 fallbacks: - - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"] - - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"] - - gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] + - qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"] + - qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"] + - gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"] ``` ### Result diff --git a/README.md b/README.md index c5f42d9..5e48169 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ CT 116 Docker stack for routing local GPU models through a unified OpenAI-compat nginx :80 → router :9000 → GPU backends ├─ qwen3.6-35B-A3B (MoE) @ 192.168.68.15:8080 [2 slots, 262K ctx] ├─ qwen3.6-27B-code (Dense) @ 192.168.68.8:8080 [2 slots, 262K ctx] - └─ gemma-4-12b (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx] + └─ gpu-vision (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx] Total: 6 concurrent slots LiteLLM :8081 (fallback) | Dashboard :3000 | Redis :6379 (local) @@ -63,7 +63,7 @@ When all GPUs are saturated, requests enter a polling queue (500ms intervals) in |-----|-------|------|-------| | Strix Halo | qwen3.6-35B-A3B (MoE) | 65GB | 2 | 262K | General quality | | RTX 3090 | qwen3.6-27B-code (Dense) | 24GB | 2 | 262K | Code, reasoning | -| RTX 5070 | gemma-4-12b (VLM) | 12GB | 2 | 262K | Speed, vision | +| RTX 5070 | gpu-vision (VLM) | 12GB | 2 | 262K | Speed, vision | ## Maintenance diff --git a/dashboard/dashboard.html b/dashboard/dashboard.html index f743140..fdee4c7 100644 --- a/dashboard/dashboard.html +++ b/dashboard/dashboard.html @@ -70,7 +70,7 @@ body{background:var(--bg);color:var(--text);font-family:-apple-system,BlinkMacSy