Merge pull request 'Reconcile upstream into CT116 production state' (#1) from sync/ct116-reconcile-20260912-095913 into main
This commit is contained in:
+11
-11
@@ -49,7 +49,7 @@
|
||||
|
||||
|
||||
|
||||
qwen3.6-35B qwen3.6-27B gemma-4-12b
|
||||
qwen3.6-35B qwen3.6-27B gpu-vision
|
||||
MoE/Strix Dense/RTX3090 VLM/RTX 5070
|
||||
:8080 (llama) :8080 (llama) :8080 (llama)
|
||||
:8090 (side) :8090 (side) :8090 (sidecar)
|
||||
@@ -84,7 +84,7 @@
|
||||
|-----|------|-----------|---------|------|---------|
|
||||
| qwen3.6-35B-A3B (MoE) | 192.168.68.15 | :8080 | :8090 | Strix Halo | 262K |
|
||||
| qwen3.6-27B-code (Dense) | 192.168.68.8 | :8080 | :8090 | RTX 3090 | 262K |
|
||||
| gemma-4-12b (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K |
|
||||
| gpu-vision (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K |
|
||||
|
||||
### 1.3 Existing LiteLLM POC on CT 116
|
||||
|
||||
@@ -249,9 +249,9 @@ model_list:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: os.environ/ROUTER_API_KEY
|
||||
|
||||
- model_name: gemma-4-12b
|
||||
- model_name: gpu-vision
|
||||
litellm_params:
|
||||
model: openai/gemma-4-12b
|
||||
model: openai/gpu-vision
|
||||
api_base: http://router:9000/v1
|
||||
api_key: os.environ/ROUTER_API_KEY
|
||||
|
||||
@@ -298,9 +298,9 @@ router_settings:
|
||||
# Fallback chains: LiteLLM retries down the chain when router returns saturated
|
||||
# This gives accurate per-model metrics because router no longer silently reroutes
|
||||
fallbacks:
|
||||
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"]
|
||||
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"]
|
||||
- gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
|
||||
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"]
|
||||
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"]
|
||||
- gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
|
||||
|
||||
# Cost tracking: map model names to per-token pricing for spend tracking
|
||||
litellm_settings:
|
||||
@@ -311,7 +311,7 @@ litellm_settings:
|
||||
qwen3.6-27B-code:
|
||||
input_cost_per_token: 0.0
|
||||
output_cost_per_token: 0.0
|
||||
gemma-4-12b:
|
||||
gpu-vision:
|
||||
input_cost_per_token: 0.0
|
||||
output_cost_per_token: 0.0
|
||||
# For internal cost allocation, set symbolic rates:
|
||||
@@ -705,9 +705,9 @@ if req != "auto":
|
||||
router_settings:
|
||||
allowed_fails: 100
|
||||
fallbacks:
|
||||
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"]
|
||||
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"]
|
||||
- gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
|
||||
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"]
|
||||
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"]
|
||||
- gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
|
||||
```
|
||||
|
||||
### Result
|
||||
|
||||
@@ -8,7 +8,7 @@ CT 116 Docker stack for routing local GPU models through a unified OpenAI-compat
|
||||
nginx :80 → router :9000 → GPU backends
|
||||
├─ qwen3.6-35B-A3B (MoE) @ 192.168.68.15:8080 [2 slots, 262K ctx]
|
||||
├─ qwen3.6-27B-code (Dense) @ 192.168.68.8:8080 [2 slots, 262K ctx]
|
||||
└─ gemma-4-12b (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx]
|
||||
└─ gpu-vision (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx]
|
||||
Total: 6 concurrent slots
|
||||
|
||||
LiteLLM :8081 (fallback) | Dashboard :3000 | Redis :6379 (local)
|
||||
@@ -63,7 +63,7 @@ When all GPUs are saturated, requests enter a polling queue (500ms intervals) in
|
||||
|-----|-------|------|-------|
|
||||
| Strix Halo | qwen3.6-35B-A3B (MoE) | 65GB | 2 | 262K | General quality |
|
||||
| RTX 3090 | qwen3.6-27B-code (Dense) | 24GB | 2 | 262K | Code, reasoning |
|
||||
| RTX 5070 | gemma-4-12b (VLM) | 12GB | 2 | 262K | Speed, vision |
|
||||
| RTX 5070 | gpu-vision (VLM) | 12GB | 2 | 262K | Speed, vision |
|
||||
|
||||
## Maintenance
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@ body{background:var(--bg);color:var(--text);font-family:-apple-system,BlinkMacSy
|
||||
</div>
|
||||
|
||||
<script>
|
||||
const COLORS={'qwen3.6-35B-A3B':'#10b981','qwen3.6-27B-code':'#8b5cf6','gemma-4-12b':'#3b82f6'};
|
||||
const COLORS={'qwen3.6-35B-A3B':'#10b981','qwen3.6-27B-code':'#8b5cf6','gpu-vision':'#3b82f6'};
|
||||
const HISTORY=[]; // rolling 60 sample history for chart
|
||||
function Q(id){return document.getElementById(id)}
|
||||
|
||||
|
||||
+11
-11
@@ -119,7 +119,7 @@ body { background: #0b0f17; color: #bcc3cd; font-family: -apple-system, BlinkMac
|
||||
<div class="d-flex gap-2">
|
||||
<select id="scatter-model" onchange="loadScatter()" style="font-size:10px;background:#1e293b;color:#94a3b8;border:1px solid #334155;border-radius:4px;padding:2px 6px">
|
||||
<option value="all">All Models</option>
|
||||
<option value="gemma-4-12b">12B VLM</option>
|
||||
<option value="gpu-vision">12B VLM</option>
|
||||
<option value="qwen3.6-27B-code">27B Dense</option>
|
||||
<option value="qwen3.6-35B-A3B">35B MoE</option>
|
||||
</select>
|
||||
@@ -136,9 +136,9 @@ body { background: #0b0f17; color: #bcc3cd; font-family: -apple-system, BlinkMac
|
||||
</div>
|
||||
|
||||
<script>
|
||||
var MC={'gemma-4-12b':'#22c55e','qwen3.6-27B-code':'#f59e0b','qwen3.6-35B-A3B':'#a78bfa'};
|
||||
var ML={'gemma-4-12b':'Gemma 4 12B','qwen3.6-27B-code':'Qwen Code','qwen3.6-35B-A3B':'Qwen MoE'};
|
||||
var GL={'qwen3.6-35B-A3B':'MoE - Strix Halo','qwen3.6-27B-code':'Dense - RTX 3090','gemma-4-12b':'VLM - RTX 5070'};
|
||||
var MC={'gpu-vision':'#22c55e','qwen3.6-27B-code':'#f59e0b','qwen3.6-35B-A3B':'#a78bfa'};
|
||||
var ML={'gpu-vision':'Gemma 4 12B','qwen3.6-27B-code':'Qwen Code','qwen3.6-35B-A3B':'Qwen MoE'};
|
||||
var GL={'qwen3.6-35B-A3B':'MoE - Strix Halo','qwen3.6-27B-code':'Dense - RTX 3090','gpu-vision':'VLM - RTX 5070'};
|
||||
function $(id){return document.getElementById(id);}
|
||||
|
||||
function render(data){
|
||||
@@ -147,7 +147,7 @@ var t=Object.values(data.route_counts||{}).reduce((a,b)=>a+b,0);
|
||||
var ta=0,tm=0;data.gpus.forEach(function(g){ta+=(g.active_requests||0);tm+=(g.max_concurrent||1)});
|
||||
$('kpi-total').textContent=t;$('kpi-active').textContent=ta+'/'+tm;$('kpi-agents').textContent=Object.keys(data.agent_counts||{}).length;
|
||||
$('update-time').textContent=new Date().toLocaleTimeString();
|
||||
var ids={'qwen3.6-35B-A3B':'gpu-moe','qwen3.6-27B-code':'gpu-dense','gemma-4-12b':'gpu-light'};
|
||||
var ids={'qwen3.6-35B-A3B':'gpu-moe','qwen3.6-27B-code':'gpu-dense','gpu-vision':'gpu-light'};
|
||||
data.gpus.forEach(function(g){
|
||||
var el=$(ids[g.id]);if(!el)return;
|
||||
var a=g.active_requests||0,mx=g.max_concurrent||1;
|
||||
@@ -179,14 +179,14 @@ var sc=pct>=100?'#ef4444':pct>=50?'#f59e0b':'#22c55e';
|
||||
var circ=188.5,dash=(pct/100)*circ;
|
||||
var h='<div class=\"d-inline-block position-relative mb-2\"><svg width=\"72\" height=\"72\"><circle cx=\"36\" cy=\"36\" r=\"30\" fill=\"none\" stroke=\"#1e293b\" stroke-width=\"6\"/><circle cx=\"36\" cy=\"36\" r=\"30\" fill=\"none\" stroke=\"'+sc+'\" stroke-width=\"6\" stroke-dasharray=\"'+dash+' '+(circ-dash)+'\" stroke-linecap=\"round\" transform=\"rotate(-90 36 36)\"/></svg><div style=\"position:absolute;top:50%;left:50%;transform:translate(-50%,-50%);text-align:center\"><div class=\"ring-label\" style=\"color:'+sc+'\">'+ta+'</div><div class=\"ring-sublabel\">/ '+tm+' slots</div></div></div>';
|
||||
h+='<div class=\"fw-bold mb-2 small\" style=\"color:'+sc+'\">'+st+'</div>';
|
||||
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gemma-4-12b':'VLM'};
|
||||
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gpu-vision':'VLM'};
|
||||
data.gpus.forEach(function(g){var a=g.active_requests||0,mx=g.max_concurrent||1,gp=mx>0?Math.round(a/mx*100):0;h+='<div class=\"d-flex align-items-center gap-2 mb-1 justify-content-center\"><span class=\"small\" style=\"min-width:32px;text-align:right;font-size:10px\">'+(lb[g.id]||g.id)+'</span><div style=\"flex:1;max-width:70px;height:3px;background:#1e293b;border-radius:2px;overflow:hidden\"><div style=\"height:100%;width:'+gp+'%;background:'+sc+';border-radius:2px\"></div></div><span class=\"small\" style=\"min-width:22px;font-size:10px\">'+a+'/'+mx+'</span></div>'});
|
||||
el.innerHTML=h;
|
||||
}
|
||||
|
||||
function renderGPUMetrics(data){
|
||||
var el=$('gpu-metrics-card');if(!el)return;
|
||||
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gemma-4-12b':'VLM'};
|
||||
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gpu-vision':'VLM'};
|
||||
var h='';data.gpus.forEach(function(g){
|
||||
var nm=lb[g.id]||g.id,tp=g.temp_c||0,ut=g.gpu_util_pct||0,pw=g.power_w||0,pl=g.power_limit_w||0;
|
||||
var tc=tp>85?'#ef4444':tp>70?'#f59e0b':'#22c55e',uc=ut>90?'#ef4444':ut>70?'#f59e0b':'#22c55e';
|
||||
@@ -219,8 +219,8 @@ function loadPerf(){fetch('/api/performance?window='+perfWindow).then(function(r
|
||||
function renderPerf(d){
|
||||
var models=d.models||[],reasons=d.reasons||[],agents=d.agents||[],sum=d.summary||{};
|
||||
// Latency bars: p50/p95/p99 per model
|
||||
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gemma-4-12b':'12B VLM'};
|
||||
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gemma-4-12b':'#22c55e'};
|
||||
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gpu-vision':'12B VLM'};
|
||||
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gpu-vision':'#22c55e'};
|
||||
if(!models.length){$('perf-latency').innerHTML='<div class="text-secondary small text-center py-4">Accumulating data...</div>';return;}
|
||||
var maxLat=Math.max(...models.map(function(m){return m.latency.p99||0}),1);
|
||||
var latHTML=models.map(function(m){
|
||||
@@ -270,8 +270,8 @@ fetch('/api/scatter?window=24&model='+m).then(function(r){return r.json()}).then
|
||||
function renderScatter(d){
|
||||
var pts=d.points||[],el=$('scatter-plot'),lg=$('scatter-legend');
|
||||
if(!pts.length){el.innerHTML='<div class="text-secondary small text-center py-5">No data yet</div>';return;}
|
||||
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gemma-4-12b':'#22c55e','unknown':'#38bdf8'};
|
||||
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gemma-4-12b':'12B VLM'};
|
||||
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gpu-vision':'#22c55e','unknown':'#38bdf8'};
|
||||
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gpu-vision':'12B VLM'};
|
||||
var maxX=Math.max.apply(null,pts.map(function(p){return p.prompt_tokens||0}))||1000;
|
||||
var maxY=Math.max.apply(null,pts.map(function(p){return p.inference_ms||0}))||5000;
|
||||
// Log scale for X axis (prompt tokens vary widely)
|
||||
|
||||
@@ -362,7 +362,7 @@ function dashboard() {
|
||||
try {
|
||||
const metrics = await fetch('/metrics/circuit-breaker').then(r => r.json());
|
||||
this.gpuHealth = [
|
||||
{ id: 'gemma3-70b', name: 'Gemma 3 70B', model: 'gemma3-70b', health_score: metrics.gemma3_70b?.gpu_health_score || 39.4, vram_pct: 45, temp: 78, load: 65, is_preferred: true },
|
||||
{ id: 'gpu-vision', name: 'Qwen3.5-9B Vision', model: 'gpu-vision', health_score: metrics.gpu_vision?.gpu_health_score || 39.4, vram_pct: 45, temp: 78, load: 65, is_preferred: true },
|
||||
{ id: 'deepseek-v3', name: 'DeepSeek V3', model: 'deepseek-v3', health_score: metrics.deepseek_v3?.gpu_health_score || 45.9, vram_pct: 60, temp: 82, load: 50, is_preferred: false },
|
||||
{ id: 'mistral-small', name: 'Mistral Small', model: 'mistral-small', health_score: metrics.mistral_small?.gpu_health_score || 35.0, vram_pct: 30, temp: 65, load: 40, is_preferred: false }
|
||||
];
|
||||
@@ -388,7 +388,7 @@ function dashboard() {
|
||||
async fetchSessionAnalytics() {
|
||||
try {
|
||||
this.sessionData = {
|
||||
distribution: { 'gemma3-70b': 45, 'deepseek-v3': 30, 'mistral-small': 25 },
|
||||
distribution: { 'gpu-vision': 45, 'deepseek-v3': 30, 'mistral-small': 25 },
|
||||
trend: Array.from({ length: 24 }, (_, i) => ({ time: `${i}:00`, sessions: Math.floor(Math.random() * 20) + 10 })),
|
||||
peaks: { '09:00': 25, '14:00': 30, '18:00': 20 }
|
||||
};
|
||||
@@ -400,7 +400,7 @@ function dashboard() {
|
||||
try {
|
||||
this.systemPerf = {
|
||||
latency: { p50: Math.floor(Math.random() * 50) + 100, p95: Math.floor(Math.random() * 200) + 250, p99: Math.floor(Math.random() * 500) + 400 },
|
||||
errorRates: { 'gemma3-70b': Math.random() * 0.01, 'deepseek-v3': Math.random() * 0.02, 'mistral-small': Math.random() * 0.015 }
|
||||
errorRates: { 'gpu-vision': Math.random() * 0.01, 'deepseek-v3': Math.random() * 0.02, 'mistral-small': Math.random() * 0.015 }
|
||||
};
|
||||
console.log('System performance loaded');
|
||||
} catch (error) { console.error('Failed to load system performance:', error); }
|
||||
@@ -434,7 +434,7 @@ function dashboard() {
|
||||
},
|
||||
|
||||
getGPUColor(name) {
|
||||
const colors = { 'gemma3-70b': '#3b82f6', 'deepseek-v3': '#8b5cf6', 'mistral-small': '#10b981' };
|
||||
const colors = { 'gpu-vision': '#3b82f6', 'deepseek-v3': '#8b5cf6', 'mistral-small': '#10b981' };
|
||||
return colors[name] || '#9ca3af';
|
||||
}
|
||||
};
|
||||
|
||||
+49
-26
@@ -1,4 +1,4 @@
|
||||
version: '3.8'
|
||||
version: "3.8"
|
||||
|
||||
services:
|
||||
redis:
|
||||
@@ -16,46 +16,70 @@ services:
|
||||
timeout: 3s
|
||||
retries: 5
|
||||
|
||||
router:
|
||||
build: ./router
|
||||
container_name: harness-router
|
||||
postgres:
|
||||
image: postgres:16-alpine
|
||||
container_name: harness-postgres
|
||||
restart: unless-stopped
|
||||
network_mode: host
|
||||
ports:
|
||||
- "127.0.0.1:5432:5432"
|
||||
environment:
|
||||
- REDIS_URL=redis://127.0.0.1:6379
|
||||
- GPU_MOE_URL=http://192.168.68.15:8080/v1
|
||||
- GPU_DENSE_URL=http://192.168.68.8:8080/v1
|
||||
- GPU_LIGHT_URL=http://192.168.68.110:8080/v1
|
||||
- API_KEYS={"sk-sys...-key":{"tier":"enterprise","agent":"admin","deprecated":true},"sk-9e6...cb64":{"tier":"enterprise","agent":"admin"},"***":{"tier":"enterprise","agent":"Abiba","deprecated":true},"sk-856...a889":{"tier":"enterprise","agent":"Abiba"},"***":{"tier":"enterprise","agent":"Mumuni","deprecated":true},"sk-b57...807e":{"tier":"enterprise","agent":"Mumuni"},"***":{"tier":"enterprise","agent":"Tanko","deprecated":true},"sk-620...eaa7":{"tier":"enterprise","agent":"Tanko"},"***":{"tier":"enterprise","agent":"Koby","deprecated":true},"sk-eb3...fdee":{"tier":"enterprise","agent":"Koby"},"***":{"tier":"enterprise","agent":"Kagenz0","deprecated":true},"sk-12b...ed9b":{"tier":"enterprise","agent":"Kagenz0"},"***":{"tier":"enterprise","agent":"Koonimo","deprecated":true},"sk-680...4dfe":{"tier":"enterprise","agent":"Koonimo"},"***":{"tier":"starter","agent":"test-starter","deprecated":true},"sk-55d...7860":{"tier":"starter","agent":"test-starter"},"sk-pro...z789":{"tier":"professional","agent":"test-pro","deprecated":true},"sk-b51...e676":{"tier":"professional","agent":"test-pro"}}
|
||||
- ADMIN_KEY=sk-adm...8814
|
||||
- POSTGRES_DB=litellm
|
||||
- POSTGRES_USER=litellm
|
||||
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD}
|
||||
volumes:
|
||||
- pgdata:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:9000/health')"]
|
||||
interval: 30s
|
||||
timeout: 15s
|
||||
retries: 3
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_healthy
|
||||
test: ["CMD-SHELL", "pg_isready -U litellm"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
retries: 5
|
||||
|
||||
litellm:
|
||||
image: ghcr.io/berriai/litellm:main-stable
|
||||
image: ghcr.io/docker.litellm.ai/berriai/litellm:1.99.1
|
||||
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
||||
container_name: harness-litellm
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "127.0.0.1:8081:4000"
|
||||
- "4000:4000"
|
||||
volumes:
|
||||
- ./litellm_config.yaml:/app/config.yaml
|
||||
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
|
||||
environment:
|
||||
- LITELLM_MASTER_KEY=sk-sys...-key
|
||||
- PROMETHEUS_EXPORTER=true
|
||||
- ENFORCE_PRISMA_MIGRATION_CHECK=true
|
||||
- LITELLM_MASTER_KEY=${LITELLM_MASTER_KEY}
|
||||
- DATABASE_URL=postgresql://litellm:${POSTGRES_PASSWORD}@postgres:5432/litellm
|
||||
- STORE_MODEL_IN_DB=True
|
||||
- LITELLM_UI_USERNAME=admin
|
||||
- LITELLM_UI_PASSWORD=${LITELLM_UI_PASSWORD}
|
||||
- UI_USERNAME=admin
|
||||
- UI_PASSWORD=${UI_PASSWORD}
|
||||
- OPENAI_API_KEY=${OPENAI_API_KEY}
|
||||
- PROXY_BASE_URL=https://litellm.sysloggh.net
|
||||
- DOCS_URL=/docs
|
||||
- ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY}
|
||||
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
|
||||
- GENERIC_CLIENT_SECRET=${GENERIC_CLIENT_SECRET}
|
||||
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
|
||||
- GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/
|
||||
- GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/
|
||||
- GENERIC_SCOPE=openid email profile
|
||||
- MCP_SERVER_RAHOS_URL=http://192.168.68.65:3100/mcp
|
||||
- MCP_SERVER_RAHOS_TRANSPORT=http
|
||||
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
|
||||
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
- "auth.sysloggh.net:192.168.68.11"
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 120s
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
|
||||
@@ -69,7 +93,7 @@ services:
|
||||
- ./nginx/nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
- ./dashboard:/opt/inference-harness/dashboard:ro
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
- "auth.sysloggh.net:192.168.68.11"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://127.0.0.1/health"]
|
||||
interval: 30s
|
||||
@@ -86,7 +110,7 @@ services:
|
||||
ports:
|
||||
- "127.0.0.1:3000:3000"
|
||||
environment:
|
||||
- REDIS_URL=redis://127.0.0.1:6379
|
||||
- REDIS_URL=redis://redis:6379
|
||||
- GPU_SIDECARS=192.168.68.15:8090,192.168.68.8:8090,192.168.68.110:8090
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:3000/health')"]
|
||||
@@ -96,8 +120,7 @@ services:
|
||||
depends_on:
|
||||
- redis
|
||||
|
||||
|
||||
volumes:
|
||||
redis-data:
|
||||
|
||||
# LiteLLM command override to load config
|
||||
# (appended to fix config loading issue)
|
||||
pgdata:
|
||||
|
||||
+62
-14
@@ -1,4 +1,4 @@
|
||||
version: '3.8'
|
||||
version: "3.8"
|
||||
|
||||
services:
|
||||
redis:
|
||||
@@ -16,43 +16,89 @@ services:
|
||||
timeout: 3s
|
||||
retries: 5
|
||||
|
||||
postgres:
|
||||
image: postgres:16-alpine
|
||||
container_name: harness-postgres
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "127.0.0.1:5432:5432"
|
||||
environment:
|
||||
- POSTGRES_DB=litellm
|
||||
- POSTGRES_USER=litellm
|
||||
- POSTGRES_PASSWORD=d9fc143e3dc1a7a8e672c359fea95c5e
|
||||
volumes:
|
||||
- pgdata:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U litellm"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
retries: 5
|
||||
|
||||
router:
|
||||
build: ./router
|
||||
container_name: harness-router
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "9000:9000"
|
||||
- "127.0.0.1:9000:9000"
|
||||
environment:
|
||||
- REDIS_URL=redis://redis:6379
|
||||
- GPU_MOE_URL=http://192.168.68.15:8080/v1
|
||||
- GPU_DENSE_URL=http://192.168.68.8:8080/v1
|
||||
- GPU_MOE_URL=http://192.168.68.15:8080/v1
|
||||
- GPU_LIGHT_URL=http://192.168.68.110:8080/v1
|
||||
- API_KEYS={"sk-syslog-local-master-key":{"tier":"enterprise","agent":"admin","deprecated":true},"sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64":{"tier":"enterprise","agent":"admin"},"sk-syslog-abiba":{"tier":"enterprise","agent":"Abiba","deprecated":true},"sk-856ffb0bbb-e5aaf78b10054eca608f8fbcbd73a889":{"tier":"enterprise","agent":"Abiba"},"sk-syslog-mumuni":{"tier":"enterprise","agent":"Mumuni","deprecated":true},"sk-b57e6e042e-47573660114f3138c852c47f62da807e":{"tier":"enterprise","agent":"Mumuni"},"sk-syslog-tanko":{"tier":"enterprise","agent":"Tanko","deprecated":true},"sk-620a05e95a-e93d875476b650a4d1137249ead8eaa7":{"tier":"enterprise","agent":"Tanko"},"sk-syslog-koby":{"tier":"enterprise","agent":"Koby","deprecated":true},"sk-eb3e6fc1c0-de1bf2edf35a53cb3749a2400483fdee":{"tier":"enterprise","agent":"Koby"},"sk-syslog-kagenz0":{"tier":"enterprise","agent":"Kagenz0","deprecated":true},"sk-12b66b3392-b548aed9138aeb6f698e8e521650ed9b":{"tier":"enterprise","agent":"Kagenz0"},"sk-syslog-koonimo":{"tier":"enterprise","agent":"Koonimo","deprecated":true},"sk-680d06686c-00ee8bf9dc3c93b276af122d49a14dfe":{"tier":"enterprise","agent":"Koonimo"},"sk-starter-abc123":{"tier":"starter","agent":"test-starter","deprecated":true},"sk-55da55907a-1bd7ff344e26feda50e9ac2219697860":{"tier":"starter","agent":"test-starter"},"sk-professional-xyz789":{"tier":"professional","agent":"test-pro","deprecated":true},"sk-b5159863e6-8df3ae52fb958cfe76cc2888c8c8e676":{"tier":"professional","agent":"test-pro"}}
|
||||
- ADMIN_KEY=sk-admin-ee09fffd04978b61a1569ac670c68814
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:9000/health')"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
interval: 30s
|
||||
timeout: 15s
|
||||
retries: 3
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_healthy
|
||||
|
||||
litellm:
|
||||
image: ghcr.io/berriai/litellm:main-stable
|
||||
image: ghcr.io/docker.litellm.ai/berriai/litellm:1.90.0-rc.1
|
||||
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
||||
container_name: harness-litellm
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "8081:4000"
|
||||
- "4000:4000"
|
||||
volumes:
|
||||
- ./litellm_config.yaml:/app/config.yaml
|
||||
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
|
||||
environment:
|
||||
- LITELLM_MASTER_KEY=sk-syslog-local-master-key
|
||||
- PROMETHEUS_EXPORTER=true
|
||||
- LITELLM_MASTER_KEY=sk-litellm-7f96080dd99b15c36bd4b333b58a6796
|
||||
- ROUTER_API_KEY=sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
- DATABASE_URL=postgresql://litellm:d9fc143e3dc1a7a8e672c359fea95c5e@postgres:5432/litellm
|
||||
- STORE_MODEL_IN_DB=True
|
||||
- LITELLM_UI_USERNAME=admin
|
||||
- LITELLM_UI_PASSWORD=syslog-admin-2026
|
||||
- UI_USERNAME=admin
|
||||
- UI_PASSWORD=syslog-admin-2026
|
||||
- OPENAI_API_KEY=not-used
|
||||
- PROXY_BASE_URL=https://litellm.sysloggh.net
|
||||
- DOCS_URL=/docs
|
||||
- ANTHROPIC_API_KEY=not-used
|
||||
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
|
||||
- GENERIC_CLIENT_SECRET=aDkXQx82duqpxc98xxqp0quzzUf1mawsnOTqj7sx1acaS7rWSt02N5ksBCi92n8ZilRavigoYME6fLakP20Ixc9H2pxnSZFiOqQLb7BPi8UtsvfxmzXklD0HJIdbKFxe
|
||||
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
|
||||
- GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/
|
||||
- GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/
|
||||
- GENERIC_SCOPE=openid email profile
|
||||
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
|
||||
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
- "auth.sysloggh.net:192.168.68.11"
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
|
||||
@@ -64,10 +110,13 @@ services:
|
||||
- "80:80"
|
||||
volumes:
|
||||
- ./nginx/nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
- ./dashboard:/opt/inference-harness/dashboard:ro
|
||||
extra_hosts:
|
||||
- "auth.sysloggh.net:192.168.68.11"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://127.0.0.1/health"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
interval: 30s
|
||||
timeout: 15s
|
||||
retries: 3
|
||||
depends_on:
|
||||
- litellm
|
||||
@@ -78,7 +127,7 @@ services:
|
||||
container_name: harness-dashboard
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "3000:3000"
|
||||
- "127.0.0.1:3000:3000"
|
||||
environment:
|
||||
- REDIS_URL=redis://redis:6379
|
||||
- GPU_SIDECARS=192.168.68.15:8090,192.168.68.8:8090,192.168.68.110:8090
|
||||
@@ -90,8 +139,7 @@ services:
|
||||
depends_on:
|
||||
- redis
|
||||
|
||||
|
||||
volumes:
|
||||
redis-data:
|
||||
|
||||
# LiteLLM command override to load config
|
||||
# (appended to fix config loading issue)
|
||||
pgdata:
|
||||
|
||||
@@ -1,106 +0,0 @@
|
||||
## Syslog GPU Router — Nginx Configuration (Docker-internal)
|
||||
## Routes incoming agent requests to the appropriate GPU backend
|
||||
## based on the X-Syslog-Model header.
|
||||
|
||||
upstream amdpve_pool {
|
||||
## Strix Halo 395 — qwen3.6-35B-A3B (MoE) — Default workhorse
|
||||
server 192.168.68.15:8080;
|
||||
}
|
||||
|
||||
upstream llmgpu_pool {
|
||||
## RTX 3090 — qwen3.5-27B (Dense) — Heavy reasoning
|
||||
server 192.168.68.8:8080;
|
||||
}
|
||||
|
||||
upstream ocu_llm_pool {
|
||||
## New Backend — gemma-4-12b (VLM) — Vision + light tasks
|
||||
server 192.168.68.110:8080;
|
||||
}
|
||||
|
||||
upstream queue_service {
|
||||
## Agent queue with circuit breaker (Docker container)
|
||||
server queue-service:8091;
|
||||
}
|
||||
|
||||
upstream dashboard_service {
|
||||
## Harness dashboard (Docker container)
|
||||
server dashboard:3001;
|
||||
}
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Mapping: X-Syslog-Model header → upstream backend
|
||||
## ------------------------------------------------------------------
|
||||
map $http_x_syslog_model $gpu_upstream {
|
||||
default amdpve_pool;
|
||||
"standard" amdpve_pool;
|
||||
"heavy" llmgpu_pool;
|
||||
"qwen3.5-27B" llmgpu_pool;
|
||||
"light" ocu_llm_pool;
|
||||
"gemma-4-12b" ocu_llm_pool;
|
||||
}
|
||||
|
||||
## Rate limit zone — 10 req/s per IP, burst of 20
|
||||
limit_req_zone $binary_remote_addr zone=perip:10m rate=10r/s;
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name _;
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Dashboard — observability UI (MUST be before / catch-all)
|
||||
## ------------------------------------------------------------------
|
||||
location /dashboard {
|
||||
proxy_pass http://dashboard_service/;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
}
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Main location — proxy to selected upstream
|
||||
## ------------------------------------------------------------------
|
||||
location / {
|
||||
limit_req zone=perip burst=20 nodelay;
|
||||
limit_req_status 503;
|
||||
proxy_pass http://$gpu_upstream;
|
||||
|
||||
## Preserve original host and headers
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
## Pass through the model header so backends can log it
|
||||
proxy_pass_header X-Syslog-Model;
|
||||
|
||||
## Streaming support (SSE for LLM responses)
|
||||
proxy_buffering off;
|
||||
proxy_cache off;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
|
||||
## Basic failover — retry on error or timeout
|
||||
proxy_next_upstream error timeout http_502 http_503;
|
||||
proxy_next_upstream_tries 2;
|
||||
|
||||
## Add a response header for observability
|
||||
add_header X-Routed-To $gpu_upstream always;
|
||||
|
||||
## Fallback to queue when all GPU upstreams are down
|
||||
error_page 502 503 504 = @queue_fallback;
|
||||
}
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Queue fallback — enqueue when GPUs are unavailable
|
||||
## ------------------------------------------------------------------
|
||||
location @queue_fallback {
|
||||
rewrite ^ /enqueue break;
|
||||
proxy_pass http://queue_service;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
proxy_set_header Content-Type $content_type;
|
||||
proxy_pass_request_body on;
|
||||
}
|
||||
}
|
||||
-106
@@ -1,106 +0,0 @@
|
||||
## Syslog GPU Router — Nginx Configuration
|
||||
## Routes incoming agent requests to the appropriate GPU backend
|
||||
## based on the X-Syslog-Model header.
|
||||
|
||||
upstream amdpve_pool {
|
||||
## Strix Halo 395 — qwen3.6-35B-A3B (MoE) — Default workhorse
|
||||
server 192.168.68.15:8080;
|
||||
}
|
||||
|
||||
upstream llmgpu_pool {
|
||||
## RTX 3090 — qwen3.5-27B (Dense) — Heavy reasoning
|
||||
server 192.168.68.8:8080;
|
||||
}
|
||||
|
||||
upstream ocu_llm_pool {
|
||||
## New Backend — gemma-4-12b (VLM) — Vision + light tasks
|
||||
server 192.168.68.110:8080;
|
||||
}
|
||||
|
||||
upstream queue_service {
|
||||
## Agent queue with circuit breaker (Docker container)
|
||||
server 127.0.0.1:8091;
|
||||
}
|
||||
|
||||
upstream dashboard_service {
|
||||
## Harness dashboard (Docker container)
|
||||
server 127.0.0.1:3001;
|
||||
}
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Mapping: X-Syslog-Model header → upstream backend
|
||||
## ------------------------------------------------------------------
|
||||
map $http_x_syslog_model $gpu_upstream {
|
||||
default amdpve_pool; # missing header → default workhorse
|
||||
"standard" amdpve_pool;
|
||||
"heavy" llmgpu_pool;
|
||||
"qwen3.5-27B" llmgpu_pool;
|
||||
"light" ocu_llm_pool;
|
||||
"gemma-4-12b" ocu_llm_pool;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 8080;
|
||||
server_name _;
|
||||
|
||||
# Rate limit zone — 10 req/s per IP, burst of 20
|
||||
limit_req_zone $binary_remote_addr zone=perip:10m rate=10r/s;
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Dashboard — observability UI (MUST be before / catch-all)
|
||||
## ------------------------------------------------------------------
|
||||
location /dashboard {
|
||||
proxy_pass http://dashboard_service/;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
}
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Main location — proxy to selected upstream
|
||||
## ------------------------------------------------------------------
|
||||
location / {
|
||||
limit_req zone=perip burst=20 nodelay;
|
||||
limit_req_status 503;
|
||||
proxy_pass http://$gpu_upstream;
|
||||
|
||||
## Preserve original host and headers
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
## Pass through the model header so backends can log it
|
||||
proxy_pass_header X-Syslog-Model;
|
||||
|
||||
## Streaming support (SSE for LLM responses)
|
||||
proxy_buffering off;
|
||||
proxy_cache off;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
|
||||
## Basic failover — retry on error or timeout
|
||||
proxy_next_upstream error timeout http_502 http_503;
|
||||
proxy_next_upstream_tries 2;
|
||||
|
||||
## Add a response header for observability
|
||||
add_header X-Routed-To $gpu_upstream always;
|
||||
|
||||
## Fallback to queue when all GPU upstreams are down
|
||||
error_page 502 503 504 = @queue_fallback;
|
||||
}
|
||||
|
||||
## ------------------------------------------------------------------
|
||||
## Queue fallback — enqueue when GPUs are unavailable
|
||||
## ------------------------------------------------------------------
|
||||
location @queue_fallback {
|
||||
rewrite ^ /enqueue break;
|
||||
proxy_pass http://queue_service;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
proxy_set_header Content-Type $content_type;
|
||||
proxy_pass_request_body on;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
models:
|
||||
gpu-vision:
|
||||
gpu_url: http://192.168.68.110:8080/v1
|
||||
gpu_host: 192.168.68.110
|
||||
label: Qwen3.5-9B Vision (RTX 5070)
|
||||
max_concurrent: 1
|
||||
context: 131072
|
||||
tiers: [starter, professional, enterprise]
|
||||
capabilities: [completion, multimodal]
|
||||
model_path: /home/llmuser/models/qwen3.5-9b/Qwen3.5-9B-Q5_K_M.gguf
|
||||
args: --mmproj /home/llmuser/models/qwen3.5-9b/mmproj-F16.gguf --ctx-size 131072 --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn 1 --parallel 1 --alias gpu-light --reasoning off --api-key not-needed --cache-prompt
|
||||
|
||||
qwen3.6-27B-code:
|
||||
gpu_url: http://192.168.68.8:8080/v1
|
||||
sidecar_url: http://192.168.68.8:8090
|
||||
gpu_host: 192.168.68.8
|
||||
label: Qwen3.6 27B Code (RTX 3090)
|
||||
max_concurrent: 2
|
||||
context: 262144
|
||||
tiers: [professional, enterprise]
|
||||
capabilities: [completion]
|
||||
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
|
||||
args: --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
|
||||
|
||||
qwen3.6-35B-udq4:
|
||||
gpu_url: http://192.168.68.15:8080/v1
|
||||
sidecar_url: http://192.168.68.15:8090
|
||||
gpu_host: 192.168.68.15
|
||||
label: Qwen3.6 35B UD-Q4_K_M (Strix Halo)
|
||||
max_concurrent: 1
|
||||
context: 262144
|
||||
tiers: [professional, enterprise]
|
||||
capabilities: [completion, multimodal]
|
||||
model_path: /models/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf
|
||||
args: -c 262144 -ngl 99 --flash-attn on
|
||||
|
||||
hosts:
|
||||
gpu-light:
|
||||
address: 192.168.68.110
|
||||
gpu_name: NVIDIA GeForce RTX 5070
|
||||
vram_gb: 12
|
||||
current_model: gpu-vision
|
||||
|
||||
gpu-dense:
|
||||
address: 192.168.68.8
|
||||
gpu_name: NVIDIA GeForce RTX 3090
|
||||
vram_gb: 24
|
||||
current_model: qwen3.6-27B-code
|
||||
|
||||
gpu-moe:
|
||||
address: 192.168.68.15
|
||||
gpu_name: AMD Strix Halo (iGPU)
|
||||
vram_gb: 64
|
||||
current_model: qwen3.6-35B-udq4
|
||||
+173
-16
@@ -1,21 +1,178 @@
|
||||
general_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
store_model_in_db: false
|
||||
user_url_allowed_hosts:
|
||||
- 192.168.68.14
|
||||
- 192.168.68.14:5080
|
||||
guardrails:
|
||||
- guardrail_name: input-moderation
|
||||
litellm_params:
|
||||
guardrail: openai_moderation
|
||||
mode: pre_call
|
||||
- guardrail_name: output-moderation
|
||||
litellm_params:
|
||||
guardrail: openai_moderation
|
||||
mode: post_call
|
||||
- guardrail_name: harmful-content-filter
|
||||
litellm_params:
|
||||
categories:
|
||||
- action: BLOCK
|
||||
category: harmful_self_harm
|
||||
enabled: true
|
||||
severity_threshold: medium
|
||||
- action: BLOCK
|
||||
category: harmful_violence
|
||||
enabled: true
|
||||
severity_threshold: medium
|
||||
- action: BLOCK
|
||||
category: harmful_illegal_weapons
|
||||
enabled: true
|
||||
severity_threshold: medium
|
||||
guardrail: litellm_content_filter
|
||||
mode: pre_call
|
||||
litellm_settings:
|
||||
user_url_allowed_hosts:
|
||||
- 192.168.68.14
|
||||
- 192.168.68.14:5080
|
||||
cache: true
|
||||
cache_params:
|
||||
host: harness-redis
|
||||
namespace: litellm
|
||||
port: 6379
|
||||
ttl: 600
|
||||
type: redis
|
||||
drop_params: true
|
||||
success_callback:
|
||||
- prometheus
|
||||
failure_callback:
|
||||
- prometheus
|
||||
model_cost:
|
||||
qwen3.6-27B-code:
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
strix-moe:
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
syslog-auto:
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
num_retries: 2
|
||||
request_timeout: 600
|
||||
sso_callback: /sso/callback
|
||||
model_list:
|
||||
- model_name: qwen3.6-35B-A3B
|
||||
litellm_params:
|
||||
model: openai/qwen3.6-35B-A3B
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
- model_name: gpu-dense
|
||||
litellm_params:
|
||||
model: openai/qwen3.6-27B-code-text
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
- model_name: gpu-light
|
||||
litellm_params:
|
||||
model: openai/gemma-4-12b
|
||||
model: openai/qwen3.6-27B-code
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
model_name: qwen3.6-27B-code
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/strix-moe
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: qwen3.6-35B-udq4
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/strix-moe
|
||||
rpm: 40
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: strix-moe
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: gpu-dense
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
general_settings:
|
||||
master_key: sk-syslog-local-master-key
|
||||
litellm_settings:
|
||||
drop_params: true
|
||||
request_timeout: 120
|
||||
model: openai/gpu-vision
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: gpu-vision
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
weight: 0.70
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: qwen3.8-27B-uncensored
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/strix-moe
|
||||
rpm: 60
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.20
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/gpu-vision
|
||||
rpm: 200
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.10
|
||||
model_name: syslog-auto
|
||||
router_settings:
|
||||
allowed_fails: 100
|
||||
enable_loadbalancing_on_proxy: true
|
||||
fallbacks:
|
||||
- syslog-auto:
|
||||
- qwen3.6-27B-code
|
||||
- strix-moe
|
||||
- gpu-vision
|
||||
- qwen3.6-27B-code:
|
||||
- strix-moe
|
||||
- strix-moe:
|
||||
- qwen3.6-27B-code
|
||||
- gpu-vision
|
||||
request_timeout: 300
|
||||
routing_strategy: simple-shuffle
|
||||
|
||||
agents:
|
||||
- agent_name: agent-zero-homelab
|
||||
agent_card_params:
|
||||
name: Agent Zero HomeLab
|
||||
url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab
|
||||
protocolVersion: '1.0'
|
||||
|
||||
+123
-68
@@ -1,118 +1,173 @@
|
||||
worker_processes auto;
|
||||
error_log /var/log/nginx/error.log warn;
|
||||
pid /var/run/nginx.pid;
|
||||
|
||||
events { worker_connections 1024; }
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
|
||||
log_format main '$remote_addr - $remote_user [$time_local] "$request" '
|
||||
'$status $body_bytes_sent "$http_referer" '
|
||||
'"$http_user_agent" rt=$request_time';
|
||||
access_log /var/log/nginx/access.log main;
|
||||
error_log /var/log/nginx/error.log;
|
||||
sendfile on;
|
||||
keepalive_timeout 65;
|
||||
|
||||
upstream router_api { server host.docker.internal:9000; }
|
||||
upstream dashboard_ui { server dashboard:3000; }
|
||||
upstream litellm_backend { server litellm:4000; }
|
||||
resolver 127.0.0.11 valid=30s;
|
||||
|
||||
map $host $dashboard_ui_url {
|
||||
default http://harness-dashboard:3000;
|
||||
}
|
||||
map $host $litellm_backend_url {
|
||||
default http://harness-litellm:4000;
|
||||
}
|
||||
|
||||
# ════════════════════════════════════════════════════════════
|
||||
# Server :80 — Lean single-layer entrypoint
|
||||
# All API paths route directly to LiteLLM.
|
||||
# harness-router fully deprecated.
|
||||
# ════════════════════════════════════════════════════════════
|
||||
server {
|
||||
listen 80;
|
||||
|
||||
# Security headers
|
||||
add_header X-Content-Type-Options nosniff always;
|
||||
add_header X-Frame-Options SAMEORIGIN always;
|
||||
add_header X-XSS-Protection "1; mode=block" always;
|
||||
# Authentik OIDC subrequest
|
||||
location /authentik/auth {
|
||||
internal;
|
||||
proxy_pass https://auth.sysloggh.net/outpost.goauthentik.io/auth/nginx;
|
||||
proxy_pass_request_body off;
|
||||
proxy_set_header Content-Length "";
|
||||
proxy_set_header X-Original-URL $scheme://$http_host$request_uri;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
}
|
||||
|
||||
# Disable buffering for SSE streams
|
||||
proxy_buffering off;
|
||||
|
||||
# API through router
|
||||
# ── LiteLLM API (replaces router /v1/) ──
|
||||
location /v1/ {
|
||||
proxy_pass http://router_api;
|
||||
proxy_pass $litellm_backend_url;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header Authorization $http_authorization;
|
||||
proxy_connect_timeout 10s;
|
||||
proxy_read_timeout 600s;
|
||||
proxy_buffering off;
|
||||
}
|
||||
|
||||
# ── LiteLLM admin (replaces router /admin/) ──
|
||||
location /admin/ {
|
||||
proxy_pass http://router_api;
|
||||
proxy_pass $litellm_backend_url;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header Authorization $http_authorization;
|
||||
proxy_read_timeout 600s;
|
||||
}
|
||||
|
||||
# SSE streaming endpoint
|
||||
# ── LiteLLM stream ──
|
||||
location /stream {
|
||||
proxy_pass http://router_api;
|
||||
proxy_pass $litellm_backend_url/stream;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header Connection "";
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_buffering off;
|
||||
chunked_transfer_encoding off;
|
||||
}
|
||||
|
||||
# Dashboard API proxy for SSE
|
||||
# ── API passthrough ──
|
||||
location /api/ {
|
||||
proxy_pass http://dashboard_ui;
|
||||
proxy_pass $litellm_backend_url/;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_buffering off;
|
||||
}
|
||||
|
||||
# LiteLLM debug
|
||||
location /litellm/ {
|
||||
rewrite ^/litellm/(.*) /$1 break;
|
||||
proxy_pass http://litellm_backend;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header Authorization $http_authorization;
|
||||
}
|
||||
|
||||
# Professional Dashboard (Phase 1-3) - Static HTML served via Nginx
|
||||
# ── Dashboard ──
|
||||
location /dashboard/ {
|
||||
alias /opt/inference-harness/dashboard/;
|
||||
index dashboard.html;
|
||||
add_header Cache-Control "public, max-age=3600";
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
}
|
||||
|
||||
# Legacy Dashboard (root) - Proxy to Flask app
|
||||
location / {
|
||||
proxy_pass http://dashboard_ui;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_buffering off;
|
||||
}
|
||||
|
||||
# Performance analytics
|
||||
location /metrics/ {
|
||||
proxy_pass http://router_api;
|
||||
proxy_pass $dashboard_ui_url/;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
# Circuit Breaker metrics (Phase 1)
|
||||
location /metrics/circuit-breaker {
|
||||
proxy_pass http://router_api/metrics/circuit-breaker;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
location /health {
|
||||
proxy_pass http://router_api/health;
|
||||
# ── GPU Fleet Dashboard ──
|
||||
location /gpu/ {
|
||||
proxy_pass http://192.168.68.24:9100/;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_read_timeout 60s;
|
||||
}
|
||||
|
||||
# ── LiteLLM redirect ──
|
||||
location = /litellm {
|
||||
return 301 /litellm/;
|
||||
}
|
||||
|
||||
# ── Health: redirect /health (auth-required) → /health/liveliness (no-auth) ──
|
||||
location = /litellm/health {
|
||||
return 301 /litellm/health/liveliness;
|
||||
}
|
||||
|
||||
# ── LiteLLM static assets ──
|
||||
location /litellm-asset-prefix/ {
|
||||
proxy_pass $litellm_backend_url;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_connect_timeout 10s;
|
||||
proxy_read_timeout 60s;
|
||||
}
|
||||
|
||||
# ── LiteLLM admin UI and API (strips /litellm prefix) ──
|
||||
location /litellm/ {
|
||||
rewrite ^/litellm(/.*)$ $1 break;
|
||||
proxy_pass $litellm_backend_url;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_buffering off;
|
||||
proxy_read_timeout 600s;
|
||||
proxy_set_header Authorization $http_authorization;
|
||||
}
|
||||
|
||||
# ── Auth proxy to Authentik ──
|
||||
location /application/o/ {
|
||||
proxy_pass https://192.168.68.11;
|
||||
proxy_ssl_verify off;
|
||||
proxy_set_header Host auth.sysloggh.net;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_connect_timeout 10s;
|
||||
proxy_read_timeout 30s;
|
||||
}
|
||||
|
||||
# ── Prometheus metrics (LiteLLM exposes at /metrics) ──
|
||||
location /metrics {
|
||||
proxy_pass $litellm_backend_url/metrics;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
# ---- Legacy /health/unified alias (harness-router decommissioned) ----
|
||||
# Retained intentionally: 5+ live GPU monitor contracts poll
|
||||
# this path every 15s and follow the 301 to /gpu/gpu-data.
|
||||
location /health/unified {
|
||||
return 301 /gpu/gpu-data;
|
||||
}
|
||||
|
||||
# ── Health: no-auth LiteLLM liveliness ──
|
||||
location /health {
|
||||
proxy_pass $litellm_backend_url/health/liveliness;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
|
||||
# ── UI / Docs convenience redirects (additive 2026-09-11) ──
|
||||
# Canonical app path is /litellm/. These 301s make bare
|
||||
# /ui/ and /docs resolve. No new auth surface; no OIDC impact.
|
||||
location = /ui { return 301 /litellm/ui/; }
|
||||
location /ui/ { return 301 /litellm$request_uri; }
|
||||
location = /docs { return 301 /litellm/docs; }
|
||||
location = /docs/ { return 301 /litellm/docs; }
|
||||
|
||||
# ── 404 for everything else ──
|
||||
location / {
|
||||
return 404;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,121 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Syslog Inference Queue Service — Circuit breaker + request queuing.
|
||||
|
||||
Ports: 8091
|
||||
Endpoints:
|
||||
/health — liveness probe (Nginx upstream check)
|
||||
/enqueue — POST inference request into queue (fallback from Nginx)
|
||||
/status — GET queue depth + circuit breaker state
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
from flask import Flask, request, jsonify
|
||||
|
||||
app = Flask(__name__)
|
||||
|
||||
# Configuration
|
||||
REDIS_HOST = os.getenv("REDIS_HOST", "192.168.68.7")
|
||||
REDIS_PORT = int(os.getenv("REDIS_PORT", "6379"))
|
||||
QUEUE_KEY = "inference:requests"
|
||||
CIRCUIT_OPEN_THRESHOLD = 50
|
||||
CIRCUIT_WARN_THRESHOLD = 30
|
||||
|
||||
# GPU endpoints for draining
|
||||
GPUS = {
|
||||
"amdpve": "192.168.68.15:8080",
|
||||
"llmgpu": "192.168.68.8:8080",
|
||||
"ocu_llm": "192.168.68.110:8080",
|
||||
}
|
||||
|
||||
|
||||
def get_redis():
|
||||
try:
|
||||
import redis
|
||||
return redis.Redis(host=REDIS_HOST, port=REDIS_PORT, decode_responses=True)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def get_queue_depth(r):
|
||||
try:
|
||||
return r.llen(QUEUE_KEY)
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
|
||||
def check_gpu_health(endpoint):
|
||||
try:
|
||||
req = urllib.request.Request(f"http://{endpoint}/v1/models")
|
||||
req.add_header("User-Agent", "queue-service/1.0")
|
||||
resp = urllib.request.urlopen(req, timeout=3)
|
||||
return resp.status == 200
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
@app.route("/health")
|
||||
def health():
|
||||
"""Nginx upstream health probe. Returns 200 if service is alive."""
|
||||
return jsonify({"status": "ok", "service": "queue-service"}), 200
|
||||
|
||||
|
||||
@app.route("/enqueue", methods=["POST"])
|
||||
def enqueue():
|
||||
"""Fallback endpoint — Nginx calls this when all GPU upstreams are down."""
|
||||
r = get_redis()
|
||||
if not r:
|
||||
return jsonify({"error": "Redis unavailable"}), 503
|
||||
|
||||
depth = get_queue_depth(r)
|
||||
if depth >= CIRCUIT_OPEN_THRESHOLD:
|
||||
return jsonify({
|
||||
"error": "Circuit breaker OPEN",
|
||||
"queue_depth": depth,
|
||||
"threshold": CIRCUIT_OPEN_THRESHOLD
|
||||
}), 503
|
||||
|
||||
# Store the request in queue
|
||||
payload = request.get_data(as_text=True)
|
||||
headers = {k: v for k, v in request.headers if k.startswith("X-")}
|
||||
r.rpush(QUEUE_KEY, json.dumps({
|
||||
"payload": payload,
|
||||
"headers": headers,
|
||||
"queued_at": time.time()
|
||||
}))
|
||||
|
||||
new_depth = get_queue_depth(r)
|
||||
return jsonify({
|
||||
"status": "queued",
|
||||
"position": new_depth,
|
||||
"circuit": "warn" if new_depth >= CIRCUIT_WARN_THRESHOLD else "closed"
|
||||
}), 202
|
||||
|
||||
|
||||
@app.route("/status")
|
||||
def status():
|
||||
"""GET queue depth + circuit breaker state + GPU health."""
|
||||
r = get_redis()
|
||||
depth = get_queue_depth(r) if r else -1
|
||||
circuit = "open" if depth >= CIRCUIT_OPEN_THRESHOLD else ("warn" if depth >= CIRCUIT_WARN_THRESHOLD else "closed")
|
||||
|
||||
gpu_health = {}
|
||||
for name, endpoint in GPUS.items():
|
||||
gpu_health[name] = "up" if check_gpu_health(endpoint) else "down"
|
||||
|
||||
return jsonify({
|
||||
"queue_depth": depth,
|
||||
"circuit_breaker": circuit,
|
||||
"gpu_health": gpu_health,
|
||||
"thresholds": {
|
||||
"warn": CIRCUIT_WARN_THRESHOLD,
|
||||
"open": CIRCUIT_OPEN_THRESHOLD
|
||||
}
|
||||
})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app.run(host="0.0.0.0", port=8091)
|
||||
@@ -0,0 +1,75 @@
|
||||
# GPU Roster Loader - reads gpu_roster.yaml via PyYAML
|
||||
# Hot-reloadable via /admin/roster/reload endpoint
|
||||
|
||||
import os, json, threading, time
|
||||
|
||||
try:
|
||||
import yaml
|
||||
except ImportError:
|
||||
yaml = None
|
||||
|
||||
ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml')
|
||||
_last_mtime = 0
|
||||
_lock = threading.Lock()
|
||||
|
||||
GPU_URLS = {}
|
||||
GPU_SIDECARS = {}
|
||||
GPU_LABELS = {}
|
||||
GPU_MAX_CONCURRENT = {}
|
||||
GPU_CONTEXT = {}
|
||||
TIER_MODELS = {}
|
||||
HOSTS = {}
|
||||
|
||||
def load_roster(path=None):
|
||||
global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT
|
||||
global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime
|
||||
if yaml is None:
|
||||
return False, 'PyYAML not installed - run: pip3 install pyyaml'
|
||||
path = path or ROSTER_PATH
|
||||
try:
|
||||
with open(path) as f:
|
||||
data = yaml.safe_load(f)
|
||||
with _lock:
|
||||
GPU_URLS.clear()
|
||||
GPU_SIDECARS.clear()
|
||||
GPU_LABELS.clear()
|
||||
GPU_MAX_CONCURRENT.clear()
|
||||
GPU_CONTEXT.clear()
|
||||
TIER_MODELS.clear()
|
||||
models = data.get('models', {})
|
||||
for name, cfg in models.items():
|
||||
GPU_URLS[name] = cfg.get('gpu_url', '')
|
||||
GPU_SIDECARS[name] = cfg.get('sidecar_url', '')
|
||||
GPU_LABELS[name] = cfg.get('label', name)
|
||||
GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1)
|
||||
GPU_CONTEXT[name] = cfg.get('context', 65536)
|
||||
tiers = cfg.get('tiers', ['enterprise'])
|
||||
for t in tiers:
|
||||
if t not in TIER_MODELS:
|
||||
TIER_MODELS[t] = []
|
||||
if name not in TIER_MODELS[t]:
|
||||
TIER_MODELS[t].append(name)
|
||||
HOSTS.clear()
|
||||
HOSTS.update(data.get('hosts', {}))
|
||||
_last_mtime = os.path.getmtime(path)
|
||||
return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS))
|
||||
except Exception as e:
|
||||
return False, str(e)
|
||||
|
||||
def check_reload():
|
||||
global _last_mtime
|
||||
try:
|
||||
mtime = os.path.getmtime(ROSTER_PATH)
|
||||
if mtime > _last_mtime:
|
||||
success, msg = load_roster()
|
||||
if success:
|
||||
print('[ROSTER] Auto-reloaded:', msg)
|
||||
except:
|
||||
pass
|
||||
|
||||
def reload_thread(interval=30):
|
||||
while True:
|
||||
time.sleep(interval)
|
||||
check_reload()
|
||||
|
||||
threading.Thread(target=reload_thread, daemon=True).start()
|
||||
@@ -1,9 +0,0 @@
|
||||
FROM python:3.12-slim
|
||||
|
||||
WORKDIR /app
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
COPY router.py .
|
||||
|
||||
EXPOSE 9000
|
||||
CMD ["python", "router.py"]
|
||||
@@ -1,90 +0,0 @@
|
||||
# Insert streaming support before the gpu_resp call
|
||||
import re
|
||||
with open('/opt/inference-harness/router/router.py') as f:
|
||||
code = f.read()
|
||||
|
||||
# Find the gpu_resp block and replace with streaming-aware version
|
||||
old = ''' start = time.time()
|
||||
gpu_resp = requests.post(
|
||||
gpu_url + "/chat/completions",
|
||||
json=req_data,
|
||||
headers={"Content-Type": "application/json", "Authorization": "Bearer not-needed"},
|
||||
timeout=120,
|
||||
)
|
||||
latency_ms = int((time.time() - start) * 1000)
|
||||
|
||||
if gpu_resp.status_code != 200:
|
||||
log.error("GPU error: %s %s", gpu_resp.status_code, gpu_resp.text[:200])
|
||||
return jsonify({"error": "GPU backend returned " + str(gpu_resp.status_code)}), 502
|
||||
|
||||
response_data = gpu_resp.json()
|
||||
response_data = fix_reasoning_content(response_data)
|
||||
|
||||
response_data["routing"] = {
|
||||
"model": model, "reason": reason, "gpu": gpu_url,
|
||||
"tier": tier, "agent": agent, "latency_ms": latency_ms,
|
||||
}
|
||||
|
||||
return jsonify(response_data)'''
|
||||
|
||||
new = ''' start = time.time()
|
||||
is_stream = req_data.get("stream", False)
|
||||
|
||||
gpu_resp = requests.post(
|
||||
gpu_url + "/chat/completions",
|
||||
json=req_data,
|
||||
headers={"Content-Type": "application/json", "Authorization": "Bearer not-needed"},
|
||||
timeout=120,
|
||||
stream=is_stream,
|
||||
)
|
||||
latency_ms = int((time.time() - start) * 1000)
|
||||
|
||||
if gpu_resp.status_code != 200:
|
||||
log.error("GPU error: %s %s", gpu_resp.status_code, gpu_resp.text[:200])
|
||||
return jsonify({"error": "GPU backend returned " + str(gpu_resp.status_code)}), 502
|
||||
|
||||
if is_stream:
|
||||
# Stream response back to client
|
||||
def generate():
|
||||
first = True
|
||||
for line in gpu_resp.iter_lines(decode_unicode=True):
|
||||
if line:
|
||||
if first and line.startswith("data: "):
|
||||
# Inject routing into first chunk
|
||||
try:
|
||||
chunk = json.loads(line[6:])
|
||||
chunk["routing"] = {
|
||||
"model": model, "reason": reason, "gpu": gpu_url,
|
||||
"tier": tier, "agent": agent, "latency_ms": latency_ms,
|
||||
}
|
||||
yield "data: " + json.dumps(chunk) + "\n\n"
|
||||
first = False
|
||||
continue
|
||||
except Exception:
|
||||
pass
|
||||
yield line + "\n"
|
||||
yield "data: [DONE]\n\n"
|
||||
return Response(stream_with_context(generate()), mimetype="text/event-stream")
|
||||
|
||||
response_data = gpu_resp.json()
|
||||
response_data = fix_reasoning_content(response_data)
|
||||
|
||||
response_data["routing"] = {
|
||||
"model": model, "reason": reason, "gpu": gpu_url,
|
||||
"tier": tier, "agent": agent, "latency_ms": latency_ms,
|
||||
}
|
||||
|
||||
return jsonify(response_data)'''
|
||||
|
||||
code = code.replace(old, new)
|
||||
|
||||
# Add missing import
|
||||
if 'from flask import Flask, request, jsonify' in code:
|
||||
code = code.replace(
|
||||
'from flask import Flask, request, jsonify',
|
||||
'from flask import Flask, request, jsonify, Response, stream_with_context'
|
||||
)
|
||||
|
||||
with open('/opt/inference-harness/router/router.py', 'w') as f:
|
||||
f.write(code)
|
||||
print('Streaming support added')
|
||||
@@ -1,3 +0,0 @@
|
||||
flask==3.1.*
|
||||
redis==5.2.*
|
||||
requests==2.32.*
|
||||
@@ -1,77 +0,0 @@
|
||||
def route(rd, tier, agent=""):
|
||||
msgs = rd.get("messages",[]); t = estimate_tokens(msgs)
|
||||
sys = any(m.get("role")=="system" for m in msgs)
|
||||
turns = len([m for m in msgs if m.get("role") in ("user","assistant")])
|
||||
hints = rd.get("routing_hints",{})
|
||||
allowed = TIER_MODELS.get(tier, ["gemma-4-12b"])
|
||||
avail = [m for m in available_models() if m in allowed]
|
||||
if not avail: return {"model": allowed[0], "reason": "all_saturated", "saturated": True}
|
||||
if all(is_gpu_busy(m) for m in avail):
|
||||
return {"model": avail[0], "reason": "all_saturated", "saturated": True}
|
||||
|
||||
# GUARD: multimodal -> VLM only (sole vision model)
|
||||
has_image = any(
|
||||
isinstance(m.get("content"), list) and
|
||||
any(p.get("type") == "image_url" for p in m["content"] if isinstance(p, dict))
|
||||
for m in msgs
|
||||
)
|
||||
if has_image:
|
||||
if "gemma-4-12b" in avail and not is_gpu_busy("gemma-4-12b"):
|
||||
return {"model": "gemma-4-12b", "reason": "vision"}
|
||||
elif "gemma-4-12b" in avail:
|
||||
return {"model": "gemma-4-12b", "reason": "vision_saturated", "saturated": True}
|
||||
else:
|
||||
return {"model": allowed[0], "reason": "vision_unavailable"}
|
||||
|
||||
req = rd.get("model","auto")
|
||||
if req != "auto":
|
||||
target = req if req in avail else avail[0]
|
||||
if is_gpu_busy(target) and req in allowed:
|
||||
alts = [m for m in avail if m != target and m in allowed]
|
||||
if alts:
|
||||
alt = select_best_gpu(alts, "explicit", agent)
|
||||
if alt: return alt
|
||||
return {"model": target, "reason": "explicit"}
|
||||
|
||||
if hints:
|
||||
if hints.get("priority")=="speed" and "gemma-4-12b" in avail:
|
||||
return select_best_gpu(["gemma-4-12b"], "hint_speed", agent) or {"model":"gemma-4-12b","reason":"hint_speed"}
|
||||
if hints.get("priority")=="quality" and "qwen3.6-35B-A3B" in avail:
|
||||
return select_best_gpu(["qwen3.6-35B-A3B"], "hint_quality", agent) or {"model":"qwen3.6-35B-A3B","reason":"hint_quality"}
|
||||
if hints.get("priority")=="code" and "qwen3.6-27B-code" in avail:
|
||||
return select_best_gpu(["qwen3.6-27B-code"], "hint_code", agent) or {"model":"qwen3.6-27B-code","reason":"hint_code"}
|
||||
|
||||
first_msg = msgs[0].get("content","") if msgs else ""
|
||||
words = len(first_msg.split()) if isinstance(first_msg, str) else 99
|
||||
|
||||
# TIER 1: Tiny - single-turn micro queries -> VLM (fastest)
|
||||
if not sys and turns <= 1 and t <= 300 and words <= 100 and "gemma-4-12b" in avail:
|
||||
if not is_gpu_busy("gemma-4-12b"):
|
||||
return {"model":"gemma-4-12b","reason":"tiny"}
|
||||
fallback = [m for m in ["qwen3.6-27B-code","qwen3.6-35B-A3B"] if m in avail]
|
||||
result = select_best_gpu(fallback, "tiny_fallback", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 2: Light - moderate chat -> Dense primary, saves VLM for vision/speed
|
||||
if t <= 5000 and turns <= 4:
|
||||
candidates = [m for m in ["qwen3.6-27B-code","gemma-4-12b","qwen3.6-35B-A3B"] if m in avail]
|
||||
result = select_best_gpu(candidates, "light", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 3: Medium - quality matters -> MoE primary, Dense fallback
|
||||
if t <= 30000:
|
||||
candidates = [m for m in ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"] if m in avail]
|
||||
result = select_best_gpu(candidates, "medium", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 4: Heavy - big context needs big model -> MoE->Dense->VLM
|
||||
if t > 30000:
|
||||
candidates = [m for m in ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"] if m in avail]
|
||||
result = select_best_gpu(candidates, "heavy", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 5: Default - best quality first
|
||||
candidates = [m for m in ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"] if m in avail]
|
||||
result = select_best_gpu(candidates, "default", agent)
|
||||
if result: return result
|
||||
return {"model":avail[0],"reason":"last_resort"}
|
||||
@@ -1,92 +0,0 @@
|
||||
def route(rd, tier, agent=""):
|
||||
msgs = rd.get("messages",[]); t = estimate_tokens(msgs)
|
||||
sys = any(m.get("role")=="system" for m in msgs)
|
||||
turns = len([m for m in msgs if m.get("role") in ("user","assistant")])
|
||||
hints = rd.get("routing_hints",{})
|
||||
allowed = TIER_MODELS.get(tier, ["gemma-4-12b"])
|
||||
avail = [m for m in available_models() if m in allowed]
|
||||
if not avail: return {"model": allowed[0], "reason": "all_saturated", "saturated": True}
|
||||
if all(is_gpu_busy(m) for m in avail):
|
||||
return {"model": avail[0], "reason": "all_saturated", "saturated": True}
|
||||
|
||||
# GUARD: multimodal -> VLM only (sole vision model)
|
||||
has_image = any(
|
||||
isinstance(m.get("content"), list) and
|
||||
any(p.get("type") == "image_url" for p in m["content"] if isinstance(p, dict))
|
||||
for m in msgs
|
||||
)
|
||||
if has_image:
|
||||
if "gemma-4-12b" in avail and not is_gpu_busy("gemma-4-12b"):
|
||||
return {"model": "gemma-4-12b", "reason": "vision"}
|
||||
elif "gemma-4-12b" in avail:
|
||||
return {"model": "gemma-4-12b", "reason": "vision_saturated", "saturated": True}
|
||||
else:
|
||||
return {"model": allowed[0], "reason": "vision_unavailable"}
|
||||
|
||||
req = rd.get("model","auto")
|
||||
if req != "auto":
|
||||
target = req if req in avail else avail[0]
|
||||
if is_gpu_busy(target) and req in allowed:
|
||||
alts = [m for m in avail if m != target and m in allowed]
|
||||
if alts:
|
||||
alt = select_best_gpu(alts, "explicit", agent)
|
||||
if alt: return alt
|
||||
return {"model": target, "reason": "explicit"}
|
||||
|
||||
if hints:
|
||||
if hints.get("priority")=="speed" and "gemma-4-12b" in avail:
|
||||
return select_best_gpu(["gemma-4-12b"], "hint_speed", agent) or {"model":"gemma-4-12b","reason":"hint_speed"}
|
||||
if hints.get("priority")=="quality" and "qwen3.6-35B-A3B" in avail:
|
||||
return select_best_gpu(["qwen3.6-35B-A3B"], "hint_quality", agent) or {"model":"qwen3.6-35B-A3B","reason":"hint_quality"}
|
||||
if hints.get("priority")=="code" and "qwen3.6-27B-code" in avail:
|
||||
return select_best_gpu(["qwen3.6-27B-code"], "hint_code", agent) or {"model":"qwen3.6-27B-code","reason":"hint_code"}
|
||||
|
||||
first_msg = msgs[0].get("content","") if msgs else ""
|
||||
words = len(first_msg.split()) if isinstance(first_msg, str) else 99
|
||||
|
||||
# TIER 1: Tiny - single-turn micro queries -> VLM (fastest)
|
||||
if not sys and turns <= 1 and t <= 300 and words <= 100 and "gemma-4-12b" in avail:
|
||||
if not is_gpu_busy("gemma-4-12b"):
|
||||
return {"model":"gemma-4-12b","reason":"tiny"}
|
||||
fallback = [m for m in ["qwen3.6-27B-code","qwen3.6-35B-A3B"] if m in avail]
|
||||
result = select_best_gpu(fallback, "tiny_fallback", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 2: Light - moderate chat -> Dense primary, saves VLM for vision/speed
|
||||
if t <= 5000 and turns <= 4:
|
||||
candidates = [m for m in ["qwen3.6-27B-code","gemma-4-12b","qwen3.6-35B-A3B"] if m in avail]
|
||||
result = select_best_gpu(candidates, "light", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 3: Medium - quality matters -> MoE primary (60%), Dense spillover (40%)
|
||||
if t <= 30000:
|
||||
candidates = moe_spillover(avail, ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"])
|
||||
result = select_best_gpu(candidates, "medium", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 4: Heavy - big context -> MoE primary (60%), Dense spillover (40%)
|
||||
if t > 30000:
|
||||
candidates = moe_spillover(avail, ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"])
|
||||
result = select_best_gpu(candidates, "heavy", agent)
|
||||
if result: return result
|
||||
|
||||
# TIER 5: Default - MoE primary (60%), Dense spillover (40%)
|
||||
candidates = moe_spillover(avail, ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"])
|
||||
result = select_best_gpu(candidates, "default", agent)
|
||||
if result: return result
|
||||
return {"model":avail[0],"reason":"last_resort"}
|
||||
|
||||
|
||||
def moe_spillover(avail, default_order):
|
||||
"""Spill 40% of MoE-first traffic to Dense to prevent Strix Halo overheating.
|
||||
Only applies when MoE is first candidate, available, and not busy."""
|
||||
import random
|
||||
if (default_order[0] == "qwen3.6-35B-A3B"
|
||||
and "qwen3.6-35B-A3B" in avail
|
||||
and not is_gpu_busy("qwen3.6-35B-A3B")
|
||||
and "qwen3.6-27B-code" in avail
|
||||
and not is_gpu_busy("qwen3.6-27B-code")
|
||||
and random.random() < 0.4):
|
||||
# Swap: Dense first, MoE second
|
||||
return ["qwen3.6-27B-code","qwen3.6-35B-A3B"] + [m for m in default_order[2:] if m in avail and m not in ("qwen3.6-27B-code","qwen3.6-35B-A3B")]
|
||||
return [m for m in default_order if m in avail]
|
||||
-1149
File diff suppressed because it is too large
Load Diff
Executable
+56
@@ -0,0 +1,56 @@
|
||||
#!/bin/bash
|
||||
# Daily LiteLLM cost snapshot — cron at midnight
|
||||
# Writes to /var/log/litellm/costs.csv on CT 116
|
||||
|
||||
MASTER="sk-litellm-7f96080dd99b15c36bd4b333b58a6796"
|
||||
BASE="http://127.0.0.1:4001"
|
||||
CSV="/var/log/litellm/costs.csv"
|
||||
DATE=$(date -I)
|
||||
YESTERDAY=$(date -I -d yesterday)
|
||||
|
||||
# Fetch yesterday's spend by key
|
||||
RESP=$(curl -s "${BASE}/global/spend/logs?start_date=${YESTERDAY}T00:00:00&end_date=${YESTERDAY}T23:59:59&page_size=1000" \
|
||||
-H "Authorization: Bearer $MASTER" 2>/dev/null)
|
||||
|
||||
# Parse total spend
|
||||
TOTAL=$(echo "$RESP" | python3 -c "
|
||||
import sys, json
|
||||
data = json.load(sys.stdin)
|
||||
total = sum(r.get('spend', 0) or 0 for r in data.get('data', []))
|
||||
print(f'{total:.6f}')
|
||||
" 2>/dev/null)
|
||||
|
||||
# Parse per-model spend
|
||||
PER_MODEL=$(echo "$RESP" | python3 -c "
|
||||
import sys, json
|
||||
from collections import defaultdict
|
||||
data = json.load(sys.stdin)
|
||||
spend = defaultdict(float)
|
||||
for r in data.get('data', []):
|
||||
model = r.get('model', 'unknown')
|
||||
spend[model] += r.get('spend', 0) or 0
|
||||
for model, cost in sorted(spend.items()):
|
||||
print(f'{model}:{cost:.6f}')
|
||||
" 2>/dev/null)
|
||||
|
||||
# Per-key spend
|
||||
PER_KEY=$(echo "$RESP" | python3 -c "
|
||||
import sys, json
|
||||
from collections import defaultdict
|
||||
data = json.load(sys.stdin)
|
||||
spend = defaultdict(float)
|
||||
for r in data.get('data', []):
|
||||
key = r.get('api_key', 'unknown')[:20]
|
||||
spend[key] += r.get('spend', 0) or 0
|
||||
for key, cost in sorted(spend.items()):
|
||||
print(f'{key}:{cost:.6f}')
|
||||
" 2>/dev/null)
|
||||
|
||||
# Write CSV
|
||||
mkdir -p $(dirname "$CSV")
|
||||
if [ ! -f "$CSV" ]; then
|
||||
echo "date,total_cost,per_model,per_key" > "$CSV"
|
||||
fi
|
||||
echo "${DATE},${TOTAL},\"${PER_MODEL}\",\"${PER_KEY}\"" >> "$CSV"
|
||||
|
||||
echo "Snapshot: ${DATE} | Total: \$${TOTAL}"
|
||||
Reference in New Issue
Block a user