Merge pull request 'Reconcile upstream into CT116 production state' (#1) from sync/ct116-reconcile-20260912-095913 into main

This commit is contained in:
2026-09-12 10:22:57 +00:00
23 changed files with 621 additions and 1906 deletions
View File
View File
+11 -11
View File
@@ -49,7 +49,7 @@
qwen3.6-35B qwen3.6-27B gemma-4-12b
qwen3.6-35B qwen3.6-27B gpu-vision
MoE/Strix Dense/RTX3090 VLM/RTX 5070
:8080 (llama) :8080 (llama) :8080 (llama)
:8090 (side) :8090 (side) :8090 (sidecar)
@@ -84,7 +84,7 @@
|-----|------|-----------|---------|------|---------|
| qwen3.6-35B-A3B (MoE) | 192.168.68.15 | :8080 | :8090 | Strix Halo | 262K |
| qwen3.6-27B-code (Dense) | 192.168.68.8 | :8080 | :8090 | RTX 3090 | 262K |
| gemma-4-12b (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K |
| gpu-vision (VLM) | 192.168.68.110 | :8080 | :8090 | RTX 5070 | 262K |
### 1.3 Existing LiteLLM POC on CT 116
@@ -249,9 +249,9 @@ model_list:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
- model_name: gemma-4-12b
- model_name: gpu-vision
litellm_params:
model: openai/gemma-4-12b
model: openai/gpu-vision
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
@@ -298,9 +298,9 @@ router_settings:
# Fallback chains: LiteLLM retries down the chain when router returns saturated
# This gives accurate per-model metrics because router no longer silently reroutes
fallbacks:
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"]
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"]
- gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"]
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"]
- gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
# Cost tracking: map model names to per-token pricing for spend tracking
litellm_settings:
@@ -311,7 +311,7 @@ litellm_settings:
qwen3.6-27B-code:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
gemma-4-12b:
gpu-vision:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
# For internal cost allocation, set symbolic rates:
@@ -705,9 +705,9 @@ if req != "auto":
router_settings:
allowed_fails: 100
fallbacks:
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gemma-4-12b"]
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gemma-4-12b"]
- gemma-4-12b: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
- qwen3.6-35B-A3B: ["qwen3.6-27B-code", "gpu-vision"]
- qwen3.6-27B-code: ["qwen3.6-35B-A3B", "gpu-vision"]
- gpu-vision: ["qwen3.6-27B-code", "qwen3.6-35B-A3B"]
```
### Result
+2 -2
View File
@@ -8,7 +8,7 @@ CT 116 Docker stack for routing local GPU models through a unified OpenAI-compat
nginx :80 → router :9000 → GPU backends
├─ qwen3.6-35B-A3B (MoE) @ 192.168.68.15:8080 [2 slots, 262K ctx]
├─ qwen3.6-27B-code (Dense) @ 192.168.68.8:8080 [2 slots, 262K ctx]
└─ gemma-4-12b (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx]
└─ gpu-vision (VLM) @ 192.168.68.110:8080 [2 slots, 262K ctx]
Total: 6 concurrent slots
LiteLLM :8081 (fallback) | Dashboard :3000 | Redis :6379 (local)
@@ -63,7 +63,7 @@ When all GPUs are saturated, requests enter a polling queue (500ms intervals) in
|-----|-------|------|-------|
| Strix Halo | qwen3.6-35B-A3B (MoE) | 65GB | 2 | 262K | General quality |
| RTX 3090 | qwen3.6-27B-code (Dense) | 24GB | 2 | 262K | Code, reasoning |
| RTX 5070 | gemma-4-12b (VLM) | 12GB | 2 | 262K | Speed, vision |
| RTX 5070 | gpu-vision (VLM) | 12GB | 2 | 262K | Speed, vision |
## Maintenance
+1 -1
View File
@@ -70,7 +70,7 @@ body{background:var(--bg);color:var(--text);font-family:-apple-system,BlinkMacSy
</div>
<script>
const COLORS={'qwen3.6-35B-A3B':'#10b981','qwen3.6-27B-code':'#8b5cf6','gemma-4-12b':'#3b82f6'};
const COLORS={'qwen3.6-35B-A3B':'#10b981','qwen3.6-27B-code':'#8b5cf6','gpu-vision':'#3b82f6'};
const HISTORY=[]; // rolling 60 sample history for chart
function Q(id){return document.getElementById(id)}
+11 -11
View File
@@ -119,7 +119,7 @@ body { background: #0b0f17; color: #bcc3cd; font-family: -apple-system, BlinkMac
<div class="d-flex gap-2">
<select id="scatter-model" onchange="loadScatter()" style="font-size:10px;background:#1e293b;color:#94a3b8;border:1px solid #334155;border-radius:4px;padding:2px 6px">
<option value="all">All Models</option>
<option value="gemma-4-12b">12B VLM</option>
<option value="gpu-vision">12B VLM</option>
<option value="qwen3.6-27B-code">27B Dense</option>
<option value="qwen3.6-35B-A3B">35B MoE</option>
</select>
@@ -136,9 +136,9 @@ body { background: #0b0f17; color: #bcc3cd; font-family: -apple-system, BlinkMac
</div>
<script>
var MC={'gemma-4-12b':'#22c55e','qwen3.6-27B-code':'#f59e0b','qwen3.6-35B-A3B':'#a78bfa'};
var ML={'gemma-4-12b':'Gemma 4 12B','qwen3.6-27B-code':'Qwen Code','qwen3.6-35B-A3B':'Qwen MoE'};
var GL={'qwen3.6-35B-A3B':'MoE - Strix Halo','qwen3.6-27B-code':'Dense - RTX 3090','gemma-4-12b':'VLM - RTX 5070'};
var MC={'gpu-vision':'#22c55e','qwen3.6-27B-code':'#f59e0b','qwen3.6-35B-A3B':'#a78bfa'};
var ML={'gpu-vision':'Gemma 4 12B','qwen3.6-27B-code':'Qwen Code','qwen3.6-35B-A3B':'Qwen MoE'};
var GL={'qwen3.6-35B-A3B':'MoE - Strix Halo','qwen3.6-27B-code':'Dense - RTX 3090','gpu-vision':'VLM - RTX 5070'};
function $(id){return document.getElementById(id);}
function render(data){
@@ -147,7 +147,7 @@ var t=Object.values(data.route_counts||{}).reduce((a,b)=>a+b,0);
var ta=0,tm=0;data.gpus.forEach(function(g){ta+=(g.active_requests||0);tm+=(g.max_concurrent||1)});
$('kpi-total').textContent=t;$('kpi-active').textContent=ta+'/'+tm;$('kpi-agents').textContent=Object.keys(data.agent_counts||{}).length;
$('update-time').textContent=new Date().toLocaleTimeString();
var ids={'qwen3.6-35B-A3B':'gpu-moe','qwen3.6-27B-code':'gpu-dense','gemma-4-12b':'gpu-light'};
var ids={'qwen3.6-35B-A3B':'gpu-moe','qwen3.6-27B-code':'gpu-dense','gpu-vision':'gpu-light'};
data.gpus.forEach(function(g){
var el=$(ids[g.id]);if(!el)return;
var a=g.active_requests||0,mx=g.max_concurrent||1;
@@ -179,14 +179,14 @@ var sc=pct>=100?'#ef4444':pct>=50?'#f59e0b':'#22c55e';
var circ=188.5,dash=(pct/100)*circ;
var h='<div class=\"d-inline-block position-relative mb-2\"><svg width=\"72\" height=\"72\"><circle cx=\"36\" cy=\"36\" r=\"30\" fill=\"none\" stroke=\"#1e293b\" stroke-width=\"6\"/><circle cx=\"36\" cy=\"36\" r=\"30\" fill=\"none\" stroke=\"'+sc+'\" stroke-width=\"6\" stroke-dasharray=\"'+dash+' '+(circ-dash)+'\" stroke-linecap=\"round\" transform=\"rotate(-90 36 36)\"/></svg><div style=\"position:absolute;top:50%;left:50%;transform:translate(-50%,-50%);text-align:center\"><div class=\"ring-label\" style=\"color:'+sc+'\">'+ta+'</div><div class=\"ring-sublabel\">/ '+tm+' slots</div></div></div>';
h+='<div class=\"fw-bold mb-2 small\" style=\"color:'+sc+'\">'+st+'</div>';
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gemma-4-12b':'VLM'};
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gpu-vision':'VLM'};
data.gpus.forEach(function(g){var a=g.active_requests||0,mx=g.max_concurrent||1,gp=mx>0?Math.round(a/mx*100):0;h+='<div class=\"d-flex align-items-center gap-2 mb-1 justify-content-center\"><span class=\"small\" style=\"min-width:32px;text-align:right;font-size:10px\">'+(lb[g.id]||g.id)+'</span><div style=\"flex:1;max-width:70px;height:3px;background:#1e293b;border-radius:2px;overflow:hidden\"><div style=\"height:100%;width:'+gp+'%;background:'+sc+';border-radius:2px\"></div></div><span class=\"small\" style=\"min-width:22px;font-size:10px\">'+a+'/'+mx+'</span></div>'});
el.innerHTML=h;
}
function renderGPUMetrics(data){
var el=$('gpu-metrics-card');if(!el)return;
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gemma-4-12b':'VLM'};
var lb={'qwen3.6-35B-A3B':'MoE','qwen3.6-27B-code':'Dense','gpu-vision':'VLM'};
var h='';data.gpus.forEach(function(g){
var nm=lb[g.id]||g.id,tp=g.temp_c||0,ut=g.gpu_util_pct||0,pw=g.power_w||0,pl=g.power_limit_w||0;
var tc=tp>85?'#ef4444':tp>70?'#f59e0b':'#22c55e',uc=ut>90?'#ef4444':ut>70?'#f59e0b':'#22c55e';
@@ -219,8 +219,8 @@ function loadPerf(){fetch('/api/performance?window='+perfWindow).then(function(r
function renderPerf(d){
var models=d.models||[],reasons=d.reasons||[],agents=d.agents||[],sum=d.summary||{};
// Latency bars: p50/p95/p99 per model
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gemma-4-12b':'12B VLM'};
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gemma-4-12b':'#22c55e'};
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gpu-vision':'12B VLM'};
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gpu-vision':'#22c55e'};
if(!models.length){$('perf-latency').innerHTML='<div class="text-secondary small text-center py-4">Accumulating data...</div>';return;}
var maxLat=Math.max(...models.map(function(m){return m.latency.p99||0}),1);
var latHTML=models.map(function(m){
@@ -270,8 +270,8 @@ fetch('/api/scatter?window=24&model='+m).then(function(r){return r.json()}).then
function renderScatter(d){
var pts=d.points||[],el=$('scatter-plot'),lg=$('scatter-legend');
if(!pts.length){el.innerHTML='<div class="text-secondary small text-center py-5">No data yet</div>';return;}
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gemma-4-12b':'#22c55e','unknown':'#38bdf8'};
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gemma-4-12b':'12B VLM'};
var mcol={'qwen3.6-35B-A3B':'#a78bfa','qwen3.6-27B-code':'#f59e0b','gpu-vision':'#22c55e','unknown':'#38bdf8'};
var mlab={'qwen3.6-35B-A3B':'35B MoE','qwen3.6-27B-code':'27B Dense','gpu-vision':'12B VLM'};
var maxX=Math.max.apply(null,pts.map(function(p){return p.prompt_tokens||0}))||1000;
var maxY=Math.max.apply(null,pts.map(function(p){return p.inference_ms||0}))||5000;
// Log scale for X axis (prompt tokens vary widely)
+4 -4
View File
@@ -362,7 +362,7 @@ function dashboard() {
try {
const metrics = await fetch('/metrics/circuit-breaker').then(r => r.json());
this.gpuHealth = [
{ id: 'gemma3-70b', name: 'Gemma 3 70B', model: 'gemma3-70b', health_score: metrics.gemma3_70b?.gpu_health_score || 39.4, vram_pct: 45, temp: 78, load: 65, is_preferred: true },
{ id: 'gpu-vision', name: 'Qwen3.5-9B Vision', model: 'gpu-vision', health_score: metrics.gpu_vision?.gpu_health_score || 39.4, vram_pct: 45, temp: 78, load: 65, is_preferred: true },
{ id: 'deepseek-v3', name: 'DeepSeek V3', model: 'deepseek-v3', health_score: metrics.deepseek_v3?.gpu_health_score || 45.9, vram_pct: 60, temp: 82, load: 50, is_preferred: false },
{ id: 'mistral-small', name: 'Mistral Small', model: 'mistral-small', health_score: metrics.mistral_small?.gpu_health_score || 35.0, vram_pct: 30, temp: 65, load: 40, is_preferred: false }
];
@@ -388,7 +388,7 @@ function dashboard() {
async fetchSessionAnalytics() {
try {
this.sessionData = {
distribution: { 'gemma3-70b': 45, 'deepseek-v3': 30, 'mistral-small': 25 },
distribution: { 'gpu-vision': 45, 'deepseek-v3': 30, 'mistral-small': 25 },
trend: Array.from({ length: 24 }, (_, i) => ({ time: `${i}:00`, sessions: Math.floor(Math.random() * 20) + 10 })),
peaks: { '09:00': 25, '14:00': 30, '18:00': 20 }
};
@@ -400,7 +400,7 @@ function dashboard() {
try {
this.systemPerf = {
latency: { p50: Math.floor(Math.random() * 50) + 100, p95: Math.floor(Math.random() * 200) + 250, p99: Math.floor(Math.random() * 500) + 400 },
errorRates: { 'gemma3-70b': Math.random() * 0.01, 'deepseek-v3': Math.random() * 0.02, 'mistral-small': Math.random() * 0.015 }
errorRates: { 'gpu-vision': Math.random() * 0.01, 'deepseek-v3': Math.random() * 0.02, 'mistral-small': Math.random() * 0.015 }
};
console.log('System performance loaded');
} catch (error) { console.error('Failed to load system performance:', error); }
@@ -434,7 +434,7 @@ function dashboard() {
},
getGPUColor(name) {
const colors = { 'gemma3-70b': '#3b82f6', 'deepseek-v3': '#8b5cf6', 'mistral-small': '#10b981' };
const colors = { 'gpu-vision': '#3b82f6', 'deepseek-v3': '#8b5cf6', 'mistral-small': '#10b981' };
return colors[name] || '#9ca3af';
}
};
+49 -26
View File
@@ -1,4 +1,4 @@
version: '3.8'
version: "3.8"
services:
redis:
@@ -16,46 +16,70 @@ services:
timeout: 3s
retries: 5
router:
build: ./router
container_name: harness-router
postgres:
image: postgres:16-alpine
container_name: harness-postgres
restart: unless-stopped
network_mode: host
ports:
- "127.0.0.1:5432:5432"
environment:
- REDIS_URL=redis://127.0.0.1:6379
- GPU_MOE_URL=http://192.168.68.15:8080/v1
- GPU_DENSE_URL=http://192.168.68.8:8080/v1
- GPU_LIGHT_URL=http://192.168.68.110:8080/v1
- API_KEYS={"sk-sys...-key":{"tier":"enterprise","agent":"admin","deprecated":true},"sk-9e6...cb64":{"tier":"enterprise","agent":"admin"},"***":{"tier":"enterprise","agent":"Abiba","deprecated":true},"sk-856...a889":{"tier":"enterprise","agent":"Abiba"},"***":{"tier":"enterprise","agent":"Mumuni","deprecated":true},"sk-b57...807e":{"tier":"enterprise","agent":"Mumuni"},"***":{"tier":"enterprise","agent":"Tanko","deprecated":true},"sk-620...eaa7":{"tier":"enterprise","agent":"Tanko"},"***":{"tier":"enterprise","agent":"Koby","deprecated":true},"sk-eb3...fdee":{"tier":"enterprise","agent":"Koby"},"***":{"tier":"enterprise","agent":"Kagenz0","deprecated":true},"sk-12b...ed9b":{"tier":"enterprise","agent":"Kagenz0"},"***":{"tier":"enterprise","agent":"Koonimo","deprecated":true},"sk-680...4dfe":{"tier":"enterprise","agent":"Koonimo"},"***":{"tier":"starter","agent":"test-starter","deprecated":true},"sk-55d...7860":{"tier":"starter","agent":"test-starter"},"sk-pro...z789":{"tier":"professional","agent":"test-pro","deprecated":true},"sk-b51...e676":{"tier":"professional","agent":"test-pro"}}
- ADMIN_KEY=sk-adm...8814
- POSTGRES_DB=litellm
- POSTGRES_USER=litellm
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD}
volumes:
- pgdata:/var/lib/postgresql/data
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:9000/health')"]
interval: 30s
timeout: 15s
retries: 3
depends_on:
redis:
condition: service_healthy
test: ["CMD-SHELL", "pg_isready -U litellm"]
interval: 5s
timeout: 3s
retries: 5
litellm:
image: ghcr.io/berriai/litellm:main-stable
image: ghcr.io/docker.litellm.ai/berriai/litellm:1.99.1
command: ["--config", "/app/config.yaml", "--port", "4000"]
container_name: harness-litellm
restart: unless-stopped
ports:
- "127.0.0.1:8081:4000"
- "4000:4000"
volumes:
- ./litellm_config.yaml:/app/config.yaml
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
environment:
- LITELLM_MASTER_KEY=sk-sys...-key
- PROMETHEUS_EXPORTER=true
- ENFORCE_PRISMA_MIGRATION_CHECK=true
- LITELLM_MASTER_KEY=${LITELLM_MASTER_KEY}
- DATABASE_URL=postgresql://litellm:${POSTGRES_PASSWORD}@postgres:5432/litellm
- STORE_MODEL_IN_DB=True
- LITELLM_UI_USERNAME=admin
- LITELLM_UI_PASSWORD=${LITELLM_UI_PASSWORD}
- UI_USERNAME=admin
- UI_PASSWORD=${UI_PASSWORD}
- OPENAI_API_KEY=${OPENAI_API_KEY}
- PROXY_BASE_URL=https://litellm.sysloggh.net
- DOCS_URL=/docs
- ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY}
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
- GENERIC_CLIENT_SECRET=${GENERIC_CLIENT_SECRET}
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
- GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/
- GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/
- GENERIC_SCOPE=openid email profile
- MCP_SERVER_RAHOS_URL=http://192.168.68.65:3100/mcp
- MCP_SERVER_RAHOS_TRANSPORT=http
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
extra_hosts:
- "host.docker.internal:host-gateway"
- "auth.sysloggh.net:192.168.68.11"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
interval: 15s
timeout: 5s
retries: 3
start_period: 120s
depends_on:
postgres:
condition: service_healthy
redis:
condition: service_healthy
@@ -69,7 +93,7 @@ services:
- ./nginx/nginx.conf:/etc/nginx/nginx.conf:ro
- ./dashboard:/opt/inference-harness/dashboard:ro
extra_hosts:
- "host.docker.internal:host-gateway"
- "auth.sysloggh.net:192.168.68.11"
healthcheck:
test: ["CMD", "curl", "-f", "http://127.0.0.1/health"]
interval: 30s
@@ -86,7 +110,7 @@ services:
ports:
- "127.0.0.1:3000:3000"
environment:
- REDIS_URL=redis://127.0.0.1:6379
- REDIS_URL=redis://redis:6379
- GPU_SIDECARS=192.168.68.15:8090,192.168.68.8:8090,192.168.68.110:8090
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:3000/health')"]
@@ -96,8 +120,7 @@ services:
depends_on:
- redis
volumes:
redis-data:
# LiteLLM command override to load config
# (appended to fix config loading issue)
pgdata:
+62 -14
View File
@@ -1,4 +1,4 @@
version: '3.8'
version: "3.8"
services:
redis:
@@ -16,43 +16,89 @@ services:
timeout: 3s
retries: 5
postgres:
image: postgres:16-alpine
container_name: harness-postgres
restart: unless-stopped
ports:
- "127.0.0.1:5432:5432"
environment:
- POSTGRES_DB=litellm
- POSTGRES_USER=litellm
- POSTGRES_PASSWORD=d9fc143e3dc1a7a8e672c359fea95c5e
volumes:
- pgdata:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm"]
interval: 5s
timeout: 3s
retries: 5
router:
build: ./router
container_name: harness-router
restart: unless-stopped
ports:
- "9000:9000"
- "127.0.0.1:9000:9000"
environment:
- REDIS_URL=redis://redis:6379
- GPU_MOE_URL=http://192.168.68.15:8080/v1
- GPU_DENSE_URL=http://192.168.68.8:8080/v1
- GPU_MOE_URL=http://192.168.68.15:8080/v1
- GPU_LIGHT_URL=http://192.168.68.110:8080/v1
- API_KEYS={"sk-syslog-local-master-key":{"tier":"enterprise","agent":"admin","deprecated":true},"sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64":{"tier":"enterprise","agent":"admin"},"sk-syslog-abiba":{"tier":"enterprise","agent":"Abiba","deprecated":true},"sk-856ffb0bbb-e5aaf78b10054eca608f8fbcbd73a889":{"tier":"enterprise","agent":"Abiba"},"sk-syslog-mumuni":{"tier":"enterprise","agent":"Mumuni","deprecated":true},"sk-b57e6e042e-47573660114f3138c852c47f62da807e":{"tier":"enterprise","agent":"Mumuni"},"sk-syslog-tanko":{"tier":"enterprise","agent":"Tanko","deprecated":true},"sk-620a05e95a-e93d875476b650a4d1137249ead8eaa7":{"tier":"enterprise","agent":"Tanko"},"sk-syslog-koby":{"tier":"enterprise","agent":"Koby","deprecated":true},"sk-eb3e6fc1c0-de1bf2edf35a53cb3749a2400483fdee":{"tier":"enterprise","agent":"Koby"},"sk-syslog-kagenz0":{"tier":"enterprise","agent":"Kagenz0","deprecated":true},"sk-12b66b3392-b548aed9138aeb6f698e8e521650ed9b":{"tier":"enterprise","agent":"Kagenz0"},"sk-syslog-koonimo":{"tier":"enterprise","agent":"Koonimo","deprecated":true},"sk-680d06686c-00ee8bf9dc3c93b276af122d49a14dfe":{"tier":"enterprise","agent":"Koonimo"},"sk-starter-abc123":{"tier":"starter","agent":"test-starter","deprecated":true},"sk-55da55907a-1bd7ff344e26feda50e9ac2219697860":{"tier":"starter","agent":"test-starter"},"sk-professional-xyz789":{"tier":"professional","agent":"test-pro","deprecated":true},"sk-b5159863e6-8df3ae52fb958cfe76cc2888c8c8e676":{"tier":"professional","agent":"test-pro"}}
- ADMIN_KEY=sk-admin-ee09fffd04978b61a1569ac670c68814
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:9000/health')"]
interval: 15s
timeout: 5s
interval: 30s
timeout: 15s
retries: 3
depends_on:
redis:
condition: service_healthy
litellm:
image: ghcr.io/berriai/litellm:main-stable
image: ghcr.io/docker.litellm.ai/berriai/litellm:1.90.0-rc.1
command: ["--config", "/app/config.yaml", "--port", "4000"]
container_name: harness-litellm
restart: unless-stopped
ports:
- "8081:4000"
- "4000:4000"
volumes:
- ./litellm_config.yaml:/app/config.yaml
- /opt/combined-ca-bundle.pem:/etc/ssl/certs/ca-certificates.crt:ro
environment:
- LITELLM_MASTER_KEY=sk-syslog-local-master-key
- PROMETHEUS_EXPORTER=true
- LITELLM_MASTER_KEY=sk-litellm-7f96080dd99b15c36bd4b333b58a6796
- ROUTER_API_KEY=sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
- DATABASE_URL=postgresql://litellm:d9fc143e3dc1a7a8e672c359fea95c5e@postgres:5432/litellm
- STORE_MODEL_IN_DB=True
- LITELLM_UI_USERNAME=admin
- LITELLM_UI_PASSWORD=syslog-admin-2026
- UI_USERNAME=admin
- UI_PASSWORD=syslog-admin-2026
- OPENAI_API_KEY=not-used
- PROXY_BASE_URL=https://litellm.sysloggh.net
- DOCS_URL=/docs
- ANTHROPIC_API_KEY=not-used
- GENERIC_CLIENT_ID=FHd7bs9dP5gHad2Ki23iUL5kQvFa0GRaj3nlLnNU
- GENERIC_CLIENT_SECRET=aDkXQx82duqpxc98xxqp0quzzUf1mawsnOTqj7sx1acaS7rWSt02N5ksBCi92n8ZilRavigoYME6fLakP20Ixc9H2pxnSZFiOqQLb7BPi8UtsvfxmzXklD0HJIdbKFxe
- GENERIC_AUTHORIZATION_ENDPOINT=https://auth.sysloggh.net/application/o/authorize/
- GENERIC_TOKEN_ENDPOINT=http://harness-nginx/application/o/token/
- GENERIC_USERINFO_ENDPOINT=http://harness-nginx/application/o/userinfo/
- GENERIC_SCOPE=openid email profile
- GENERIC_ROLE_MAPPINGS_DEFAULT_ROLE=proxy_admin
- SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
extra_hosts:
- "host.docker.internal:host-gateway"
- "auth.sysloggh.net:192.168.68.11"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
interval: 15s
timeout: 5s
retries: 3
depends_on:
postgres:
condition: service_healthy
redis:
condition: service_healthy
@@ -64,10 +110,13 @@ services:
- "80:80"
volumes:
- ./nginx/nginx.conf:/etc/nginx/nginx.conf:ro
- ./dashboard:/opt/inference-harness/dashboard:ro
extra_hosts:
- "auth.sysloggh.net:192.168.68.11"
healthcheck:
test: ["CMD", "curl", "-f", "http://127.0.0.1/health"]
interval: 15s
timeout: 5s
interval: 30s
timeout: 15s
retries: 3
depends_on:
- litellm
@@ -78,7 +127,7 @@ services:
container_name: harness-dashboard
restart: unless-stopped
ports:
- "3000:3000"
- "127.0.0.1:3000:3000"
environment:
- REDIS_URL=redis://redis:6379
- GPU_SIDECARS=192.168.68.15:8090,192.168.68.8:8090,192.168.68.110:8090
@@ -90,8 +139,7 @@ services:
depends_on:
- redis
volumes:
redis-data:
# LiteLLM command override to load config
# (appended to fix config loading issue)
pgdata:
-106
View File
@@ -1,106 +0,0 @@
## Syslog GPU Router — Nginx Configuration (Docker-internal)
## Routes incoming agent requests to the appropriate GPU backend
## based on the X-Syslog-Model header.
upstream amdpve_pool {
## Strix Halo 395 — qwen3.6-35B-A3B (MoE) — Default workhorse
server 192.168.68.15:8080;
}
upstream llmgpu_pool {
## RTX 3090 — qwen3.5-27B (Dense) — Heavy reasoning
server 192.168.68.8:8080;
}
upstream ocu_llm_pool {
## New Backend — gemma-4-12b (VLM) — Vision + light tasks
server 192.168.68.110:8080;
}
upstream queue_service {
## Agent queue with circuit breaker (Docker container)
server queue-service:8091;
}
upstream dashboard_service {
## Harness dashboard (Docker container)
server dashboard:3001;
}
## ------------------------------------------------------------------
## Mapping: X-Syslog-Model header → upstream backend
## ------------------------------------------------------------------
map $http_x_syslog_model $gpu_upstream {
default amdpve_pool;
"standard" amdpve_pool;
"heavy" llmgpu_pool;
"qwen3.5-27B" llmgpu_pool;
"light" ocu_llm_pool;
"gemma-4-12b" ocu_llm_pool;
}
## Rate limit zone — 10 req/s per IP, burst of 20
limit_req_zone $binary_remote_addr zone=perip:10m rate=10r/s;
server {
listen 80;
server_name _;
## ------------------------------------------------------------------
## Dashboard — observability UI (MUST be before / catch-all)
## ------------------------------------------------------------------
location /dashboard {
proxy_pass http://dashboard_service/;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
}
## ------------------------------------------------------------------
## Main location — proxy to selected upstream
## ------------------------------------------------------------------
location / {
limit_req zone=perip burst=20 nodelay;
limit_req_status 503;
proxy_pass http://$gpu_upstream;
## Preserve original host and headers
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
## Pass through the model header so backends can log it
proxy_pass_header X-Syslog-Model;
## Streaming support (SSE for LLM responses)
proxy_buffering off;
proxy_cache off;
proxy_read_timeout 300s;
proxy_send_timeout 300s;
## Basic failover — retry on error or timeout
proxy_next_upstream error timeout http_502 http_503;
proxy_next_upstream_tries 2;
## Add a response header for observability
add_header X-Routed-To $gpu_upstream always;
## Fallback to queue when all GPU upstreams are down
error_page 502 503 504 = @queue_fallback;
}
## ------------------------------------------------------------------
## Queue fallback — enqueue when GPUs are unavailable
## ------------------------------------------------------------------
location @queue_fallback {
rewrite ^ /enqueue break;
proxy_pass http://queue_service;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header Content-Type $content_type;
proxy_pass_request_body on;
}
}
-106
View File
@@ -1,106 +0,0 @@
## Syslog GPU Router — Nginx Configuration
## Routes incoming agent requests to the appropriate GPU backend
## based on the X-Syslog-Model header.
upstream amdpve_pool {
## Strix Halo 395 — qwen3.6-35B-A3B (MoE) — Default workhorse
server 192.168.68.15:8080;
}
upstream llmgpu_pool {
## RTX 3090 — qwen3.5-27B (Dense) — Heavy reasoning
server 192.168.68.8:8080;
}
upstream ocu_llm_pool {
## New Backend — gemma-4-12b (VLM) — Vision + light tasks
server 192.168.68.110:8080;
}
upstream queue_service {
## Agent queue with circuit breaker (Docker container)
server 127.0.0.1:8091;
}
upstream dashboard_service {
## Harness dashboard (Docker container)
server 127.0.0.1:3001;
}
## ------------------------------------------------------------------
## Mapping: X-Syslog-Model header → upstream backend
## ------------------------------------------------------------------
map $http_x_syslog_model $gpu_upstream {
default amdpve_pool; # missing header → default workhorse
"standard" amdpve_pool;
"heavy" llmgpu_pool;
"qwen3.5-27B" llmgpu_pool;
"light" ocu_llm_pool;
"gemma-4-12b" ocu_llm_pool;
}
server {
listen 8080;
server_name _;
# Rate limit zone — 10 req/s per IP, burst of 20
limit_req_zone $binary_remote_addr zone=perip:10m rate=10r/s;
## ------------------------------------------------------------------
## Dashboard — observability UI (MUST be before / catch-all)
## ------------------------------------------------------------------
location /dashboard {
proxy_pass http://dashboard_service/;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
}
## ------------------------------------------------------------------
## Main location — proxy to selected upstream
## ------------------------------------------------------------------
location / {
limit_req zone=perip burst=20 nodelay;
limit_req_status 503;
proxy_pass http://$gpu_upstream;
## Preserve original host and headers
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
## Pass through the model header so backends can log it
proxy_pass_header X-Syslog-Model;
## Streaming support (SSE for LLM responses)
proxy_buffering off;
proxy_cache off;
proxy_read_timeout 300s;
proxy_send_timeout 300s;
## Basic failover — retry on error or timeout
proxy_next_upstream error timeout http_502 http_503;
proxy_next_upstream_tries 2;
## Add a response header for observability
add_header X-Routed-To $gpu_upstream always;
## Fallback to queue when all GPU upstreams are down
error_page 502 503 504 = @queue_fallback;
}
## ------------------------------------------------------------------
## Queue fallback — enqueue when GPUs are unavailable
## ------------------------------------------------------------------
location @queue_fallback {
rewrite ^ /enqueue break;
proxy_pass http://queue_service;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header Content-Type $content_type;
proxy_pass_request_body on;
}
}
+54
View File
@@ -0,0 +1,54 @@
models:
gpu-vision:
gpu_url: http://192.168.68.110:8080/v1
gpu_host: 192.168.68.110
label: Qwen3.5-9B Vision (RTX 5070)
max_concurrent: 1
context: 131072
tiers: [starter, professional, enterprise]
capabilities: [completion, multimodal]
model_path: /home/llmuser/models/qwen3.5-9b/Qwen3.5-9B-Q5_K_M.gguf
args: --mmproj /home/llmuser/models/qwen3.5-9b/mmproj-F16.gguf --ctx-size 131072 --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn 1 --parallel 1 --alias gpu-light --reasoning off --api-key not-needed --cache-prompt
qwen3.6-27B-code:
gpu_url: http://192.168.68.8:8080/v1
sidecar_url: http://192.168.68.8:8090
gpu_host: 192.168.68.8
label: Qwen3.6 27B Code (RTX 3090)
max_concurrent: 2
context: 262144
tiers: [professional, enterprise]
capabilities: [completion]
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
args: --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
qwen3.6-35B-udq4:
gpu_url: http://192.168.68.15:8080/v1
sidecar_url: http://192.168.68.15:8090
gpu_host: 192.168.68.15
label: Qwen3.6 35B UD-Q4_K_M (Strix Halo)
max_concurrent: 1
context: 262144
tiers: [professional, enterprise]
capabilities: [completion, multimodal]
model_path: /models/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf
args: -c 262144 -ngl 99 --flash-attn on
hosts:
gpu-light:
address: 192.168.68.110
gpu_name: NVIDIA GeForce RTX 5070
vram_gb: 12
current_model: gpu-vision
gpu-dense:
address: 192.168.68.8
gpu_name: NVIDIA GeForce RTX 3090
vram_gb: 24
current_model: qwen3.6-27B-code
gpu-moe:
address: 192.168.68.15
gpu_name: AMD Strix Halo (iGPU)
vram_gb: 64
current_model: qwen3.6-35B-udq4
+173 -16
View File
@@ -1,21 +1,178 @@
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
store_model_in_db: false
user_url_allowed_hosts:
- 192.168.68.14
- 192.168.68.14:5080
guardrails:
- guardrail_name: input-moderation
litellm_params:
guardrail: openai_moderation
mode: pre_call
- guardrail_name: output-moderation
litellm_params:
guardrail: openai_moderation
mode: post_call
- guardrail_name: harmful-content-filter
litellm_params:
categories:
- action: BLOCK
category: harmful_self_harm
enabled: true
severity_threshold: medium
- action: BLOCK
category: harmful_violence
enabled: true
severity_threshold: medium
- action: BLOCK
category: harmful_illegal_weapons
enabled: true
severity_threshold: medium
guardrail: litellm_content_filter
mode: pre_call
litellm_settings:
user_url_allowed_hosts:
- 192.168.68.14
- 192.168.68.14:5080
cache: true
cache_params:
host: harness-redis
namespace: litellm
port: 6379
ttl: 600
type: redis
drop_params: true
success_callback:
- prometheus
failure_callback:
- prometheus
model_cost:
qwen3.6-27B-code:
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
strix-moe:
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
syslog-auto:
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
num_retries: 2
request_timeout: 600
sso_callback: /sso/callback
model_list:
- model_name: qwen3.6-35B-A3B
litellm_params:
model: openai/qwen3.6-35B-A3B
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
- model_name: gpu-dense
litellm_params:
model: openai/qwen3.6-27B-code-text
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
- model_name: gpu-light
litellm_params:
model: openai/gemma-4-12b
model: openai/qwen3.6-27B-code
timeout: 300
model_info:
max_input_tokens: 131072
model_name: qwen3.6-27B-code
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
model: openai/strix-moe
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: qwen3.6-35B-udq4
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
model: openai/strix-moe
rpm: 40
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: strix-moe
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: gpu-dense
- litellm_params:
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
general_settings:
master_key: sk-syslog-local-master-key
litellm_settings:
drop_params: true
request_timeout: 120
model: openai/gpu-vision
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: gpu-vision
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
weight: 0.70
model_name: syslog-auto
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: qwen3.8-27B-uncensored
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
model: openai/strix-moe
rpm: 60
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.20
model_name: syslog-auto
- litellm_params:
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
model: openai/gpu-vision
rpm: 200
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.10
model_name: syslog-auto
router_settings:
allowed_fails: 100
enable_loadbalancing_on_proxy: true
fallbacks:
- syslog-auto:
- qwen3.6-27B-code
- strix-moe
- gpu-vision
- qwen3.6-27B-code:
- strix-moe
- strix-moe:
- qwen3.6-27B-code
- gpu-vision
request_timeout: 300
routing_strategy: simple-shuffle
agents:
- agent_name: agent-zero-homelab
agent_card_params:
name: Agent Zero HomeLab
url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab
protocolVersion: '1.0'
+123 -68
View File
@@ -1,118 +1,173 @@
worker_processes auto;
error_log /var/log/nginx/error.log warn;
pid /var/run/nginx.pid;
events { worker_connections 1024; }
events {
worker_connections 1024;
}
http {
include /etc/nginx/mime.types;
default_type application/octet-stream;
log_format main '$remote_addr - $remote_user [$time_local] "$request" '
'$status $body_bytes_sent "$http_referer" '
'"$http_user_agent" rt=$request_time';
access_log /var/log/nginx/access.log main;
error_log /var/log/nginx/error.log;
sendfile on;
keepalive_timeout 65;
upstream router_api { server host.docker.internal:9000; }
upstream dashboard_ui { server dashboard:3000; }
upstream litellm_backend { server litellm:4000; }
resolver 127.0.0.11 valid=30s;
map $host $dashboard_ui_url {
default http://harness-dashboard:3000;
}
map $host $litellm_backend_url {
default http://harness-litellm:4000;
}
# ════════════════════════════════════════════════════════════
# Server :80 — Lean single-layer entrypoint
# All API paths route directly to LiteLLM.
# harness-router fully deprecated.
# ════════════════════════════════════════════════════════════
server {
listen 80;
# Security headers
add_header X-Content-Type-Options nosniff always;
add_header X-Frame-Options SAMEORIGIN always;
add_header X-XSS-Protection "1; mode=block" always;
# Authentik OIDC subrequest
location /authentik/auth {
internal;
proxy_pass https://auth.sysloggh.net/outpost.goauthentik.io/auth/nginx;
proxy_pass_request_body off;
proxy_set_header Content-Length "";
proxy_set_header X-Original-URL $scheme://$http_host$request_uri;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
}
# Disable buffering for SSE streams
proxy_buffering off;
# API through router
# ── LiteLLM API (replaces router /v1/) ──
location /v1/ {
proxy_pass http://router_api;
proxy_pass $litellm_backend_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header Authorization $http_authorization;
proxy_connect_timeout 10s;
proxy_read_timeout 600s;
proxy_buffering off;
}
# ── LiteLLM admin (replaces router /admin/) ──
location /admin/ {
proxy_pass http://router_api;
proxy_pass $litellm_backend_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header Authorization $http_authorization;
proxy_read_timeout 600s;
}
# SSE streaming endpoint
# ── LiteLLM stream ──
location /stream {
proxy_pass http://router_api;
proxy_pass $litellm_backend_url/stream;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header Connection "";
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_buffering off;
chunked_transfer_encoding off;
}
# Dashboard API proxy for SSE
# ── API passthrough ──
location /api/ {
proxy_pass http://dashboard_ui;
proxy_pass $litellm_backend_url/;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_buffering off;
}
# LiteLLM debug
location /litellm/ {
rewrite ^/litellm/(.*) /$1 break;
proxy_pass http://litellm_backend;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header Authorization $http_authorization;
}
# Professional Dashboard (Phase 1-3) - Static HTML served via Nginx
# ── Dashboard ──
location /dashboard/ {
alias /opt/inference-harness/dashboard/;
index dashboard.html;
add_header Cache-Control "public, max-age=3600";
add_header X-Content-Type-Options nosniff;
}
# Legacy Dashboard (root) - Proxy to Flask app
location / {
proxy_pass http://dashboard_ui;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_buffering off;
}
# Performance analytics
location /metrics/ {
proxy_pass http://router_api;
proxy_pass $dashboard_ui_url/;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
# Circuit Breaker metrics (Phase 1)
location /metrics/circuit-breaker {
proxy_pass http://router_api/metrics/circuit-breaker;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
location /health {
proxy_pass http://router_api/health;
# ── GPU Fleet Dashboard ──
location /gpu/ {
proxy_pass http://192.168.68.24:9100/;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_read_timeout 60s;
}
# ── LiteLLM redirect ──
location = /litellm {
return 301 /litellm/;
}
# ── Health: redirect /health (auth-required) → /health/liveliness (no-auth) ──
location = /litellm/health {
return 301 /litellm/health/liveliness;
}
# ── LiteLLM static assets ──
location /litellm-asset-prefix/ {
proxy_pass $litellm_backend_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_connect_timeout 10s;
proxy_read_timeout 60s;
}
# ── LiteLLM admin UI and API (strips /litellm prefix) ──
location /litellm/ {
rewrite ^/litellm(/.*)$ $1 break;
proxy_pass $litellm_backend_url;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_buffering off;
proxy_read_timeout 600s;
proxy_set_header Authorization $http_authorization;
}
# ── Auth proxy to Authentik ──
location /application/o/ {
proxy_pass https://192.168.68.11;
proxy_ssl_verify off;
proxy_set_header Host auth.sysloggh.net;
proxy_set_header X-Real-IP $remote_addr;
proxy_connect_timeout 10s;
proxy_read_timeout 30s;
}
# ── Prometheus metrics (LiteLLM exposes at /metrics) ──
location /metrics {
proxy_pass $litellm_backend_url/metrics;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
# ---- Legacy /health/unified alias (harness-router decommissioned) ----
# Retained intentionally: 5+ live GPU monitor contracts poll
# this path every 15s and follow the 301 to /gpu/gpu-data.
location /health/unified {
return 301 /gpu/gpu-data;
}
# ── Health: no-auth LiteLLM liveliness ──
location /health {
proxy_pass $litellm_backend_url/health/liveliness;
proxy_http_version 1.1;
proxy_set_header Host $host;
}
# ── UI / Docs convenience redirects (additive 2026-09-11) ──
# Canonical app path is /litellm/. These 301s make bare
# /ui/ and /docs resolve. No new auth surface; no OIDC impact.
location = /ui { return 301 /litellm/ui/; }
location /ui/ { return 301 /litellm$request_uri; }
location = /docs { return 301 /litellm/docs; }
location = /docs/ { return 301 /litellm/docs; }
# ── 404 for everything else ──
location / {
return 404;
}
}
}
-121
View File
@@ -1,121 +0,0 @@
#!/usr/bin/env python3
"""Syslog Inference Queue Service — Circuit breaker + request queuing.
Ports: 8091
Endpoints:
/health — liveness probe (Nginx upstream check)
/enqueue — POST inference request into queue (fallback from Nginx)
/status — GET queue depth + circuit breaker state
"""
import json
import os
import sys
import time
import urllib.request
from flask import Flask, request, jsonify
app = Flask(__name__)
# Configuration
REDIS_HOST = os.getenv("REDIS_HOST", "192.168.68.7")
REDIS_PORT = int(os.getenv("REDIS_PORT", "6379"))
QUEUE_KEY = "inference:requests"
CIRCUIT_OPEN_THRESHOLD = 50
CIRCUIT_WARN_THRESHOLD = 30
# GPU endpoints for draining
GPUS = {
"amdpve": "192.168.68.15:8080",
"llmgpu": "192.168.68.8:8080",
"ocu_llm": "192.168.68.110:8080",
}
def get_redis():
try:
import redis
return redis.Redis(host=REDIS_HOST, port=REDIS_PORT, decode_responses=True)
except Exception:
return None
def get_queue_depth(r):
try:
return r.llen(QUEUE_KEY)
except Exception:
return 0
def check_gpu_health(endpoint):
try:
req = urllib.request.Request(f"http://{endpoint}/v1/models")
req.add_header("User-Agent", "queue-service/1.0")
resp = urllib.request.urlopen(req, timeout=3)
return resp.status == 200
except Exception:
return False
@app.route("/health")
def health():
"""Nginx upstream health probe. Returns 200 if service is alive."""
return jsonify({"status": "ok", "service": "queue-service"}), 200
@app.route("/enqueue", methods=["POST"])
def enqueue():
"""Fallback endpoint — Nginx calls this when all GPU upstreams are down."""
r = get_redis()
if not r:
return jsonify({"error": "Redis unavailable"}), 503
depth = get_queue_depth(r)
if depth >= CIRCUIT_OPEN_THRESHOLD:
return jsonify({
"error": "Circuit breaker OPEN",
"queue_depth": depth,
"threshold": CIRCUIT_OPEN_THRESHOLD
}), 503
# Store the request in queue
payload = request.get_data(as_text=True)
headers = {k: v for k, v in request.headers if k.startswith("X-")}
r.rpush(QUEUE_KEY, json.dumps({
"payload": payload,
"headers": headers,
"queued_at": time.time()
}))
new_depth = get_queue_depth(r)
return jsonify({
"status": "queued",
"position": new_depth,
"circuit": "warn" if new_depth >= CIRCUIT_WARN_THRESHOLD else "closed"
}), 202
@app.route("/status")
def status():
"""GET queue depth + circuit breaker state + GPU health."""
r = get_redis()
depth = get_queue_depth(r) if r else -1
circuit = "open" if depth >= CIRCUIT_OPEN_THRESHOLD else ("warn" if depth >= CIRCUIT_WARN_THRESHOLD else "closed")
gpu_health = {}
for name, endpoint in GPUS.items():
gpu_health[name] = "up" if check_gpu_health(endpoint) else "down"
return jsonify({
"queue_depth": depth,
"circuit_breaker": circuit,
"gpu_health": gpu_health,
"thresholds": {
"warn": CIRCUIT_WARN_THRESHOLD,
"open": CIRCUIT_OPEN_THRESHOLD
}
})
if __name__ == "__main__":
app.run(host="0.0.0.0", port=8091)
+75
View File
@@ -0,0 +1,75 @@
# GPU Roster Loader - reads gpu_roster.yaml via PyYAML
# Hot-reloadable via /admin/roster/reload endpoint
import os, json, threading, time
try:
import yaml
except ImportError:
yaml = None
ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml')
_last_mtime = 0
_lock = threading.Lock()
GPU_URLS = {}
GPU_SIDECARS = {}
GPU_LABELS = {}
GPU_MAX_CONCURRENT = {}
GPU_CONTEXT = {}
TIER_MODELS = {}
HOSTS = {}
def load_roster(path=None):
global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT
global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime
if yaml is None:
return False, 'PyYAML not installed - run: pip3 install pyyaml'
path = path or ROSTER_PATH
try:
with open(path) as f:
data = yaml.safe_load(f)
with _lock:
GPU_URLS.clear()
GPU_SIDECARS.clear()
GPU_LABELS.clear()
GPU_MAX_CONCURRENT.clear()
GPU_CONTEXT.clear()
TIER_MODELS.clear()
models = data.get('models', {})
for name, cfg in models.items():
GPU_URLS[name] = cfg.get('gpu_url', '')
GPU_SIDECARS[name] = cfg.get('sidecar_url', '')
GPU_LABELS[name] = cfg.get('label', name)
GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1)
GPU_CONTEXT[name] = cfg.get('context', 65536)
tiers = cfg.get('tiers', ['enterprise'])
for t in tiers:
if t not in TIER_MODELS:
TIER_MODELS[t] = []
if name not in TIER_MODELS[t]:
TIER_MODELS[t].append(name)
HOSTS.clear()
HOSTS.update(data.get('hosts', {}))
_last_mtime = os.path.getmtime(path)
return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS))
except Exception as e:
return False, str(e)
def check_reload():
global _last_mtime
try:
mtime = os.path.getmtime(ROSTER_PATH)
if mtime > _last_mtime:
success, msg = load_roster()
if success:
print('[ROSTER] Auto-reloaded:', msg)
except:
pass
def reload_thread(interval=30):
while True:
time.sleep(interval)
check_reload()
threading.Thread(target=reload_thread, daemon=True).start()
-9
View File
@@ -1,9 +0,0 @@
FROM python:3.12-slim
WORKDIR /app
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY router.py .
EXPOSE 9000
CMD ["python", "router.py"]
-90
View File
@@ -1,90 +0,0 @@
# Insert streaming support before the gpu_resp call
import re
with open('/opt/inference-harness/router/router.py') as f:
code = f.read()
# Find the gpu_resp block and replace with streaming-aware version
old = ''' start = time.time()
gpu_resp = requests.post(
gpu_url + "/chat/completions",
json=req_data,
headers={"Content-Type": "application/json", "Authorization": "Bearer not-needed"},
timeout=120,
)
latency_ms = int((time.time() - start) * 1000)
if gpu_resp.status_code != 200:
log.error("GPU error: %s %s", gpu_resp.status_code, gpu_resp.text[:200])
return jsonify({"error": "GPU backend returned " + str(gpu_resp.status_code)}), 502
response_data = gpu_resp.json()
response_data = fix_reasoning_content(response_data)
response_data["routing"] = {
"model": model, "reason": reason, "gpu": gpu_url,
"tier": tier, "agent": agent, "latency_ms": latency_ms,
}
return jsonify(response_data)'''
new = ''' start = time.time()
is_stream = req_data.get("stream", False)
gpu_resp = requests.post(
gpu_url + "/chat/completions",
json=req_data,
headers={"Content-Type": "application/json", "Authorization": "Bearer not-needed"},
timeout=120,
stream=is_stream,
)
latency_ms = int((time.time() - start) * 1000)
if gpu_resp.status_code != 200:
log.error("GPU error: %s %s", gpu_resp.status_code, gpu_resp.text[:200])
return jsonify({"error": "GPU backend returned " + str(gpu_resp.status_code)}), 502
if is_stream:
# Stream response back to client
def generate():
first = True
for line in gpu_resp.iter_lines(decode_unicode=True):
if line:
if first and line.startswith("data: "):
# Inject routing into first chunk
try:
chunk = json.loads(line[6:])
chunk["routing"] = {
"model": model, "reason": reason, "gpu": gpu_url,
"tier": tier, "agent": agent, "latency_ms": latency_ms,
}
yield "data: " + json.dumps(chunk) + "\n\n"
first = False
continue
except Exception:
pass
yield line + "\n"
yield "data: [DONE]\n\n"
return Response(stream_with_context(generate()), mimetype="text/event-stream")
response_data = gpu_resp.json()
response_data = fix_reasoning_content(response_data)
response_data["routing"] = {
"model": model, "reason": reason, "gpu": gpu_url,
"tier": tier, "agent": agent, "latency_ms": latency_ms,
}
return jsonify(response_data)'''
code = code.replace(old, new)
# Add missing import
if 'from flask import Flask, request, jsonify' in code:
code = code.replace(
'from flask import Flask, request, jsonify',
'from flask import Flask, request, jsonify, Response, stream_with_context'
)
with open('/opt/inference-harness/router/router.py', 'w') as f:
f.write(code)
print('Streaming support added')
-3
View File
@@ -1,3 +0,0 @@
flask==3.1.*
redis==5.2.*
requests==2.32.*
-77
View File
@@ -1,77 +0,0 @@
def route(rd, tier, agent=""):
msgs = rd.get("messages",[]); t = estimate_tokens(msgs)
sys = any(m.get("role")=="system" for m in msgs)
turns = len([m for m in msgs if m.get("role") in ("user","assistant")])
hints = rd.get("routing_hints",{})
allowed = TIER_MODELS.get(tier, ["gemma-4-12b"])
avail = [m for m in available_models() if m in allowed]
if not avail: return {"model": allowed[0], "reason": "all_saturated", "saturated": True}
if all(is_gpu_busy(m) for m in avail):
return {"model": avail[0], "reason": "all_saturated", "saturated": True}
# GUARD: multimodal -> VLM only (sole vision model)
has_image = any(
isinstance(m.get("content"), list) and
any(p.get("type") == "image_url" for p in m["content"] if isinstance(p, dict))
for m in msgs
)
if has_image:
if "gemma-4-12b" in avail and not is_gpu_busy("gemma-4-12b"):
return {"model": "gemma-4-12b", "reason": "vision"}
elif "gemma-4-12b" in avail:
return {"model": "gemma-4-12b", "reason": "vision_saturated", "saturated": True}
else:
return {"model": allowed[0], "reason": "vision_unavailable"}
req = rd.get("model","auto")
if req != "auto":
target = req if req in avail else avail[0]
if is_gpu_busy(target) and req in allowed:
alts = [m for m in avail if m != target and m in allowed]
if alts:
alt = select_best_gpu(alts, "explicit", agent)
if alt: return alt
return {"model": target, "reason": "explicit"}
if hints:
if hints.get("priority")=="speed" and "gemma-4-12b" in avail:
return select_best_gpu(["gemma-4-12b"], "hint_speed", agent) or {"model":"gemma-4-12b","reason":"hint_speed"}
if hints.get("priority")=="quality" and "qwen3.6-35B-A3B" in avail:
return select_best_gpu(["qwen3.6-35B-A3B"], "hint_quality", agent) or {"model":"qwen3.6-35B-A3B","reason":"hint_quality"}
if hints.get("priority")=="code" and "qwen3.6-27B-code" in avail:
return select_best_gpu(["qwen3.6-27B-code"], "hint_code", agent) or {"model":"qwen3.6-27B-code","reason":"hint_code"}
first_msg = msgs[0].get("content","") if msgs else ""
words = len(first_msg.split()) if isinstance(first_msg, str) else 99
# TIER 1: Tiny - single-turn micro queries -> VLM (fastest)
if not sys and turns <= 1 and t <= 300 and words <= 100 and "gemma-4-12b" in avail:
if not is_gpu_busy("gemma-4-12b"):
return {"model":"gemma-4-12b","reason":"tiny"}
fallback = [m for m in ["qwen3.6-27B-code","qwen3.6-35B-A3B"] if m in avail]
result = select_best_gpu(fallback, "tiny_fallback", agent)
if result: return result
# TIER 2: Light - moderate chat -> Dense primary, saves VLM for vision/speed
if t <= 5000 and turns <= 4:
candidates = [m for m in ["qwen3.6-27B-code","gemma-4-12b","qwen3.6-35B-A3B"] if m in avail]
result = select_best_gpu(candidates, "light", agent)
if result: return result
# TIER 3: Medium - quality matters -> MoE primary, Dense fallback
if t <= 30000:
candidates = [m for m in ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"] if m in avail]
result = select_best_gpu(candidates, "medium", agent)
if result: return result
# TIER 4: Heavy - big context needs big model -> MoE->Dense->VLM
if t > 30000:
candidates = [m for m in ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"] if m in avail]
result = select_best_gpu(candidates, "heavy", agent)
if result: return result
# TIER 5: Default - best quality first
candidates = [m for m in ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"] if m in avail]
result = select_best_gpu(candidates, "default", agent)
if result: return result
return {"model":avail[0],"reason":"last_resort"}
-92
View File
@@ -1,92 +0,0 @@
def route(rd, tier, agent=""):
msgs = rd.get("messages",[]); t = estimate_tokens(msgs)
sys = any(m.get("role")=="system" for m in msgs)
turns = len([m for m in msgs if m.get("role") in ("user","assistant")])
hints = rd.get("routing_hints",{})
allowed = TIER_MODELS.get(tier, ["gemma-4-12b"])
avail = [m for m in available_models() if m in allowed]
if not avail: return {"model": allowed[0], "reason": "all_saturated", "saturated": True}
if all(is_gpu_busy(m) for m in avail):
return {"model": avail[0], "reason": "all_saturated", "saturated": True}
# GUARD: multimodal -> VLM only (sole vision model)
has_image = any(
isinstance(m.get("content"), list) and
any(p.get("type") == "image_url" for p in m["content"] if isinstance(p, dict))
for m in msgs
)
if has_image:
if "gemma-4-12b" in avail and not is_gpu_busy("gemma-4-12b"):
return {"model": "gemma-4-12b", "reason": "vision"}
elif "gemma-4-12b" in avail:
return {"model": "gemma-4-12b", "reason": "vision_saturated", "saturated": True}
else:
return {"model": allowed[0], "reason": "vision_unavailable"}
req = rd.get("model","auto")
if req != "auto":
target = req if req in avail else avail[0]
if is_gpu_busy(target) and req in allowed:
alts = [m for m in avail if m != target and m in allowed]
if alts:
alt = select_best_gpu(alts, "explicit", agent)
if alt: return alt
return {"model": target, "reason": "explicit"}
if hints:
if hints.get("priority")=="speed" and "gemma-4-12b" in avail:
return select_best_gpu(["gemma-4-12b"], "hint_speed", agent) or {"model":"gemma-4-12b","reason":"hint_speed"}
if hints.get("priority")=="quality" and "qwen3.6-35B-A3B" in avail:
return select_best_gpu(["qwen3.6-35B-A3B"], "hint_quality", agent) or {"model":"qwen3.6-35B-A3B","reason":"hint_quality"}
if hints.get("priority")=="code" and "qwen3.6-27B-code" in avail:
return select_best_gpu(["qwen3.6-27B-code"], "hint_code", agent) or {"model":"qwen3.6-27B-code","reason":"hint_code"}
first_msg = msgs[0].get("content","") if msgs else ""
words = len(first_msg.split()) if isinstance(first_msg, str) else 99
# TIER 1: Tiny - single-turn micro queries -> VLM (fastest)
if not sys and turns <= 1 and t <= 300 and words <= 100 and "gemma-4-12b" in avail:
if not is_gpu_busy("gemma-4-12b"):
return {"model":"gemma-4-12b","reason":"tiny"}
fallback = [m for m in ["qwen3.6-27B-code","qwen3.6-35B-A3B"] if m in avail]
result = select_best_gpu(fallback, "tiny_fallback", agent)
if result: return result
# TIER 2: Light - moderate chat -> Dense primary, saves VLM for vision/speed
if t <= 5000 and turns <= 4:
candidates = [m for m in ["qwen3.6-27B-code","gemma-4-12b","qwen3.6-35B-A3B"] if m in avail]
result = select_best_gpu(candidates, "light", agent)
if result: return result
# TIER 3: Medium - quality matters -> MoE primary (60%), Dense spillover (40%)
if t <= 30000:
candidates = moe_spillover(avail, ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"])
result = select_best_gpu(candidates, "medium", agent)
if result: return result
# TIER 4: Heavy - big context -> MoE primary (60%), Dense spillover (40%)
if t > 30000:
candidates = moe_spillover(avail, ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"])
result = select_best_gpu(candidates, "heavy", agent)
if result: return result
# TIER 5: Default - MoE primary (60%), Dense spillover (40%)
candidates = moe_spillover(avail, ["qwen3.6-35B-A3B","qwen3.6-27B-code","gemma-4-12b"])
result = select_best_gpu(candidates, "default", agent)
if result: return result
return {"model":avail[0],"reason":"last_resort"}
def moe_spillover(avail, default_order):
"""Spill 40% of MoE-first traffic to Dense to prevent Strix Halo overheating.
Only applies when MoE is first candidate, available, and not busy."""
import random
if (default_order[0] == "qwen3.6-35B-A3B"
and "qwen3.6-35B-A3B" in avail
and not is_gpu_busy("qwen3.6-35B-A3B")
and "qwen3.6-27B-code" in avail
and not is_gpu_busy("qwen3.6-27B-code")
and random.random() < 0.4):
# Swap: Dense first, MoE second
return ["qwen3.6-27B-code","qwen3.6-35B-A3B"] + [m for m in default_order[2:] if m in avail and m not in ("qwen3.6-27B-code","qwen3.6-35B-A3B")]
return [m for m in default_order if m in avail]
-1149
View File
File diff suppressed because it is too large Load Diff
+56
View File
@@ -0,0 +1,56 @@
#!/bin/bash
# Daily LiteLLM cost snapshot — cron at midnight
# Writes to /var/log/litellm/costs.csv on CT 116
MASTER="sk-litellm-7f96080dd99b15c36bd4b333b58a6796"
BASE="http://127.0.0.1:4001"
CSV="/var/log/litellm/costs.csv"
DATE=$(date -I)
YESTERDAY=$(date -I -d yesterday)
# Fetch yesterday's spend by key
RESP=$(curl -s "${BASE}/global/spend/logs?start_date=${YESTERDAY}T00:00:00&end_date=${YESTERDAY}T23:59:59&page_size=1000" \
-H "Authorization: Bearer $MASTER" 2>/dev/null)
# Parse total spend
TOTAL=$(echo "$RESP" | python3 -c "
import sys, json
data = json.load(sys.stdin)
total = sum(r.get('spend', 0) or 0 for r in data.get('data', []))
print(f'{total:.6f}')
" 2>/dev/null)
# Parse per-model spend
PER_MODEL=$(echo "$RESP" | python3 -c "
import sys, json
from collections import defaultdict
data = json.load(sys.stdin)
spend = defaultdict(float)
for r in data.get('data', []):
model = r.get('model', 'unknown')
spend[model] += r.get('spend', 0) or 0
for model, cost in sorted(spend.items()):
print(f'{model}:{cost:.6f}')
" 2>/dev/null)
# Per-key spend
PER_KEY=$(echo "$RESP" | python3 -c "
import sys, json
from collections import defaultdict
data = json.load(sys.stdin)
spend = defaultdict(float)
for r in data.get('data', []):
key = r.get('api_key', 'unknown')[:20]
spend[key] += r.get('spend', 0) or 0
for key, cost in sorted(spend.items()):
print(f'{key}:{cost:.6f}')
" 2>/dev/null)
# Write CSV
mkdir -p $(dirname "$CSV")
if [ ! -f "$CSV" ]; then
echo "date,total_cost,per_model,per_key" > "$CSV"
fi
echo "${DATE},${TOTAL},\"${PER_MODEL}\",\"${PER_KEY}\"" >> "$CSV"
echo "Snapshot: ${DATE} | Total: \$${TOTAL}"