- Added /metrics/gpu-health endpoint with live health scores (VRAM 40%, temp 30%, load 30%) - Added /metrics/latency endpoint for dashboard KPIs - Added GPU_LABELS for human-readable model names - Dashboard v2: rewired to real data endpoints - KPI cards: GPUs online, circuit trips, avg latency, req/min, active requests - Health scores from actual gpu_health_score() function - Rolling 60-sample history chart (real data, no simulation) - Status: green/yellow/red based on tripped circuits - No CDN dependency (pure CSS) - Auto-refresh every 15s - nginx: /dashboard/ serves static files with cache headers - docker-compose: dashboard volume mount Co-authored-by: Abiba <abiba@sysloggh.com>
103 lines
4.2 KiB
YAML
103 lines
4.2 KiB
YAML
version: '3.8'
|
|
|
|
services:
|
|
redis:
|
|
image: redis:7-alpine
|
|
container_name: harness-redis
|
|
restart: unless-stopped
|
|
ports:
|
|
- "127.0.0.1:6379:6379"
|
|
volumes:
|
|
- redis-data:/data
|
|
command: redis-server --appendonly yes --maxmemory 256mb --maxmemory-policy allkeys-lru
|
|
healthcheck:
|
|
test: ["CMD", "redis-cli", "ping"]
|
|
interval: 10s
|
|
timeout: 3s
|
|
retries: 5
|
|
|
|
router:
|
|
build: ./router
|
|
container_name: harness-router
|
|
restart: unless-stopped
|
|
ports:
|
|
- "127.0.0.1:9000:9000"
|
|
environment:
|
|
- REDIS_URL=redis://redis:6379
|
|
- GPU_MOE_URL=http://192.168.68.15:8080/v1
|
|
- GPU_DENSE_URL=http://192.168.68.8:8080/v1
|
|
- GPU_LIGHT_URL=http://192.168.68.110:8080/v1
|
|
- API_KEYS={"sk-syslog-local-master-key":{"tier":"enterprise","agent":"admin","deprecated":true},"sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64":{"tier":"enterprise","agent":"admin"},"sk-syslog-abiba":{"tier":"enterprise","agent":"Abiba","deprecated":true},"sk-856ffb0bbb-e5aaf78b10054eca608f8fbcbd73a889":{"tier":"enterprise","agent":"Abiba"},"sk-syslog-mumuni":{"tier":"enterprise","agent":"Mumuni","deprecated":true},"sk-b57e6e042e-47573660114f3138c852c47f62da807e":{"tier":"enterprise","agent":"Mumuni"},"sk-syslog-tanko":{"tier":"enterprise","agent":"Tanko","deprecated":true},"sk-620a05e95a-e93d875476b650a4d1137249ead8eaa7":{"tier":"enterprise","agent":"Tanko"},"sk-syslog-koby":{"tier":"enterprise","agent":"Koby","deprecated":true},"sk-eb3e6fc1c0-de1bf2edf35a53cb3749a2400483fdee":{"tier":"enterprise","agent":"Koby"},"sk-syslog-kagenz0":{"tier":"enterprise","agent":"Kagenz0","deprecated":true},"sk-12b66b3392-b548aed9138aeb6f698e8e521650ed9b":{"tier":"enterprise","agent":"Kagenz0"},"sk-syslog-koonimo":{"tier":"enterprise","agent":"Koonimo","deprecated":true},"sk-680d06686c-00ee8bf9dc3c93b276af122d49a14dfe":{"tier":"enterprise","agent":"Koonimo"},"sk-starter-abc123":{"tier":"starter","agent":"test-starter","deprecated":true},"sk-55da55907a-1bd7ff344e26feda50e9ac2219697860":{"tier":"starter","agent":"test-starter"},"sk-professional-xyz789":{"tier":"professional","agent":"test-pro","deprecated":true},"sk-b5159863e6-8df3ae52fb958cfe76cc2888c8c8e676":{"tier":"professional","agent":"test-pro"}}
|
|
- ADMIN_KEY=sk-admin-ee09fffd04978b61a1569ac670c68814
|
|
healthcheck:
|
|
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:9000/health')"]
|
|
interval: 30s
|
|
timeout: 15s
|
|
retries: 3
|
|
depends_on:
|
|
redis:
|
|
condition: service_healthy
|
|
|
|
litellm:
|
|
image: ghcr.io/berriai/litellm:main-stable
|
|
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
|
container_name: harness-litellm
|
|
restart: unless-stopped
|
|
ports:
|
|
- "127.0.0.1:8081:4000"
|
|
volumes:
|
|
- ./litellm_config.yaml:/app/config.yaml
|
|
environment:
|
|
- LITELLM_MASTER_KEY=sk-sys...-key
|
|
extra_hosts:
|
|
- "host.docker.internal:host-gateway"
|
|
healthcheck:
|
|
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
retries: 3
|
|
depends_on:
|
|
redis:
|
|
condition: service_healthy
|
|
|
|
nginx:
|
|
image: nginx:alpine
|
|
container_name: harness-nginx
|
|
restart: unless-stopped
|
|
ports:
|
|
- "80:80"
|
|
volumes:
|
|
- ./nginx/nginx.conf:/etc/nginx/nginx.conf:ro
|
|
- ./dashboard:/opt/inference-harness/dashboard:ro
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-f", "http://127.0.0.1/health"]
|
|
interval: 30s
|
|
timeout: 15s
|
|
retries: 3
|
|
depends_on:
|
|
- litellm
|
|
- dashboard
|
|
|
|
dashboard:
|
|
build: ./dashboard
|
|
container_name: harness-dashboard
|
|
restart: unless-stopped
|
|
ports:
|
|
- "127.0.0.1:3000:3000"
|
|
environment:
|
|
- REDIS_URL=redis://redis:6379
|
|
- GPU_SIDECARS=192.168.68.15:8090,192.168.68.8:8090,192.168.68.110:8090
|
|
healthcheck:
|
|
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:3000/health')"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
retries: 3
|
|
depends_on:
|
|
- redis
|
|
|
|
volumes:
|
|
redis-data:
|
|
|
|
# LiteLLM command override to load config
|
|
# (appended to fix config loading issue)
|