From 8210fd905cfc30e46b418600673e7eb45d9aa786 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 21:39:50 +0000 Subject: [PATCH] fix: align contracts to 4-name LiteLLM registry (2026-09-12) - Remove retired names (qwen3.6-27B-code, qwen3.6-35B-udq4) from live alias claims - Update Strix Halo model to Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf (strix-moe, 256K ctx) - Fix litellm-health step 7 probe to gpu-vision (monitor key scoped) - Move qwen3.6-27B-code/35B-udq4 from raw-but-live to non-resolving in audit - Fold in pm2-self-heal: remove spoton-service (live PM2 set is 4/4) - Update hermes templates, key enforcement, timeout tables to live names --- audit-hermes-config.py | 3 +-- gpu-fleet.prose.md | 10 +++++----- gpu-self-heal.prose.md | 4 ++-- hermes-agent-baseline.prose.md | 3 +-- hermes-config-template.prose.md | 10 +++++----- hermes-key-enforcement.prose.md | 2 +- inference-optimization.prose.md | 2 +- litellm-client-timeouts.prose.md | 8 +++----- litellm-health.prose.md | 2 +- pm2-self-heal.prose.md | 1 - tests/test_audit_hermes_config_alias.py | 14 +++++++------- 11 files changed, 27 insertions(+), 32 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index faf6f79..2a69e1b 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -245,11 +245,10 @@ def audit(path): "gemma-4-12b": "gpu-vision", "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", "ornith-1.0-35b": "strix-moe", - } - raw_but_live = { "qwen3.6-27B-code": "gpu-dense", "qwen3.6-35B-udq4": "strix-moe", } + raw_but_live = {} for field_path, value in _iter_model_values(cfg): if value in non_resolving: check( diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index 3d4d13e..490649d 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -91,7 +91,7 @@ Single source of truth for models, aliases, rpm caps, weights and fallback chain CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not duplicate those values in contracts — read them there. -**Backward compatibility**: Old model-specific names (qwen3.6-27B-code, qwen3.5-9b-it) still work +**No backward compatibility**: Old model-specific names (qwen3.6-27B-code, qwen3.5-9b-it) are retired as of 2026-09-12 but are deprecated for agent configs. Only the stable aliases survive model swaps. `gemma-4-12b` is retired and returns 400 `Invalid model name`. @@ -207,7 +207,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. - **Router startup race**: Compose router.py doesn't call load_roster(). Reload thread sleeps 30s first. Fix: trigger roster reload via SSH after restart, or rebuild image with startup load_roster(). - **LiteLLM /metrics**: Requires auth. Prometheus uses `/health/liveliness` as workaround. -- **Strix Halo GPU**: Vulkan is the working backend (ROCm/HIP path abandoned — HSA runtime blocked on Debian 13). Build at `/root/llama.cpp/build-vk/`, commit `4fc4ec5` (2026-07-01), ggml 0.15.3 shared-lib arch. Mesa RADV 25.0.7, KHR_coopmat fast path active. ~70 tok/s gen, 532 tok/s prompt. Service: `strix-server.service` on port 8080, model: `qwen3.6-35B-udq4`, alias `strix-moe`, 128K context, flash-attn + q4 KV, multimodal (mmproj loaded). +- **Strix Halo GPU**: Vulkan is the working backend (ROCm/HIP path abandoned — HSA runtime blocked on Debian 13). Build at `/root/llama.cpp/build-vk/`, commit `4fc4ec5` (2026-07-01), ggml 0.15.3 shared-lib arch. Mesa RADV 25.0.7, KHR_coopmat fast path active. ~70 tok/s gen, 532 tok/s prompt. Service: `strix-server.service` on port 8080, model: `Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf`, alias `strix-moe`, 256K context (n_ctx 262144), --parallel 2 --kv-unified, flash-attn + q4 KV, multimodal (mmproj loaded). - **Port conflict detection (2026-07-05)**: All 3 GPU wrappers now detect ghost processes squatting port 8080 before starting. `.8` and `.110` use inline pre-start check in `llama-wrapper.sh`; `.15` uses `/usr/local/bin/port-cleanup.sh` ExecStartPre. Replaces the blanket `pkill -9 -x llama-server` on .15 which would kill ALL llama-server instances regardless of port. Ghost detection was the root cause of .8 crash-looping for 27+ restarts (stale pid 25836 squatting 8080 after OOM kill). - **Strix Halo thermal safeguard (2026-07-02)**: `strix-server.service` has `-n 8192` (hard generation cap per request). Without it, `--predict` defaults to -1 (infinity) — a runaway request from .123 (old Mumuni CT114 — now inside Abiba CT100 at .24) decoded 39,868 tokens over 24 min, pushing Tctl to 98°C (crit 89.8°C) and throttling 70→29 t/s. The cap bounds worst-case generation to ~5 min. Do NOT remove `-n` without a replacement ceiling. Sustained load hits ~84°C even at 92s; the APU is fanless/low-flow. Clients MUST also set `max_tokens`. - **Port 8080 firewall**: amdpve iptables restricts 8080 to 192.168.68.116 (LiteLLM/router host) only. All inbound connections are from .116 (LiteLLM proxied via nginx). Localhost curls hang (SYN dropped). Always test from .116. @@ -223,10 +223,10 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. | GPU | Model | Gen tok/s | Prompt tok/s | Baseline | Context | |-----|-------|-----------|--------------|----------|---------| | RTX 3090 (.8) | Qwen3.8-27B-Uncensored-Q4_K_M | **TBD** | — | — | **128K** | -| Strix Halo (.15) | qwen3.6-35B-udq4 | **65** | 140 | — | **128K** | +| Strix Halo (.15) | Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf (strix-moe) | **65** | 140 | — | **256K** | -Benchmarks from 2026-07-17. Strix Halo model: qwen3.6-35B-udq4. RTX 5070 MTP provides 2.7x speedup over pre-upgrade 70 tok/s. -All 3 GPUs now at 128K context (2026-07-17, reduced from 256K for stability). +Benchmarks from 2026-07-17. Strix Halo model: Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf (alias strix-moe), n_ctx 262144, --parallel 2 --kv-unified. RTX 5070 MTP provides 2.7x speedup over pre-upgrade 70 tok/s. +GPU contexts: RTX 3090 (.8) and RTX 5070 (.110) at 128K; Strix Halo (.15) at 256K (2026-09-12). Benchmarks run through LiteLLM proxy (192.168.68.116:4000) every 5 minutes. Degradation alerts fire at 30% (warning) and 50% (critical) below baseline. diff --git a/gpu-self-heal.prose.md b/gpu-self-heal.prose.md index 32019ca..89bbd1b 100644 --- a/gpu-self-heal.prose.md +++ b/gpu-self-heal.prose.md @@ -48,7 +48,7 @@ depends_on: | Alias | GPU | Host | Model | VRAM | Ctx | tok/s | Role | |-------|-----|------|-------|------|-----|-------|------| -| `strix-moe` | Strix Halo 64GB | ct15 (.15:8080) | qwen3.6-35B-udq4 | ~10/64GB (16%) | 128K | 62.9 | Compression, summarization, long docs | +| `strix-moe` | Strix Halo 64GB | ct15 (.15:8080) | Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf | ~10/64GB (16%) | 256K | 62.9 | Compression, summarization, long docs | Key notes: - All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path. @@ -145,7 +145,7 @@ Key notes: - **Detect**: Benchmark tok/s vs baseline for each GPU at current context (all 128K) - RTX 3090 (128K ctx, ThinkingCap): baseline 74.8 tok/s — currently at 74.9 (100%) - RTX 5070 (128K ctx, HauhauCS QAT): baseline 165.2 tok/s — currently at 169.6 (103%) - - Strix Halo (128K ctx, qwen3.6-35B-udq4): baseline 70.5 tok/s — currently at 62.9 (89%) + - Strix Halo (256K ctx, Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf): baseline 70.5 tok/s — currently at 62.9 (89%) - **Fix**: - If tok/s > baseline → context has headroom, consider increasing - If tok/s < 90% baseline → reduce context by 25% and retest diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index 9158303..2e2eb9a 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -200,8 +200,7 @@ the registry before applying. Key is injected via `infisical run --` wrapper at { "id": "syslog-auto" }, { "id": "strix-moe" }, { "id": "gpu-dense" }, - { "id": "gpu-vision" }, - { "id": "qwen3.6-27B-code" } + { "id": "gpu-vision" } ] } } diff --git a/hermes-config-template.prose.md b/hermes-config-template.prose.md index ea700c8..1a98545 100644 --- a/hermes-config-template.prose.md +++ b/hermes-config-template.prose.md @@ -92,7 +92,7 @@ work immediately after restart. ```yaml # ─── Model Selection ─── model: - default: # e.g., strix-moe, qwen3.6-27B-code, syslog-auto + default: # e.g., strix-moe, gpu-dense, syslog-auto provider: harness base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated path; /v1 also OK api_key_env: LITELLM_API_KEY # Injected via infisical run -- wrapper @@ -149,7 +149,7 @@ compression: # Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU. # gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning. # Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead. -# NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired +# NEVER use retired model names (qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired # and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents. auxiliary: vision: @@ -173,10 +173,10 @@ auxiliary: timeout: 300 # gpu-fleet: 300s for large-history summarization (was 60) # ─── Delegation / Heavy Aux (use gpu-dense = RTX 3090) ─── -# delegation.model and x_search.model use gpu-dense (NOT raw qwen3.6-27B-code). +# delegation.model and x_search.model use gpu-dense (NOT retired raw name). delegation: - model: gpu-dense # stable alias for RTX 3090 (was raw qwen3.6-27B-code) + model: gpu-dense # stable alias for RTX 3090 provider: harness base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK api_key_env: LITELLM_API_KEY @@ -273,7 +273,7 @@ The following MUST be identical across ALL profiles: - The `compression: max_context_window: 131072` MUST match actual GPU capacity (128K) ### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16) -- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations +- **RTX 3090 (24GB, 128K ctx, gpu-dense)**: Heavy reasoning, code gen, long conversations - **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K) - **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs - Agent profiles MUST route auxiliary tasks to the correct GPU: diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index 96b76d3..3c5e42c 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -182,7 +182,7 @@ The agent picks up the new key via `infisical run --` at gateway startup. # re-verify before applying. litellm_settings: default_key_generate_params: - models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] + models: ["syslog-auto", "gpu-dense", "gpu-vision", "strix-moe"] duration: null # ← permanent max_budget: 100 metadata: diff --git a/inference-optimization.prose.md b/inference-optimization.prose.md index 69a0d96..b0ee937 100644 --- a/inference-optimization.prose.md +++ b/inference-optimization.prose.md @@ -101,5 +101,5 @@ call enable-prompt-caching call verify-latency host: 192.168.68.116 - models: [syslog-auto, qwen3.6-27B-code, gpu-vision] + models: [syslog-auto, gpu-dense, gpu-vision, strix-moe] ``` diff --git a/litellm-client-timeouts.prose.md b/litellm-client-timeouts.prose.md index 7987ac7..810f1c1 100644 --- a/litellm-client-timeouts.prose.md +++ b/litellm-client-timeouts.prose.md @@ -26,9 +26,8 @@ description: > | Model | avg latency | avg TTFT | p-profile (24h) | |---|---|---|---| | syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load | -| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto | | strix-moe | 7.5s | — | Strix Halo, healthy | -| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy | +| gpu-vision (retired gemma-4-12b, RTX 5070) | 2.6s | — | RTX 5070, healthy | Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged backend), full recovery 07:00-08:00 with ZERO client failures once requests @@ -56,9 +55,8 @@ proxy queuing. - vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`. - compression: 300s (keep — this was already raised from 60 per gpu-fleet). - **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090 - (qwen3.6-27B-code backend, 23.0s avg) is the same speed class as - syslog-auto; delegation defaults that assume fast responses will 408 the - same way. + (gpu-dense backend) is the same speed class as syslog-auto; delegation + defaults that assume fast responses will 408 the same way. ### 3. Retry policy — backoff, not repetition diff --git a/litellm-health.prose.md b/litellm-health.prose.md index 4a430ca..2061843 100644 --- a/litellm-health.prose.md +++ b/litellm-health.prose.md @@ -144,7 +144,7 @@ contracts — read them there. Do not re-add retired names (`gemma-4-12b`, `gpu- 7. **Check model inference via LiteLLM** — Test one model on each GPU host. The health check runs on the **backend edge**, not the public edge, so these paths carry the `/litellm/` prefix: - - POST http://{{backend_host}}/litellm/v1/chat/completions model=qwen3.6-27B-code → expect 200 (RTX 3090, .8) + - POST http://{{backend_host}}/litellm/v1/chat/completions model=gpu-vision → expect 200 (RTX 5070, .110) - POST http://{{backend_host}}/litellm/v1/chat/completions model=gpu-vision → expect 200 (RTX 5070, .110) - POST http://{{backend_host}}/litellm/v1/chat/completions model=strix-moe → expect 200 (Strix Halo, .15) - Auth uses the dedicated `monitor` agent key, read on CT 116 from diff --git a/pm2-self-heal.prose.md b/pm2-self-heal.prose.md index 6081167..e3df0e6 100644 --- a/pm2-self-heal.prose.md +++ b/pm2-self-heal.prose.md @@ -9,7 +9,6 @@ description: > - abiba-telegram: { status: "online", uptime: string, restarts: number } - abiba-zulip: { status: "online", uptime: string, restarts: number } - gitea-runner: { status: "online", uptime: string, restarts: number } -- spoton-service: { status: "online", uptime: string, restarts: number } - zulip-watchdog: { status: "online", uptime: string, restarts: number } - last_check: timestamp diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index a129f03..d1c479e 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -132,20 +132,20 @@ def test_retired_alias_in_custom_providers_is_rejected(tmp_path): assert "RESULT: FAIL" in out -def test_raw_but_live_alias_warns_but_passes(tmp_path): - """Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs.""" +def test_retired_raw_name_fails(tmp_path): + """Retired raw names no longer resolve (400), so they fail; the 2026-09-12 registry change moved qwen3.6-27B-code from raw-but-live to non-resolving.""" code, out = _run_config( tmp_path, - "raw-qwen.yaml", + "retired-qwen.yaml", BASE.format(alias="gpu-vision").replace( "delegation:\n provider: harness", "delegation:\n provider: harness\n model: qwen3.6-27B-code", ), ) - assert code == 0, out - assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out - assert "prefer the stable alias gpu-dense" in out - assert "RESULT: PASS" in out + assert code != 0, out + assert "delegation.model = 'qwen3.6-27B-code' is retired and no longer resolves" in out + assert "use gpu-dense" in out + assert "RESULT: FAIL" in out def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):