From 221f9f79f3afec7e5be5cf9f4f34c4b87f8b6126 Mon Sep 17 00:00:00 2001 From: abiba Date: Sat, 12 Sep 2026 16:11:46 +0000 Subject: [PATCH 1/8] fix(audit): reject retired aliases; sweep gpu-light/gemma-4-12b to gpu-vision audit-hermes-config.py Rule 8 required auxiliary.vision.model and auxiliary.web_extract.model to equal the retired 'gpu-light', so a config adopting the live canonical 'gpu-vision' FAILED our own audit - the audit was enforcing a dead alias (400 Invalid model name). Rule 8 now requires gpu-vision; retired names gpu-light/crew-auto join the raw-name rejection set; the guidance message names the live aliases. Sweep of the remaining references: gpu-self-heal stops canonicalizing gpu-light; hermes-config-template, hermes-agent-baseline, hermes-key-enforcement, inference-optimization, litellm-client-timeouts and gpu-fleet now use the live gpu-vision alias. Where a file restated model/rpm/weight/fallback state it now points at CT 116 /opt/inference-harness/litellm_config.yaml instead of duplicating it. koby's .129 config is report-only and recorded, not edited. Adds tests/test_audit_hermes_config_alias.py: executes the audit CLI and asserts gpu-vision passes while gpu-light and gemma-4-12b fail. --- audit-hermes-config.py | 20 +++--- gpu-fleet.prose.md | 2 +- gpu-self-heal.prose.md | 17 ++--- hermes-agent-baseline.prose.md | 9 ++- hermes-config-template.prose.md | 29 ++++---- hermes-key-enforcement.prose.md | 6 +- inference-optimization.prose.md | 2 +- litellm-api-keys.prose.md | 4 +- litellm-client-timeouts.prose.md | 4 +- tests/test_audit_hermes_config_alias.py | 89 +++++++++++++++++++++++++ 10 files changed, 139 insertions(+), 43 deletions(-) create mode 100644 tests/test_audit_hermes_config_alias.py diff --git a/audit-hermes-config.py b/audit-hermes-config.py index 773fff2..b0f3323 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -89,15 +89,16 @@ def audit(path): ) # --- Rule 8: GPU Workload Distribution --- + # gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision. check( - aux.get("vision", {}).get("model") == "gpu-light", + aux.get("vision", {}).get("model") == "gpu-vision", "Rule 8", - f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias", + f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias", ) check( - aux.get("web_extract", {}).get("model") == "gpu-light", + aux.get("web_extract", {}).get("model") == "gpu-vision", "Rule 8", - f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias", + f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias", ) # --- Rule 9: Compression Threshold --- @@ -178,8 +179,11 @@ def audit(path): f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", ) - # --- No raw model names (Rule 7/8 spirit) --- - raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} + # --- No raw or retired model names (Rule 7/8 spirit) --- + # Retired names are rejected by the alias rules above; a config that names them directly is + # flagged here too. gpu-light/gemma-4-12b/crew-auto were retired 2026-09-12. + raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b", + "gpu-light", "crew-auto"} for section_path, section_dict in [ ("model", model), ("compression", comp), ("auxiliary.vision", aux.get("vision", {})), @@ -191,8 +195,8 @@ def audit(path): if m in raw_names: warn( "Rule 7/8", - f"{section_path}.model = {m!r} — raw model name, use stable alias instead " - f"(gpu-light, gpu-dense, strix-moe, syslog-auto)", + f"{section_path}.model = {m!r} — raw or retired model name, use a live stable alias " + f"instead (gpu-vision, gpu-dense, strix-moe, syslog-auto)", ) # --- Report --- diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index fefebba..1bef170 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -216,7 +216,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. - **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled. - **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds. - **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down. -- **Alias-retirement follow-up (2026-09-12)**: the agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, and the executable `audit-hermes-config.py` still reference the retired `gpu-light`/`gemma-4-12b` names, and `audit-hermes-config.py` Rule 8 currently fails a config whose vision/web_extract model is `gpu-vision`. These are tracked separately and are intentionally NOT updated in this change. +- **Alias-retirement sweep (2026-09-12)**: `gemma-4-12b`, `gpu-light` and `crew-auto` are retired and replaced by `gpu-vision` / no cap respectively. The agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, `gpu-self-heal.prose.md`, `hermes-key-enforcement.prose.md`, `inference-optimization.prose.md`, `litellm-client-timeouts.prose.md` and the executable `audit-hermes-config.py` were all updated to the live canonical alias in the same change. **koby's config on .129 still names `gpu-light` (and `gemma-4-E4B`); .129 is report-only, so that is recorded for its owner and NOT edited here.** ## GPU Inference Benchmarks (Current) diff --git a/gpu-self-heal.prose.md b/gpu-self-heal.prose.md index 04ef8e4..cdc6264 100644 --- a/gpu-self-heal.prose.md +++ b/gpu-self-heal.prose.md @@ -12,7 +12,7 @@ description: > Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing. Benchmark baselines refreshed to live values. Prometheus exporters removed — not deployed; fall back to direct sidecar probes. - Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet. + Stable role-based aliases (strix-moe, gpu-dense, gpu-vision) from gpu-fleet. agent: abiba depends_on: - gpu-monitor.prose.md (live data source on .24:9100) @@ -52,8 +52,8 @@ depends_on: Key notes: - All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path. -- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. -- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was. +- Stable aliases (gpu-dense, gpu-vision, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. The retired names `gpu-light` and `gemma-4-12b` were superseded by `gpu-vision` on 2026-09-12 and no longer resolve (400 `Invalid model name`). +- The RTX 5070 is the fastest endpoint per token — `gpu-vision` is its canonical alias. Route vision/web/light work there first. The RTX 5070 model is multimodal (image+text). - Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads. - RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom). - RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes). @@ -157,13 +157,14 @@ Key notes: ### Rule 10: Workload Distribution Optimization (updated 2026-07-18) - **Detect**: GPU roles misaligned with hardware capabilities - **Target distribution**: - - RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM). - - RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM). - - Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM). + - RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). + - RTX 5070 (gpu-vision, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. + - Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). - **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first. +- **Weights are not restated here** — the live `syslog-auto` pool weights and rpm caps live in CT 116 `/opt/inference-harness/litellm_config.yaml`, the single source of truth. - **Fix**: - Alert if any GPU is handling workload outside its designated role - - Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe) + - Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-vision, strix-moe) - Track per-GPU request distribution via LiteLLM spend logs - **Verify**: Each GPU's request pattern matches its designated role within 24h - **Escalate**: If role mismatch persists >48h → agent alias audit needed @@ -311,6 +312,6 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab If monitor response > 1MB, log a warning and skip the cycle rather than crashing. ### L6: Stable Aliases Replace Model Names -- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15. +- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15; `gpu-light` was superseded by `gpu-vision` on 2026-09-12. - Self-heal must use aliases for reporting and alerting, not model-specific names. - **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier. diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index 5fcd5f1..c9513fc 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -80,7 +80,7 @@ custom_providers: auxiliary: vision: provider: harness - model: gemma-4-12b # or syslog-auto + model: gpu-vision # RTX 5070 stable alias (or syslog-auto) base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround @@ -95,7 +95,7 @@ auxiliary: threshold: 0.65 target_ratio: 0.3 provider: harness - model: syslog-auto # or gemma-4-12b + model: syslog-auto # or gpu-vision base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround @@ -197,9 +197,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup: { "id": "syslog-auto" }, { "id": "strix-moe" }, { "id": "gpu-dense" }, - { "id": "gpu-light" }, - { "id": "qwen3.6-27B-code" }, - { "id": "gemma-4-12b" } + { "id": "gpu-vision" }, + { "id": "qwen3.6-27B-code" } ] } } diff --git a/hermes-config-template.prose.md b/hermes-config-template.prose.md index 076ff04..485f2ee 100644 --- a/hermes-config-template.prose.md +++ b/hermes-config-template.prose.md @@ -102,7 +102,7 @@ model: # and falls back to 256K when /v1/models lacks a context # field (llama-server does). Without this override, agents # silently run syslog-auto at 256K (verified 2026-08-09). - # Set 65536 if using gemma-4-12b directly (tight VRAM). + # Set 65536 if pinning a single model directly (tight VRAM). fallback_providers: provider: deepseek @@ -143,25 +143,25 @@ compression: # ─── Auxiliary Tasks (CONSISTENCY RULE) ─── # All auxiliary services MUST use identical model, base_url, and api_key_env: -# model: gpu-light # stable alias (NOT raw "gemma-4-12b") +# model: gpu-vision # stable alias (NOT a raw model name) # base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK # api_key_env: LITELLM_API_KEY # Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU. -# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning. +# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning. # Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead. -# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4) -# in agent configs — use the stable aliases so model swaps don't break agents. +# NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired +# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents. auxiliary: vision: provider: harness - model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b) + model: gpu-vision # stable alias for RTX 5070 base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK api_key_env: LITELLM_API_KEY timeout: 60 download_timeout: 30 web_extract: provider: harness - model: gpu-light # stable alias for RTX 5070 + model: gpu-vision # stable alias for RTX 5070 base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK api_key_env: LITELLM_API_KEY timeout: 30 @@ -252,10 +252,11 @@ The following MUST be identical across ALL profiles: - For agents needing longer outputs: raise to 8192, but never omit ### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16) -- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized) +- Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized) - Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized) - **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b` - (it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls. + (do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml` + is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls. - **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.** The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can @@ -273,11 +274,11 @@ The following MUST be identical across ALL profiles: ### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16) - **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations -- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K) +- **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K) - **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs - Agent profiles MUST route auxiliary tasks to the correct GPU: - - `auxiliary.vision.model: gemma-4-12b` (RTX 5070) - - `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070) + - `auxiliary.vision.model: gpu-vision` (RTX 5070) + - `auxiliary.web_extract.model: gpu-vision` (RTX 5070) - `auxiliary.compression.model: syslog-auto` (Strix Halo) - Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing - For 128K context window: `threshold: 0.65` (fires at ~85K tokens) @@ -297,8 +298,8 @@ The following MUST be identical across ALL profiles: - **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10. - **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto` - **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json` -- `syslog-auto` is the LiteLLM routing model — it load-balances between strix-moe - and qwen3.6-27B-code, with gemma-4-12b as fallback. Using it protects against: +- `syslog-auto` is the LiteLLM routing model — it load-balances across the live pool + (see CT 116 `/opt/inference-harness/litellm_config.yaml` for the current members and weights). Using it protects against: - Model name typos that cause 403 errors and silent worker failures - Single GPU downtime (routing falls back automatically) - Key/model authorization mismatches diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index 5e46736..45d0296 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -178,7 +178,7 @@ The agent picks up the new key via `infisical run --` at gateway startup. # In litellm_config.yaml — ensures all future keys inherit these defaults: litellm_settings: default_key_generate_params: - models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b"] + models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] duration: null # ← permanent max_budget: 100 metadata: @@ -286,13 +286,13 @@ auxiliary: api_key: sk- # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain) api_key_env: LITELLM_API_KEY base_url: http://192.168.68.116/litellm/v1 - model: gemma-4-12b + model: gpu-vision provider: harness compression: api_key: sk- # ← workaround (same as above) api_key_env: LITELLM_API_KEY base_url: http://192.168.68.116/litellm/v1 - model: gemma-4-12b + model: gpu-vision provider: harness ``` diff --git a/inference-optimization.prose.md b/inference-optimization.prose.md index fd28d22..bd5dbb2 100644 --- a/inference-optimization.prose.md +++ b/inference-optimization.prose.md @@ -99,5 +99,5 @@ call enable-prompt-caching call verify-latency host: 192.168.68.116 - models: [syslog-auto, qwen3.6-27B-code, gemma-4-12b] + models: [syslog-auto, qwen3.6-27B-code, gpu-vision] ``` diff --git a/litellm-api-keys.prose.md b/litellm-api-keys.prose.md index c42897d..27b8e94 100644 --- a/litellm-api-keys.prose.md +++ b/litellm-api-keys.prose.md @@ -68,7 +68,9 @@ description: > - Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date) - Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" } - Duration is null (permanent) — inherited from litellm default_key_generate_params - - Set models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b", "strix-moe", "gpu-dense", "gpu-light", "qwen3.6-35B-udq4"] + - Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped, + and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add + retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12). - Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed). - Return the new key 5. **If action == "rotate"**: diff --git a/litellm-client-timeouts.prose.md b/litellm-client-timeouts.prose.md index 4e7048f..a7d5aca 100644 --- a/litellm-client-timeouts.prose.md +++ b/litellm-client-timeouts.prose.md @@ -28,7 +28,7 @@ description: > | syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load | | qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto | | strix-moe | 7.5s | — | Strix Halo, healthy | -| gemma-4-12b | 2.6s | — | RTX 5070, healthy | +| gpu-vision | 2.6s | — | RTX 5070, healthy | Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged backend), full recovery 07:00-08:00 with ZERO client failures once requests @@ -53,7 +53,7 @@ proxy queuing. ### 2. Auxiliary tasks — keep template timeouts, one correction -- vision: 60s (keep), web_extract: 30s (keep) — gemma-4-12b averages 2.6s; +- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s; these are fine. - compression: 300s (keep — this was already raised from 60 per gpu-fleet). - **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090 diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py new file mode 100644 index 0000000..4416318 --- /dev/null +++ b/tests/test_audit_hermes_config_alias.py @@ -0,0 +1,89 @@ +"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py. + +WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias. +Rule 8 required `auxiliary.vision.model == "gpu-light"` and +`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor +`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the +live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias +therefore FAILED our own audit, so the audit was actively enforcing a broken config. + +These tests execute the real CLI (`python3 audit-hermes-config.py `) and assert +observable behaviour — exit code and the emitted rule message — for the live alias and +for both retired names. No network, vault, or SSH access is required. +""" +from __future__ import annotations + +import pathlib +import subprocess +import sys + +ROOT = pathlib.Path(__file__).resolve().parent.parent +AUDIT = ROOT / "audit-hermes-config.py" + +BASE = """ +model: + api_key: "" + api_key_env: LITELLM_API_KEY + base_url: http://192.168.68.116/v1 + max_tokens: 4096 + default: syslog-auto + provider: harness +fallback_providers: + provider: deepseek + model: deepseek-v4-flash + api_key_env: DEEPSEEK_API_KEY +compression: + model: syslog-auto + provider: harness + threshold: 0.65 + max_context_window: 131072 +auxiliary: + vision: + model: {alias} + provider: harness + web_extract: + model: {alias} + provider: harness + compression: + model: syslog-auto + provider: harness +delegation: + provider: harness +custom_providers: + - name: harness + key_env: LITELLM_API_KEY + base_url: http://192.168.68.116/v1 +""" + + +def _run(tmp_path, alias): + cfg = tmp_path / f"{alias}.yaml" + cfg.write_text(BASE.format(alias=alias)) + proc = subprocess.run( + [sys.executable, str(AUDIT), str(cfg)], + capture_output=True, text=True, + ) + return proc.returncode, proc.stdout + + +def test_live_canonical_alias_passes(tmp_path): + """The RTX 5070 alias that actually resolves must satisfy Rule 8.""" + code, out = _run(tmp_path, "gpu-vision") + assert code == 0, out + assert "RESULT: PASS" in out + + +def test_retired_gpu_light_is_rejected(tmp_path): + """A config pinned to the retired alias must fail, not pass.""" + code, out = _run(tmp_path, "gpu-light") + assert code == 1, out + assert "auxiliary.vision.model must be gpu-vision" in out + assert "RESULT: FAIL" in out + + +def test_retired_gemma_is_rejected(tmp_path): + """The retired raw model name must fail Rule 8 as well.""" + code, out = _run(tmp_path, "gemma-4-12b") + assert code == 1, out + assert "auxiliary.vision.model must be gpu-vision" in out + assert "RESULT: FAIL" in out From 9cac3589cf82500ac58917b3283a2cb6afb58c9f Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:19:43 +0000 Subject: [PATCH 2/8] no-mistakes(review): Fix compression example alias; make retired aliases fail audit --- audit-hermes-config.py | 25 ++++++++++++++++++------- hermes-agent-baseline.prose.md | 6 ++++-- hermes-key-enforcement.prose.md | 8 ++++++-- inference-optimization.prose.md | 2 ++ tests/test_audit_hermes_config_alias.py | 18 ++++++++++++++++++ 5 files changed, 48 insertions(+), 11 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index b0f3323..4d14234 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -180,10 +180,15 @@ def audit(path): ) # --- No raw or retired model names (Rule 7/8 spirit) --- - # Retired names are rejected by the alias rules above; a config that names them directly is - # flagged here too. gpu-light/gemma-4-12b/crew-auto were retired 2026-09-12. - raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b", - "gpu-light", "crew-auto"} + # Retired names are a hard failure in every audited model-bearing section: a config that + # names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were + # retired 2026-09-12. Raw-but-live names are a warning only. + retired_names = { + "gpu-light": "gpu-vision", + "gemma-4-12b": "gpu-vision", + "crew-auto": "no replacement alias (64K crew cap removed)", + } + raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} for section_path, section_dict in [ ("model", model), ("compression", comp), ("auxiliary.vision", aux.get("vision", {})), @@ -192,11 +197,17 @@ def audit(path): ("delegation", deleg), ]: m = section_dict.get("model", "") - if m in raw_names: + if m in retired_names: + check( + False, + "Rule 7/8", + f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}", + ) + elif m in raw_names: warn( "Rule 7/8", - f"{section_path}.model = {m!r} — raw or retired model name, use a live stable alias " - f"instead (gpu-vision, gpu-dense, strix-moe, syslog-auto)", + f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead " + f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)", ) # --- Report --- diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index c9513fc..b77044c 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -95,7 +95,7 @@ auxiliary: threshold: 0.65 target_ratio: 0.3 provider: harness - model: syslog-auto # or gpu-vision + model: syslog-auto # Rule 7: compression must be syslog-auto base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround @@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension. Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`. **models.json** — Must only list models authorized for the agent's LiteLLM key. -Key is injected via `infisical run --` wrapper at PM2 startup: +`/v1/models` is key-scoped and the live registry is CT 116 +`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read +the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup: ```json { "providers": { diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index 45d0296..fb7886c 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -175,7 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup. - **Max budget**: $100 per key (config default). ```yaml -# In litellm_config.yaml — ensures all future keys inherit these defaults: +# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block +# today, and a key generated with no explicit models comes back with an EMPTY models list. This is +# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped +# `/v1/models` view), so re-read the live registry at CT 116 +# /opt/inference-harness/litellm_config.yaml before applying. litellm_settings: default_key_generate_params: models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] @@ -292,7 +296,7 @@ auxiliary: api_key: sk- # ← workaround (same as above) api_key_env: LITELLM_API_KEY base_url: http://192.168.68.116/litellm/v1 - model: gpu-vision + model: syslog-auto provider: harness ``` diff --git a/inference-optimization.prose.md b/inference-optimization.prose.md index bd5dbb2..beb18a2 100644 --- a/inference-optimization.prose.md +++ b/inference-optimization.prose.md @@ -96,6 +96,8 @@ call enable-prompt-caching hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110] -- Phase 5: Verify end-to-end latency +-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is +-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use. call verify-latency host: 192.168.68.116 diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index 4416318..f14016a 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -87,3 +87,21 @@ def test_retired_gemma_is_rejected(tmp_path): assert code == 1, out assert "auxiliary.vision.model must be gpu-vision" in out assert "RESULT: FAIL" in out + + +def test_retired_alias_in_delegation_is_rejected(tmp_path): + """delegation.model has no dedicated value rule, so a retired name there used to PASS.""" + cfg = tmp_path / "delegation-gpu-light.yaml" + cfg.write_text( + BASE.format(alias="gpu-vision").replace( + "delegation:\n provider: harness", + "delegation:\n provider: harness\n model: gpu-light", + ) + ) + proc = subprocess.run( + [sys.executable, str(AUDIT), str(cfg)], + capture_output=True, text=True, + ) + assert proc.returncode == 1, proc.stdout + assert "delegation.model = 'gpu-light' is retired" in proc.stdout + assert "RESULT: FAIL" in proc.stdout From 9e581ab20388503e6f6ea976a377a46471a17332 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:26:14 +0000 Subject: [PATCH 3/8] no-mistakes(review): Complete retired-alias field coverage; fix misleading example labels --- audit-hermes-config.py | 72 +++++++++++++++++-------- hermes-agent-baseline.prose.md | 2 +- hermes-key-enforcement.prose.md | 10 ++-- tests/test_audit_hermes_config_alias.py | 62 ++++++++++++++++----- 4 files changed, 105 insertions(+), 41 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index 4d14234..a55442c 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -36,6 +36,43 @@ def warn(rule, message): WARNINGS.append(f"[{rule}] {message}") +def _provider_model_fields(label, value): + """Model-bearing fields from a dict-shaped or list-shaped provider section.""" + fields = [] + if isinstance(value, dict): + fields.append((f"{label}.model", value.get("model"))) + elif isinstance(value, list): + for i, item in enumerate(value): + if isinstance(item, dict): + fields.append((f"{label}[{i}].model", item.get("model"))) + return fields + + +def _model_name_fields(cfg): + """Every model-name-bearing field in an agent config, as (path, value) pairs.""" + fields = [] + model = cfg.get("model") or {} + if isinstance(model, dict): + for key, value in model.items(): + if key == "default" or "model" in key: + fields.append((f"model.{key}", value)) + comp = cfg.get("compression") or {} + if isinstance(comp, dict): + fields.append(("compression.model", comp.get("model"))) + aux = cfg.get("auxiliary") or {} + if isinstance(aux, dict): + for name in ("vision", "web_extract", "compression"): + section = aux.get(name) or {} + if isinstance(section, dict): + fields.append((f"auxiliary.{name}.model", section.get("model"))) + deleg = cfg.get("delegation") or {} + if isinstance(deleg, dict): + fields.append(("delegation.model", deleg.get("model"))) + fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers"))) + fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers"))) + return fields + + def audit(path): with open(path) as f: cfg = yaml.safe_load(f) @@ -180,34 +217,23 @@ def audit(path): ) # --- No raw or retired model names (Rule 7/8 spirit) --- - # Retired names are a hard failure in every audited model-bearing section: a config that - # names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were - # retired 2026-09-12. Raw-but-live names are a warning only. - retired_names = { - "gpu-light": "gpu-vision", + # Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets + # 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ + # crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. + bad_model_names = { "gemma-4-12b": "gpu-vision", - "crew-auto": "no replacement alias (64K crew cap removed)", + "gpu-light": "gpu-vision", + "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", + "qwen3.6-27B-code": "gpu-dense", + "qwen3.6-35B-udq4": "strix-moe", + "ornith-1.0-35b": "strix-moe", } - raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} - for section_path, section_dict in [ - ("model", model), ("compression", comp), - ("auxiliary.vision", aux.get("vision", {})), - ("auxiliary.web_extract", aux.get("web_extract", {})), - ("auxiliary.compression", aux.get("compression", {})), - ("delegation", deleg), - ]: - m = section_dict.get("model", "") - if m in retired_names: + for field_path, value in _model_name_fields(cfg): + if value in bad_model_names: check( False, "Rule 7/8", - f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}", - ) - elif m in raw_names: - warn( - "Rule 7/8", - f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead " - f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)", + f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}", ) # --- Report --- diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index b77044c..71ab79f 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -80,7 +80,7 @@ custom_providers: auxiliary: vision: provider: harness - model: gpu-vision # RTX 5070 stable alias (or syslog-auto) + model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux) base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index fb7886c..b712c0e 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -175,11 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup. - **Max budget**: $100 per key (config default). ```yaml -# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block -# today, and a key generated with no explicit models comes back with an EMPTY models list. This is -# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped -# `/v1/models` view), so re-read the live registry at CT 116 -# /opt/inference-harness/litellm_config.yaml before applying. +# NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no +# default_key_generate_params block today, and a key generated with no explicit models comes back +# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to +# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and +# re-verify before applying. litellm_settings: default_key_generate_params: models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index f14016a..b489d67 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -56,9 +56,9 @@ custom_providers: """ -def _run(tmp_path, alias): - cfg = tmp_path / f"{alias}.yaml" - cfg.write_text(BASE.format(alias=alias)) +def _run_config(tmp_path, name, text): + cfg = tmp_path / name + cfg.write_text(text) proc = subprocess.run( [sys.executable, str(AUDIT), str(cfg)], capture_output=True, text=True, @@ -66,6 +66,10 @@ def _run(tmp_path, alias): return proc.returncode, proc.stdout +def _run(tmp_path, alias): + return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias)) + + def test_live_canonical_alias_passes(tmp_path): """The RTX 5070 alias that actually resolves must satisfy Rule 8.""" code, out = _run(tmp_path, "gpu-vision") @@ -89,19 +93,53 @@ def test_retired_gemma_is_rejected(tmp_path): assert "RESULT: FAIL" in out +def test_corrected_compression_example_passes(tmp_path): + """The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS.""" + code, out = _run(tmp_path, "gpu-vision") + assert code == 0, out + assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out + assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out + assert "RESULT: PASS" in out + + def test_retired_alias_in_delegation_is_rejected(tmp_path): """delegation.model has no dedicated value rule, so a retired name there used to PASS.""" - cfg = tmp_path / "delegation-gpu-light.yaml" - cfg.write_text( + code, out = _run_config( + tmp_path, + "delegation-gpu-light.yaml", BASE.format(alias="gpu-vision").replace( "delegation:\n provider: harness", "delegation:\n provider: harness\n model: gpu-light", - ) + ), ) - proc = subprocess.run( - [sys.executable, str(AUDIT), str(cfg)], - capture_output=True, text=True, + assert code == 1, out + assert "delegation.model = 'gpu-light' is retired" in out + assert "RESULT: FAIL" in out + + +def test_retired_alias_in_custom_providers_is_rejected(tmp_path): + """custom_providers[*].model is model-bearing; a retired name there must fail.""" + code, out = _run_config( + tmp_path, + "custom-provider-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace( + " - name: harness\n key_env: LITELLM_API_KEY", + " - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY", + ), ) - assert proc.returncode == 1, proc.stdout - assert "delegation.model = 'gpu-light' is retired" in proc.stdout - assert "RESULT: FAIL" in proc.stdout + assert code == 1, out + assert "custom_providers[0].model = 'gpu-light'" in out + assert "RESULT: FAIL" in out + + +def test_raw_alias_is_rejected(tmp_path): + """Raw-but-live model names must fail too, pointing at the stable alias.""" + code, out = _run_config( + tmp_path, + "raw-qwen.yaml", + BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"), + ) + assert code == 1, out + assert "model.default = 'qwen3.6-27B-code'" in out + assert "gpu-dense" in out + assert "RESULT: FAIL" in out From ce48070f21fdb021ac60fa6bc423dab5d776587f Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:33:58 +0000 Subject: [PATCH 4/8] no-mistakes(review): Derive model fields recursively; fix historical latency and key claims --- audit-hermes-config.py | 86 +++++++++++++++---------- hermes-key-enforcement.prose.md | 2 +- litellm-client-timeouts.prose.md | 5 +- tests/test_audit_hermes_config_alias.py | 43 +++++++++++++ 4 files changed, 97 insertions(+), 39 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index a55442c..561dc99 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -36,41 +36,57 @@ def warn(rule, message): WARNINGS.append(f"[{rule}] {message}") -def _provider_model_fields(label, value): - """Model-bearing fields from a dict-shaped or list-shaped provider section.""" - fields = [] - if isinstance(value, dict): - fields.append((f"{label}.model", value.get("model"))) - elif isinstance(value, list): - for i, item in enumerate(value): - if isinstance(item, dict): - fields.append((f"{label}[{i}].model", item.get("model"))) - return fields +# Derivation rule: a model name is any scalar under a mapping key named `model` or +# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model +# name lives under `default`/`model`/`model_name` inside that section, so it is descended +# specially. The only other exception is key `models` (litellm key-generation params carry a +# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by +# hand. +MODEL_KEYS = ("model", "model_name") +MODEL_SECTION_KEYS = ("default", "model", "model_name") +MODEL_LIST_KEYS = ("models",) -def _model_name_fields(cfg): - """Every model-name-bearing field in an agent config, as (path, value) pairs.""" - fields = [] - model = cfg.get("model") or {} - if isinstance(model, dict): - for key, value in model.items(): - if key == "default" or "model" in key: - fields.append((f"model.{key}", value)) - comp = cfg.get("compression") or {} - if isinstance(comp, dict): - fields.append(("compression.model", comp.get("model"))) - aux = cfg.get("auxiliary") or {} - if isinstance(aux, dict): - for name in ("vision", "web_extract", "compression"): - section = aux.get(name) or {} - if isinstance(section, dict): - fields.append((f"auxiliary.{name}.model", section.get("model"))) - deleg = cfg.get("delegation") or {} - if isinstance(deleg, dict): - fields.append(("delegation.model", deleg.get("model"))) - fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers"))) - fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers"))) - return fields +def _iter_model_values(node, path=""): + """Yield (path, value) for every model-name-bearing scalar in a config.""" + if isinstance(node, dict): + for key, value in node.items(): + child = f"{path}.{key}" if path else key + if key in MODEL_KEYS: + if isinstance(value, dict): + for subkey in MODEL_SECTION_KEYS: + subvalue = value.get(subkey) + if isinstance(subvalue, str): + yield (f"{child}.{subkey}", subvalue) + for subkey, subvalue in value.items(): + if isinstance(subvalue, (dict, list)): + yield from _iter_model_values(subvalue, f"{child}.{subkey}") + elif isinstance(value, list): + yield from _iter_model_values(value, child) + else: + yield (child, value) + elif key in MODEL_LIST_KEYS: + yield from _iter_model_list(value, child) + elif isinstance(value, (dict, list)): + yield from _iter_model_values(value, child) + elif isinstance(node, list): + for i, item in enumerate(node): + yield from _iter_model_values(item, f"{path}[{i}]") + + +def _iter_model_list(node, path): + """Yield scalars under an allowlisted `models` key (list of names or list of dicts).""" + if isinstance(node, list): + for i, item in enumerate(node): + yield from _iter_model_list(item, f"{path}[{i}]") + elif isinstance(node, dict): + for key, value in node.items(): + if key in MODEL_KEYS and isinstance(value, str): + yield (f"{path}.{key}", value) + elif isinstance(value, (dict, list)): + yield from _iter_model_list(value, f"{path}.{key}") + else: + yield (path, node) def audit(path): @@ -217,7 +233,7 @@ def audit(path): ) # --- No raw or retired model names (Rule 7/8 spirit) --- - # Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets + # Any retired/raw name found by the derivation above is a hard failure: such a config gets # 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ # crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. bad_model_names = { @@ -228,7 +244,7 @@ def audit(path): "qwen3.6-35B-udq4": "strix-moe", "ornith-1.0-35b": "strix-moe", } - for field_path, value in _model_name_fields(cfg): + for field_path, value in _iter_model_values(cfg): if value in bad_model_names: check( False, diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index b712c0e..96b76d3 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -169,7 +169,7 @@ The agent picks up the new key via `infisical run --` at gateway startup. **Keys are permanent and use bare agent name aliases.** -- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`. +- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.) - **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity. - **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven. - **Max budget**: $100 per key (config default). diff --git a/litellm-client-timeouts.prose.md b/litellm-client-timeouts.prose.md index a7d5aca..7987ac7 100644 --- a/litellm-client-timeouts.prose.md +++ b/litellm-client-timeouts.prose.md @@ -28,7 +28,7 @@ description: > | syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load | | qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto | | strix-moe | 7.5s | — | Strix Halo, healthy | -| gpu-vision | 2.6s | — | RTX 5070, healthy | +| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy | Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged backend), full recovery 07:00-08:00 with ZERO client failures once requests @@ -53,8 +53,7 @@ proxy queuing. ### 2. Auxiliary tasks — keep template timeouts, one correction -- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s; - these are fine. +- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`. - compression: 300s (keep — this was already raised from 60 per gpu-fleet). - **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090 (qwen3.6-27B-code backend, 23.0s avg) is the same speed class as diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index b489d67..f4db4d9 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -143,3 +143,46 @@ def test_raw_alias_is_rejected(tmp_path): assert "model.default = 'qwen3.6-27B-code'" in out assert "gpu-dense" in out assert "RESULT: FAIL" in out + + +def test_retired_alias_in_fallback_providers_is_rejected(tmp_path): + """fallback_providers.model is model-bearing; a retired name there must fail.""" + code, out = _run_config( + tmp_path, + "fallback-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"), + ) + assert code == 1, out + assert "fallback_providers.model = 'gpu-light'" in out + assert "RESULT: FAIL" in out + + +def test_retired_alias_in_x_search_is_rejected(tmp_path): + """x_search.model was previously not enumerated; the derivation must catch it.""" + code, out = _run_config( + tmp_path, + "x-search-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace( + "delegation:\n provider: harness", + "delegation:\n provider: harness\nx_search:\n model: gpu-light", + ), + ) + assert code == 1, out + assert "x_search.model = 'gpu-light'" in out + assert "RESULT: FAIL" in out + + +def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path): + """A nested auxiliary sub-block outside the named three must still be derived.""" + code, out = _run_config( + tmp_path, + "nested-aux-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace( + " compression:\n model: syslog-auto\n provider: harness\ndelegation:", + " compression:\n model: syslog-auto\n provider: harness\n" + " tasks:\n summarize:\n model: gpu-light\ndelegation:", + ), + ) + assert code == 1, out + assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out + assert "RESULT: FAIL" in out From 3d55764799658a761c9094971727f8e9ee496afe Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:41:35 +0000 Subject: [PATCH 5/8] no-mistakes(review): Split retired-alias audit into fail vs warn; fix key claim --- audit-hermes-config.py | 31 +++++++++++++++++-------- litellm-api-keys.prose.md | 2 +- tests/test_audit_hermes_config_alias.py | 17 ++++++++------ 3 files changed, 32 insertions(+), 18 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index 561dc99..faf6f79 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -232,24 +232,35 @@ def audit(path): f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", ) - # --- No raw or retired model names (Rule 7/8 spirit) --- - # Any retired/raw name found by the derivation above is a hard failure: such a config gets - # 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ - # crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. - bad_model_names = { - "gemma-4-12b": "gpu-vision", + # --- Retired/raw model names (Rule 7/8 spirit) --- + # The audit's job is to catch configs that are BROKEN, not to enforce a style preference. + # NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL: + # gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision + # crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe + # RESOLVING names (verified 200) are discouraged but working, so they only WARN: + # qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe + # Failing a working alias would reject valid configs - the exact defect this change fixes. + non_resolving = { "gpu-light": "gpu-vision", + "gemma-4-12b": "gpu-vision", "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", - "qwen3.6-27B-code": "gpu-dense", - "qwen3.6-35B-udq4": "strix-moe", "ornith-1.0-35b": "strix-moe", } + raw_but_live = { + "qwen3.6-27B-code": "gpu-dense", + "qwen3.6-35B-udq4": "strix-moe", + } for field_path, value in _iter_model_values(cfg): - if value in bad_model_names: + if value in non_resolving: check( False, "Rule 7/8", - f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}", + f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}", + ) + elif value in raw_but_live: + warn( + "Rule 7/8", + f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}", ) # --- Report --- diff --git a/litellm-api-keys.prose.md b/litellm-api-keys.prose.md index 27b8e94..1c1e2df 100644 --- a/litellm-api-keys.prose.md +++ b/litellm-api-keys.prose.md @@ -67,7 +67,7 @@ description: > 4. **If action == "create"**: - Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date) - Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" } - - Duration is null (permanent) — inherited from litellm default_key_generate_params + - Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.) - Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped, and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12). diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index f4db4d9..a129f03 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -132,17 +132,20 @@ def test_retired_alias_in_custom_providers_is_rejected(tmp_path): assert "RESULT: FAIL" in out -def test_raw_alias_is_rejected(tmp_path): - """Raw-but-live model names must fail too, pointing at the stable alias.""" +def test_raw_but_live_alias_warns_but_passes(tmp_path): + """Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs.""" code, out = _run_config( tmp_path, "raw-qwen.yaml", - BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"), + BASE.format(alias="gpu-vision").replace( + "delegation:\n provider: harness", + "delegation:\n provider: harness\n model: qwen3.6-27B-code", + ), ) - assert code == 1, out - assert "model.default = 'qwen3.6-27B-code'" in out - assert "gpu-dense" in out - assert "RESULT: FAIL" in out + assert code == 0, out + assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out + assert "prefer the stable alias gpu-dense" in out + assert "RESULT: PASS" in out def test_retired_alias_in_fallback_providers_is_rejected(tmp_path): From 1f1b47f59df2da7a6d89185271edf29ebf6dd5a1 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:51:21 +0000 Subject: [PATCH 6/8] no-mistakes(document): Sweep residual gemma aliases; align compression rule contradiction --- gpu-self-heal.prose.md | 2 +- hermes-config-template.prose.md | 6 +++--- inference-optimization.prose.md | 6 +++--- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/gpu-self-heal.prose.md b/gpu-self-heal.prose.md index cdc6264..32019ca 100644 --- a/gpu-self-heal.prose.md +++ b/gpu-self-heal.prose.md @@ -64,7 +64,7 @@ Key notes: - **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls - **Fix**: 1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control) - 2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gemma → qwen, qwen → gemma) + 2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gpu-vision → gpu-dense, gpu-dense → gpu-vision) 3. If all GPUs hot, alert about cooling infrastructure - **Verify**: Temp drops below 80°C within 5 minutes - **Escalate after**: 3 verification failures → Zulip alert diff --git a/hermes-config-template.prose.md b/hermes-config-template.prose.md index 485f2ee..ea700c8 100644 --- a/hermes-config-template.prose.md +++ b/hermes-config-template.prose.md @@ -253,8 +253,8 @@ The following MUST be identical across ALL profiles: ### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16) - Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized) -- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized) -- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b` +- Compression uses `syslog-auto` — the Strix Halo weighted pool (64GB, 128K ctx, compression-optimized); do NOT pin `compression.model` to `strix-moe` (audit Rule 7 rejects it) +- **`ornith-1.0-35b` is NOT a valid compression model name** — LiteLLM does not serve it (do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml` is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls. - **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.** @@ -265,7 +265,7 @@ The following MUST be identical across ALL profiles: - All auxiliary services MUST use identical routing: - `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK) - `api_key_env: LITELLM_API_KEY` -- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably +- **Do NOT use `syslog-auto` for `vision`/`web_extract`** — it routes unpredictably; compression is the deliberate exception (see the OPERATIONAL DECISION above) - **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo (64GB UMA, 128K context) — the designated compression GPU. This frees the RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning. diff --git a/inference-optimization.prose.md b/inference-optimization.prose.md index beb18a2..69a0d96 100644 --- a/inference-optimization.prose.md +++ b/inference-optimization.prose.md @@ -23,7 +23,7 @@ management, and prompt caching — without sacrificing agent capability. .123, any others on .129/.122) including compression, model, context_window, prompt_caching, memory settings - `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080, - qwen .8:8080, gemma .110:8080) + gpu-dense .8:8080, gpu-vision .110:8080) ### Maintains @@ -56,8 +56,8 @@ duration. **Context is the root cause.** Every ~46K prompt token costs ~87s of prefill time at 532 tok/s. Fix context first, routing second. -- **Route by task**: qwen for code/standard queries; gemma for - compression/auxiliary; strix-moe for compression tasks. +- **Route by task**: gpu-dense for code/standard queries; gpu-vision for + vision/web-auxiliary; syslog-auto for compression. - **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should compact at 51K, not 85K. Target 15% tail (not 30%). - **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these From 78b501798f181fc8172dd5a2fe82d6d6c6251e82 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:54:41 +0000 Subject: [PATCH 7/8] no-mistakes(document): Sweep residual gemma labels; align compression rule contradiction --- gpu-fleet.prose.md | 14 +++++++------- gpu-monitor.prose.md | 4 ++-- infrastructure-control.prose.md | 2 +- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index 1bef170..3d4d13e 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -239,7 +239,7 @@ History stored at `/root/data/toks-history.json` with 7-day rolling window. ### Stable Aliases — CRITICAL All agent configs MUST use stable role-based aliases, never model-specific names: -- `compression.model: strix-moe` +- `compression.model: syslog-auto` - `auxiliary.vision.model: gpu-vision` - `delegation.model: gpu-dense` - `auxiliary.web_extract.model: gpu-vision` @@ -249,25 +249,25 @@ When the underlying model is swapped, only the LiteLLM config changes — agent ### Context Windows - RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K** - **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek) -- Compression threshold 0.60: fires at ~77K (~51K headroom before 128K ceiling) +- Compression threshold 0.65 (audit Rule 9): fires at ~85K (~43K headroom before 128K ceiling) - **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K) -- Mumuni compression model alias: `strix-moe` +- Mumuni compression model alias: `syslog-auto` ### Mumuni Agent Profile -Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs: +Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs. The compression values below are the current required values per template Rules 7/9 and `audit-hermes-config.py`; whether Mumuni's LIVE config currently complies is a separate operational question. | Setting | Value | Notes | |---------|-------|-------| | `model.default` | `syslog-auto` | Balanced default (pool router) | | `model.provider` | `custom:litellm` | LiteLLM on CT116 | -| `compression.model` | `strix-moe` | Stable alias — survives model swaps | -| `aux.compression.model` | `strix-moe` | Compression auxiliary model | +| `compression.model` | `syslog-auto` | Rule 7: auto-routing, prevents Strix Halo overload | +| `aux.compression.model` | `syslog-auto` | Must match `compression.model` (Rule 7) | | `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) | | `aux.web_extract.model` | `gpu-vision` | Web extraction | | `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) | | `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling | -| `compression.threshold` | 0.60 | Triggers at ~77K (~60% of 128K) — optimized for 128K context | +| `compression.threshold` | 0.65 | Rule 9: triggers at ~85K for a 128K window | | `compression.target_ratio` | 0.3 | Compresses to ~38K | | `compression.protect_last_n` | 40 | Preserves last 40 messages | | `memory.memory_char_limit` | 800 | Brief memory entries | diff --git a/gpu-monitor.prose.md b/gpu-monitor.prose.md index a6808e1..72bad0d 100644 --- a/gpu-monitor.prose.md +++ b/gpu-monitor.prose.md @@ -27,8 +27,8 @@ agent: abiba ┌──────┐ ┌──────┐ ┌────────┐ │.8:8080│ │.110 │ │.116:80 │ │RTX3090│ │:8080 │ │nginx │ -│gemma │ │RTX5070│ │router │ -└──────┘ │qwen27B│ │LiteLLM │ +│qwen │ │RTX5070│ │router │ +└──────┘ │vision │ │LiteLLM │ └──────┘ │dashboard│ └────────┘ ``` diff --git a/infrastructure-control.prose.md b/infrastructure-control.prose.md index b9b682b..d7a4a36 100644 --- a/infrastructure-control.prose.md +++ b/infrastructure-control.prose.md @@ -222,7 +222,7 @@ description: > **Prometheus targets**: - 192.168.68.8:9400 (RTX 3090 — qwen) -- 192.168.68.110:9400 (RTX 5070 — gemma) +- 192.168.68.110:9400 (RTX 5070 — gpu-vision) - 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4) - harness-litellm:4000 (LiteLLM health) From a2edc2f56f1ff510c43b680ee5455ea5eeeceb93 Mon Sep 17 00:00:00 2001 From: abiba Date: Sat, 12 Sep 2026 17:13:54 +0000 Subject: [PATCH 8/8] ci(pr-pipeline): fix flaky frontmatter check (grep -q SIGPIPE under pipefail) The validate job runs with bash -e -o pipefail. `echo "$FM" | grep -q '^name:'` lets grep exit on first match, which can SIGPIPE the echo; pipefail then reports the pipeline non-zero and the || branch raises a false "Missing name/description". The flagged file set varied run to run (and included files untouched by the PR) while a fresh clone of the same commit passes the identical check. Reproduced: the old form failed 3 of 5 local runs under the same shell flags, the herestring form passed 5 of 5. Use herestrings so no pipe can be broken. --- .gitea/workflows/pr-pipeline.yaml | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/.gitea/workflows/pr-pipeline.yaml b/.gitea/workflows/pr-pipeline.yaml index 7525b10..9ffe552 100644 --- a/.gitea/workflows/pr-pipeline.yaml +++ b/.gitea/workflows/pr-pipeline.yaml @@ -43,17 +43,21 @@ jobs: echo "=== Prose Contract Frontmatter Validation ===" FAILED=0 for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do + # NOTE: use herestrings, not `echo "$FM" | grep ...`. Under the runner's + # `-e -o pipefail`, `grep -q` exits on first match and can SIGPIPE the + # producer, making the pipeline report non-zero and raising a false + # "Missing name/description" whose file set varies run to run. FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d') [ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; } - KIND=$(echo "$FM" | grep '^kind:' | awk '{print $2}') + KIND=$(grep '^kind:' <<< "$FM" | awk '{print $2}') case "$KIND" in function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;; *) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;; esac - echo "$FM" | grep -q '^name:' || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); } - echo "$FM" | grep -q '^description:' || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); } + grep -q '^name:' <<< "$FM" || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); } + grep -q '^description:' <<< "$FM" || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); } done [ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; } echo "✅ Frontmatter validation passed"