From 221f9f79f3afec7e5be5cf9f4f34c4b87f8b6126 Mon Sep 17 00:00:00 2001 From: abiba Date: Sat, 12 Sep 2026 16:11:46 +0000 Subject: [PATCH] fix(audit): reject retired aliases; sweep gpu-light/gemma-4-12b to gpu-vision audit-hermes-config.py Rule 8 required auxiliary.vision.model and auxiliary.web_extract.model to equal the retired 'gpu-light', so a config adopting the live canonical 'gpu-vision' FAILED our own audit - the audit was enforcing a dead alias (400 Invalid model name). Rule 8 now requires gpu-vision; retired names gpu-light/crew-auto join the raw-name rejection set; the guidance message names the live aliases. Sweep of the remaining references: gpu-self-heal stops canonicalizing gpu-light; hermes-config-template, hermes-agent-baseline, hermes-key-enforcement, inference-optimization, litellm-client-timeouts and gpu-fleet now use the live gpu-vision alias. Where a file restated model/rpm/weight/fallback state it now points at CT 116 /opt/inference-harness/litellm_config.yaml instead of duplicating it. koby's .129 config is report-only and recorded, not edited. Adds tests/test_audit_hermes_config_alias.py: executes the audit CLI and asserts gpu-vision passes while gpu-light and gemma-4-12b fail. --- audit-hermes-config.py | 20 +++--- gpu-fleet.prose.md | 2 +- gpu-self-heal.prose.md | 17 ++--- hermes-agent-baseline.prose.md | 9 ++- hermes-config-template.prose.md | 29 ++++---- hermes-key-enforcement.prose.md | 6 +- inference-optimization.prose.md | 2 +- litellm-api-keys.prose.md | 4 +- litellm-client-timeouts.prose.md | 4 +- tests/test_audit_hermes_config_alias.py | 89 +++++++++++++++++++++++++ 10 files changed, 139 insertions(+), 43 deletions(-) create mode 100644 tests/test_audit_hermes_config_alias.py diff --git a/audit-hermes-config.py b/audit-hermes-config.py index 773fff2..b0f3323 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -89,15 +89,16 @@ def audit(path): ) # --- Rule 8: GPU Workload Distribution --- + # gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision. check( - aux.get("vision", {}).get("model") == "gpu-light", + aux.get("vision", {}).get("model") == "gpu-vision", "Rule 8", - f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias", + f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias", ) check( - aux.get("web_extract", {}).get("model") == "gpu-light", + aux.get("web_extract", {}).get("model") == "gpu-vision", "Rule 8", - f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias", + f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias", ) # --- Rule 9: Compression Threshold --- @@ -178,8 +179,11 @@ def audit(path): f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", ) - # --- No raw model names (Rule 7/8 spirit) --- - raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} + # --- No raw or retired model names (Rule 7/8 spirit) --- + # Retired names are rejected by the alias rules above; a config that names them directly is + # flagged here too. gpu-light/gemma-4-12b/crew-auto were retired 2026-09-12. + raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b", + "gpu-light", "crew-auto"} for section_path, section_dict in [ ("model", model), ("compression", comp), ("auxiliary.vision", aux.get("vision", {})), @@ -191,8 +195,8 @@ def audit(path): if m in raw_names: warn( "Rule 7/8", - f"{section_path}.model = {m!r} — raw model name, use stable alias instead " - f"(gpu-light, gpu-dense, strix-moe, syslog-auto)", + f"{section_path}.model = {m!r} — raw or retired model name, use a live stable alias " + f"instead (gpu-vision, gpu-dense, strix-moe, syslog-auto)", ) # --- Report --- diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index fefebba..1bef170 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -216,7 +216,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. - **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled. - **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds. - **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down. -- **Alias-retirement follow-up (2026-09-12)**: the agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, and the executable `audit-hermes-config.py` still reference the retired `gpu-light`/`gemma-4-12b` names, and `audit-hermes-config.py` Rule 8 currently fails a config whose vision/web_extract model is `gpu-vision`. These are tracked separately and are intentionally NOT updated in this change. +- **Alias-retirement sweep (2026-09-12)**: `gemma-4-12b`, `gpu-light` and `crew-auto` are retired and replaced by `gpu-vision` / no cap respectively. The agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, `gpu-self-heal.prose.md`, `hermes-key-enforcement.prose.md`, `inference-optimization.prose.md`, `litellm-client-timeouts.prose.md` and the executable `audit-hermes-config.py` were all updated to the live canonical alias in the same change. **koby's config on .129 still names `gpu-light` (and `gemma-4-E4B`); .129 is report-only, so that is recorded for its owner and NOT edited here.** ## GPU Inference Benchmarks (Current) diff --git a/gpu-self-heal.prose.md b/gpu-self-heal.prose.md index 04ef8e4..cdc6264 100644 --- a/gpu-self-heal.prose.md +++ b/gpu-self-heal.prose.md @@ -12,7 +12,7 @@ description: > Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing. Benchmark baselines refreshed to live values. Prometheus exporters removed — not deployed; fall back to direct sidecar probes. - Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet. + Stable role-based aliases (strix-moe, gpu-dense, gpu-vision) from gpu-fleet. agent: abiba depends_on: - gpu-monitor.prose.md (live data source on .24:9100) @@ -52,8 +52,8 @@ depends_on: Key notes: - All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path. -- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. -- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was. +- Stable aliases (gpu-dense, gpu-vision, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. The retired names `gpu-light` and `gemma-4-12b` were superseded by `gpu-vision` on 2026-09-12 and no longer resolve (400 `Invalid model name`). +- The RTX 5070 is the fastest endpoint per token — `gpu-vision` is its canonical alias. Route vision/web/light work there first. The RTX 5070 model is multimodal (image+text). - Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads. - RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom). - RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes). @@ -157,13 +157,14 @@ Key notes: ### Rule 10: Workload Distribution Optimization (updated 2026-07-18) - **Detect**: GPU roles misaligned with hardware capabilities - **Target distribution**: - - RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM). - - RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM). - - Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM). + - RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). + - RTX 5070 (gpu-vision, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. + - Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). - **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first. +- **Weights are not restated here** — the live `syslog-auto` pool weights and rpm caps live in CT 116 `/opt/inference-harness/litellm_config.yaml`, the single source of truth. - **Fix**: - Alert if any GPU is handling workload outside its designated role - - Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe) + - Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-vision, strix-moe) - Track per-GPU request distribution via LiteLLM spend logs - **Verify**: Each GPU's request pattern matches its designated role within 24h - **Escalate**: If role mismatch persists >48h → agent alias audit needed @@ -311,6 +312,6 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab If monitor response > 1MB, log a warning and skip the cycle rather than crashing. ### L6: Stable Aliases Replace Model Names -- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15. +- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15; `gpu-light` was superseded by `gpu-vision` on 2026-09-12. - Self-heal must use aliases for reporting and alerting, not model-specific names. - **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier. diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index 5fcd5f1..c9513fc 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -80,7 +80,7 @@ custom_providers: auxiliary: vision: provider: harness - model: gemma-4-12b # or syslog-auto + model: gpu-vision # RTX 5070 stable alias (or syslog-auto) base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround @@ -95,7 +95,7 @@ auxiliary: threshold: 0.65 target_ratio: 0.3 provider: harness - model: syslog-auto # or gemma-4-12b + model: syslog-auto # or gpu-vision base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround @@ -197,9 +197,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup: { "id": "syslog-auto" }, { "id": "strix-moe" }, { "id": "gpu-dense" }, - { "id": "gpu-light" }, - { "id": "qwen3.6-27B-code" }, - { "id": "gemma-4-12b" } + { "id": "gpu-vision" }, + { "id": "qwen3.6-27B-code" } ] } } diff --git a/hermes-config-template.prose.md b/hermes-config-template.prose.md index 076ff04..485f2ee 100644 --- a/hermes-config-template.prose.md +++ b/hermes-config-template.prose.md @@ -102,7 +102,7 @@ model: # and falls back to 256K when /v1/models lacks a context # field (llama-server does). Without this override, agents # silently run syslog-auto at 256K (verified 2026-08-09). - # Set 65536 if using gemma-4-12b directly (tight VRAM). + # Set 65536 if pinning a single model directly (tight VRAM). fallback_providers: provider: deepseek @@ -143,25 +143,25 @@ compression: # ─── Auxiliary Tasks (CONSISTENCY RULE) ─── # All auxiliary services MUST use identical model, base_url, and api_key_env: -# model: gpu-light # stable alias (NOT raw "gemma-4-12b") +# model: gpu-vision # stable alias (NOT a raw model name) # base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK # api_key_env: LITELLM_API_KEY # Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU. -# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning. +# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning. # Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead. -# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4) -# in agent configs — use the stable aliases so model swaps don't break agents. +# NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired +# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents. auxiliary: vision: provider: harness - model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b) + model: gpu-vision # stable alias for RTX 5070 base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK api_key_env: LITELLM_API_KEY timeout: 60 download_timeout: 30 web_extract: provider: harness - model: gpu-light # stable alias for RTX 5070 + model: gpu-vision # stable alias for RTX 5070 base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK api_key_env: LITELLM_API_KEY timeout: 30 @@ -252,10 +252,11 @@ The following MUST be identical across ALL profiles: - For agents needing longer outputs: raise to 8192, but never omit ### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16) -- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized) +- Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized) - Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized) - **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b` - (it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls. + (do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml` + is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls. - **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.** The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can @@ -273,11 +274,11 @@ The following MUST be identical across ALL profiles: ### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16) - **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations -- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K) +- **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K) - **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs - Agent profiles MUST route auxiliary tasks to the correct GPU: - - `auxiliary.vision.model: gemma-4-12b` (RTX 5070) - - `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070) + - `auxiliary.vision.model: gpu-vision` (RTX 5070) + - `auxiliary.web_extract.model: gpu-vision` (RTX 5070) - `auxiliary.compression.model: syslog-auto` (Strix Halo) - Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing - For 128K context window: `threshold: 0.65` (fires at ~85K tokens) @@ -297,8 +298,8 @@ The following MUST be identical across ALL profiles: - **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10. - **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto` - **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json` -- `syslog-auto` is the LiteLLM routing model — it load-balances between strix-moe - and qwen3.6-27B-code, with gemma-4-12b as fallback. Using it protects against: +- `syslog-auto` is the LiteLLM routing model — it load-balances across the live pool + (see CT 116 `/opt/inference-harness/litellm_config.yaml` for the current members and weights). Using it protects against: - Model name typos that cause 403 errors and silent worker failures - Single GPU downtime (routing falls back automatically) - Key/model authorization mismatches diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index 5e46736..45d0296 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -178,7 +178,7 @@ The agent picks up the new key via `infisical run --` at gateway startup. # In litellm_config.yaml — ensures all future keys inherit these defaults: litellm_settings: default_key_generate_params: - models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b"] + models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] duration: null # ← permanent max_budget: 100 metadata: @@ -286,13 +286,13 @@ auxiliary: api_key: sk- # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain) api_key_env: LITELLM_API_KEY base_url: http://192.168.68.116/litellm/v1 - model: gemma-4-12b + model: gpu-vision provider: harness compression: api_key: sk- # ← workaround (same as above) api_key_env: LITELLM_API_KEY base_url: http://192.168.68.116/litellm/v1 - model: gemma-4-12b + model: gpu-vision provider: harness ``` diff --git a/inference-optimization.prose.md b/inference-optimization.prose.md index fd28d22..bd5dbb2 100644 --- a/inference-optimization.prose.md +++ b/inference-optimization.prose.md @@ -99,5 +99,5 @@ call enable-prompt-caching call verify-latency host: 192.168.68.116 - models: [syslog-auto, qwen3.6-27B-code, gemma-4-12b] + models: [syslog-auto, qwen3.6-27B-code, gpu-vision] ``` diff --git a/litellm-api-keys.prose.md b/litellm-api-keys.prose.md index c42897d..27b8e94 100644 --- a/litellm-api-keys.prose.md +++ b/litellm-api-keys.prose.md @@ -68,7 +68,9 @@ description: > - Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date) - Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" } - Duration is null (permanent) — inherited from litellm default_key_generate_params - - Set models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b", "strix-moe", "gpu-dense", "gpu-light", "qwen3.6-35B-udq4"] + - Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped, + and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add + retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12). - Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed). - Return the new key 5. **If action == "rotate"**: diff --git a/litellm-client-timeouts.prose.md b/litellm-client-timeouts.prose.md index 4e7048f..a7d5aca 100644 --- a/litellm-client-timeouts.prose.md +++ b/litellm-client-timeouts.prose.md @@ -28,7 +28,7 @@ description: > | syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load | | qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto | | strix-moe | 7.5s | — | Strix Halo, healthy | -| gemma-4-12b | 2.6s | — | RTX 5070, healthy | +| gpu-vision | 2.6s | — | RTX 5070, healthy | Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged backend), full recovery 07:00-08:00 with ZERO client failures once requests @@ -53,7 +53,7 @@ proxy queuing. ### 2. Auxiliary tasks — keep template timeouts, one correction -- vision: 60s (keep), web_extract: 30s (keep) — gemma-4-12b averages 2.6s; +- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s; these are fine. - compression: 300s (keep — this was already raised from 60 per gpu-fleet). - **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090 diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py new file mode 100644 index 0000000..4416318 --- /dev/null +++ b/tests/test_audit_hermes_config_alias.py @@ -0,0 +1,89 @@ +"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py. + +WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias. +Rule 8 required `auxiliary.vision.model == "gpu-light"` and +`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor +`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the +live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias +therefore FAILED our own audit, so the audit was actively enforcing a broken config. + +These tests execute the real CLI (`python3 audit-hermes-config.py `) and assert +observable behaviour — exit code and the emitted rule message — for the live alias and +for both retired names. No network, vault, or SSH access is required. +""" +from __future__ import annotations + +import pathlib +import subprocess +import sys + +ROOT = pathlib.Path(__file__).resolve().parent.parent +AUDIT = ROOT / "audit-hermes-config.py" + +BASE = """ +model: + api_key: "" + api_key_env: LITELLM_API_KEY + base_url: http://192.168.68.116/v1 + max_tokens: 4096 + default: syslog-auto + provider: harness +fallback_providers: + provider: deepseek + model: deepseek-v4-flash + api_key_env: DEEPSEEK_API_KEY +compression: + model: syslog-auto + provider: harness + threshold: 0.65 + max_context_window: 131072 +auxiliary: + vision: + model: {alias} + provider: harness + web_extract: + model: {alias} + provider: harness + compression: + model: syslog-auto + provider: harness +delegation: + provider: harness +custom_providers: + - name: harness + key_env: LITELLM_API_KEY + base_url: http://192.168.68.116/v1 +""" + + +def _run(tmp_path, alias): + cfg = tmp_path / f"{alias}.yaml" + cfg.write_text(BASE.format(alias=alias)) + proc = subprocess.run( + [sys.executable, str(AUDIT), str(cfg)], + capture_output=True, text=True, + ) + return proc.returncode, proc.stdout + + +def test_live_canonical_alias_passes(tmp_path): + """The RTX 5070 alias that actually resolves must satisfy Rule 8.""" + code, out = _run(tmp_path, "gpu-vision") + assert code == 0, out + assert "RESULT: PASS" in out + + +def test_retired_gpu_light_is_rejected(tmp_path): + """A config pinned to the retired alias must fail, not pass.""" + code, out = _run(tmp_path, "gpu-light") + assert code == 1, out + assert "auxiliary.vision.model must be gpu-vision" in out + assert "RESULT: FAIL" in out + + +def test_retired_gemma_is_rejected(tmp_path): + """The retired raw model name must fail Rule 8 as well.""" + code, out = _run(tmp_path, "gemma-4-12b") + assert code == 1, out + assert "auxiliary.vision.model must be gpu-vision" in out + assert "RESULT: FAIL" in out