diff --git a/audit-hermes-config.py b/audit-hermes-config.py index 561dc99..faf6f79 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -232,24 +232,35 @@ def audit(path): f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", ) - # --- No raw or retired model names (Rule 7/8 spirit) --- - # Any retired/raw name found by the derivation above is a hard failure: such a config gets - # 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ - # crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. - bad_model_names = { - "gemma-4-12b": "gpu-vision", + # --- Retired/raw model names (Rule 7/8 spirit) --- + # The audit's job is to catch configs that are BROKEN, not to enforce a style preference. + # NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL: + # gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision + # crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe + # RESOLVING names (verified 200) are discouraged but working, so they only WARN: + # qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe + # Failing a working alias would reject valid configs - the exact defect this change fixes. + non_resolving = { "gpu-light": "gpu-vision", + "gemma-4-12b": "gpu-vision", "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", - "qwen3.6-27B-code": "gpu-dense", - "qwen3.6-35B-udq4": "strix-moe", "ornith-1.0-35b": "strix-moe", } + raw_but_live = { + "qwen3.6-27B-code": "gpu-dense", + "qwen3.6-35B-udq4": "strix-moe", + } for field_path, value in _iter_model_values(cfg): - if value in bad_model_names: + if value in non_resolving: check( False, "Rule 7/8", - f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}", + f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}", + ) + elif value in raw_but_live: + warn( + "Rule 7/8", + f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}", ) # --- Report --- diff --git a/litellm-api-keys.prose.md b/litellm-api-keys.prose.md index 27b8e94..1c1e2df 100644 --- a/litellm-api-keys.prose.md +++ b/litellm-api-keys.prose.md @@ -67,7 +67,7 @@ description: > 4. **If action == "create"**: - Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date) - Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" } - - Duration is null (permanent) — inherited from litellm default_key_generate_params + - Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.) - Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped, and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12). diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index f4db4d9..a129f03 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -132,17 +132,20 @@ def test_retired_alias_in_custom_providers_is_rejected(tmp_path): assert "RESULT: FAIL" in out -def test_raw_alias_is_rejected(tmp_path): - """Raw-but-live model names must fail too, pointing at the stable alias.""" +def test_raw_but_live_alias_warns_but_passes(tmp_path): + """Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs.""" code, out = _run_config( tmp_path, "raw-qwen.yaml", - BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"), + BASE.format(alias="gpu-vision").replace( + "delegation:\n provider: harness", + "delegation:\n provider: harness\n model: qwen3.6-27B-code", + ), ) - assert code == 1, out - assert "model.default = 'qwen3.6-27B-code'" in out - assert "gpu-dense" in out - assert "RESULT: FAIL" in out + assert code == 0, out + assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out + assert "prefer the stable alias gpu-dense" in out + assert "RESULT: PASS" in out def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):