no-mistakes(review): Split retired-alias audit into fail vs warn; fix key claim

This commit is contained in:
root
2026-09-12 16:41:35 +00:00
parent ce48070f21
commit 3d55764799
3 changed files with 32 additions and 18 deletions
+21 -10
View File
@@ -232,24 +232,35 @@ def audit(path):
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
) )
# --- No raw or retired model names (Rule 7/8 spirit) --- # --- Retired/raw model names (Rule 7/8 spirit) ---
# Any retired/raw name found by the derivation above is a hard failure: such a config gets # The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ # NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. # gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
bad_model_names = { # crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
"gemma-4-12b": "gpu-vision", # RESOLVING names (verified 200) are discouraged but working, so they only WARN:
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
# Failing a working alias would reject valid configs - the exact defect this change fixes.
non_resolving = {
"gpu-light": "gpu-vision", "gpu-light": "gpu-vision",
"gemma-4-12b": "gpu-vision",
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
"qwen3.6-27B-code": "gpu-dense",
"qwen3.6-35B-udq4": "strix-moe",
"ornith-1.0-35b": "strix-moe", "ornith-1.0-35b": "strix-moe",
} }
raw_but_live = {
"qwen3.6-27B-code": "gpu-dense",
"qwen3.6-35B-udq4": "strix-moe",
}
for field_path, value in _iter_model_values(cfg): for field_path, value in _iter_model_values(cfg):
if value in bad_model_names: if value in non_resolving:
check( check(
False, False,
"Rule 7/8", "Rule 7/8",
f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}", f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
)
elif value in raw_but_live:
warn(
"Rule 7/8",
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
) )
# --- Report --- # --- Report ---
+1 -1
View File
@@ -67,7 +67,7 @@ description: >
4. **If action == "create"**: 4. **If action == "create"**:
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date) - Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" } - Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
- Duration is null (permanent) — inherited from litellm default_key_generate_params - Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
- Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped, - Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12). retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
+10 -7
View File
@@ -132,17 +132,20 @@ def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
assert "RESULT: FAIL" in out assert "RESULT: FAIL" in out
def test_raw_alias_is_rejected(tmp_path): def test_raw_but_live_alias_warns_but_passes(tmp_path):
"""Raw-but-live model names must fail too, pointing at the stable alias.""" """Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs."""
code, out = _run_config( code, out = _run_config(
tmp_path, tmp_path,
"raw-qwen.yaml", "raw-qwen.yaml",
BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"), BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
),
) )
assert code == 1, out assert code == 0, out
assert "model.default = 'qwen3.6-27B-code'" in out assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out
assert "gpu-dense" in out assert "prefer the stable alias gpu-dense" in out
assert "RESULT: FAIL" in out assert "RESULT: PASS" in out
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path): def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):