no-mistakes(review): Split retired-alias audit into fail vs warn; fix key claim
This commit is contained in:
+21
-10
@@ -232,24 +232,35 @@ def audit(path):
|
|||||||
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- No raw or retired model names (Rule 7/8 spirit) ---
|
# --- Retired/raw model names (Rule 7/8 spirit) ---
|
||||||
# Any retired/raw name found by the derivation above is a hard failure: such a config gets
|
# The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
|
||||||
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/
|
# NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
|
||||||
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead.
|
# gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
|
||||||
bad_model_names = {
|
# crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
|
||||||
"gemma-4-12b": "gpu-vision",
|
# RESOLVING names (verified 200) are discouraged but working, so they only WARN:
|
||||||
|
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
|
||||||
|
# Failing a working alias would reject valid configs - the exact defect this change fixes.
|
||||||
|
non_resolving = {
|
||||||
"gpu-light": "gpu-vision",
|
"gpu-light": "gpu-vision",
|
||||||
|
"gemma-4-12b": "gpu-vision",
|
||||||
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
|
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
|
||||||
"qwen3.6-27B-code": "gpu-dense",
|
|
||||||
"qwen3.6-35B-udq4": "strix-moe",
|
|
||||||
"ornith-1.0-35b": "strix-moe",
|
"ornith-1.0-35b": "strix-moe",
|
||||||
}
|
}
|
||||||
|
raw_but_live = {
|
||||||
|
"qwen3.6-27B-code": "gpu-dense",
|
||||||
|
"qwen3.6-35B-udq4": "strix-moe",
|
||||||
|
}
|
||||||
for field_path, value in _iter_model_values(cfg):
|
for field_path, value in _iter_model_values(cfg):
|
||||||
if value in bad_model_names:
|
if value in non_resolving:
|
||||||
check(
|
check(
|
||||||
False,
|
False,
|
||||||
"Rule 7/8",
|
"Rule 7/8",
|
||||||
f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}",
|
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
|
||||||
|
)
|
||||||
|
elif value in raw_but_live:
|
||||||
|
warn(
|
||||||
|
"Rule 7/8",
|
||||||
|
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- Report ---
|
# --- Report ---
|
||||||
|
|||||||
@@ -67,7 +67,7 @@ description: >
|
|||||||
4. **If action == "create"**:
|
4. **If action == "create"**:
|
||||||
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
|
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
|
||||||
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
|
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
|
||||||
- Duration is null (permanent) — inherited from litellm default_key_generate_params
|
- Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||||
- Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
|
- Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
|
||||||
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
|
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
|
||||||
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
|
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
|
||||||
|
|||||||
@@ -132,17 +132,20 @@ def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
|
|||||||
assert "RESULT: FAIL" in out
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
def test_raw_alias_is_rejected(tmp_path):
|
def test_raw_but_live_alias_warns_but_passes(tmp_path):
|
||||||
"""Raw-but-live model names must fail too, pointing at the stable alias."""
|
"""Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs."""
|
||||||
code, out = _run_config(
|
code, out = _run_config(
|
||||||
tmp_path,
|
tmp_path,
|
||||||
"raw-qwen.yaml",
|
"raw-qwen.yaml",
|
||||||
BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"),
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
"delegation:\n provider: harness",
|
||||||
|
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
|
||||||
|
),
|
||||||
)
|
)
|
||||||
assert code == 1, out
|
assert code == 0, out
|
||||||
assert "model.default = 'qwen3.6-27B-code'" in out
|
assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out
|
||||||
assert "gpu-dense" in out
|
assert "prefer the stable alias gpu-dense" in out
|
||||||
assert "RESULT: FAIL" in out
|
assert "RESULT: PASS" in out
|
||||||
|
|
||||||
|
|
||||||
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
||||||
|
|||||||
Reference in New Issue
Block a user