fix(audit): stop requiring the retired gpu-light alias; derive model fields; sweep retired names #80

Merged
abiba-bot merged 8 commits from fix/retired-alias-sweep-20260912 into master 2026-09-12 17:15:06 +00:00
4 changed files with 105 additions and 41 deletions
Showing only changes of commit 9e581ab203 - Show all commits
+49 -23
View File
@@ -36,6 +36,43 @@ def warn(rule, message):
WARNINGS.append(f"[{rule}] {message}")
def _provider_model_fields(label, value):
"""Model-bearing fields from a dict-shaped or list-shaped provider section."""
fields = []
if isinstance(value, dict):
fields.append((f"{label}.model", value.get("model")))
elif isinstance(value, list):
for i, item in enumerate(value):
if isinstance(item, dict):
fields.append((f"{label}[{i}].model", item.get("model")))
return fields
def _model_name_fields(cfg):
"""Every model-name-bearing field in an agent config, as (path, value) pairs."""
fields = []
model = cfg.get("model") or {}
if isinstance(model, dict):
for key, value in model.items():
if key == "default" or "model" in key:
fields.append((f"model.{key}", value))
comp = cfg.get("compression") or {}
if isinstance(comp, dict):
fields.append(("compression.model", comp.get("model")))
aux = cfg.get("auxiliary") or {}
if isinstance(aux, dict):
for name in ("vision", "web_extract", "compression"):
section = aux.get(name) or {}
if isinstance(section, dict):
fields.append((f"auxiliary.{name}.model", section.get("model")))
deleg = cfg.get("delegation") or {}
if isinstance(deleg, dict):
fields.append(("delegation.model", deleg.get("model")))
fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers")))
fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers")))
return fields
def audit(path):
with open(path) as f:
cfg = yaml.safe_load(f)
@@ -180,34 +217,23 @@ def audit(path):
)
# --- No raw or retired model names (Rule 7/8 spirit) ---
# Retired names are a hard failure in every audited model-bearing section: a config that
# names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were
# retired 2026-09-12. Raw-but-live names are a warning only.
retired_names = {
"gpu-light": "gpu-vision",
# Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead.
bad_model_names = {
"gemma-4-12b": "gpu-vision",
"crew-auto": "no replacement alias (64K crew cap removed)",
"gpu-light": "gpu-vision",
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
"qwen3.6-27B-code": "gpu-dense",
"qwen3.6-35B-udq4": "strix-moe",
"ornith-1.0-35b": "strix-moe",
}
raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
for section_path, section_dict in [
("model", model), ("compression", comp),
("auxiliary.vision", aux.get("vision", {})),
("auxiliary.web_extract", aux.get("web_extract", {})),
("auxiliary.compression", aux.get("compression", {})),
("delegation", deleg),
]:
m = section_dict.get("model", "")
if m in retired_names:
for field_path, value in _model_name_fields(cfg):
if value in bad_model_names:
check(
False,
"Rule 7/8",
f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}",
)
elif m in raw_names:
warn(
"Rule 7/8",
f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead "
f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)",
f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}",
)
# --- Report ---
+1 -1
View File
@@ -80,7 +80,7 @@ custom_providers:
auxiliary:
vision:
provider: harness
model: gpu-vision # RTX 5070 stable alias (or syslog-auto)
model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux)
base_url: http://192.168.68.116/litellm/v1
api_key_env: LITELLM_API_KEY
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
+5 -5
View File
@@ -175,11 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup.
- **Max budget**: $100 per key (config default).
```yaml
# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block
# today, and a key generated with no explicit models comes back with an EMPTY models list. This is
# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped
# `/v1/models` view), so re-read the live registry at CT 116
# /opt/inference-harness/litellm_config.yaml before applying.
# NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no
# default_key_generate_params block today, and a key generated with no explicit models comes back
# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to
# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and
# re-verify before applying.
litellm_settings:
default_key_generate_params:
models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
+50 -12
View File
@@ -56,9 +56,9 @@ custom_providers:
"""
def _run(tmp_path, alias):
cfg = tmp_path / f"{alias}.yaml"
cfg.write_text(BASE.format(alias=alias))
def _run_config(tmp_path, name, text):
cfg = tmp_path / name
cfg.write_text(text)
proc = subprocess.run(
[sys.executable, str(AUDIT), str(cfg)],
capture_output=True, text=True,
@@ -66,6 +66,10 @@ def _run(tmp_path, alias):
return proc.returncode, proc.stdout
def _run(tmp_path, alias):
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
def test_live_canonical_alias_passes(tmp_path):
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
code, out = _run(tmp_path, "gpu-vision")
@@ -89,19 +93,53 @@ def test_retired_gemma_is_rejected(tmp_path):
assert "RESULT: FAIL" in out
def test_corrected_compression_example_passes(tmp_path):
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "RESULT: PASS" in out
def test_retired_alias_in_delegation_is_rejected(tmp_path):
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
cfg = tmp_path / "delegation-gpu-light.yaml"
cfg.write_text(
code, out = _run_config(
tmp_path,
"delegation-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: gpu-light",
)
),
)
proc = subprocess.run(
[sys.executable, str(AUDIT), str(cfg)],
capture_output=True, text=True,
assert code == 1, out
assert "delegation.model = 'gpu-light' is retired" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"custom-provider-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" - name: harness\n key_env: LITELLM_API_KEY",
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
),
)
assert proc.returncode == 1, proc.stdout
assert "delegation.model = 'gpu-light' is retired" in proc.stdout
assert "RESULT: FAIL" in proc.stdout
assert code == 1, out
assert "custom_providers[0].model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_raw_alias_is_rejected(tmp_path):
"""Raw-but-live model names must fail too, pointing at the stable alias."""
code, out = _run_config(
tmp_path,
"raw-qwen.yaml",
BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"),
)
assert code == 1, out
assert "model.default = 'qwen3.6-27B-code'" in out
assert "gpu-dense" in out
assert "RESULT: FAIL" in out