no-mistakes(review): Fix compression example alias; make retired aliases fail audit

This commit is contained in:
root
2026-09-12 16:19:43 +00:00
parent 221f9f79f3
commit 9cac3589cf
5 changed files with 48 additions and 11 deletions
+18 -7
View File
@@ -180,10 +180,15 @@ def audit(path):
)
# --- No raw or retired model names (Rule 7/8 spirit) ---
# Retired names are rejected by the alias rules above; a config that names them directly is
# flagged here too. gpu-light/gemma-4-12b/crew-auto were retired 2026-09-12.
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b",
"gpu-light", "crew-auto"}
# Retired names are a hard failure in every audited model-bearing section: a config that
# names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were
# retired 2026-09-12. Raw-but-live names are a warning only.
retired_names = {
"gpu-light": "gpu-vision",
"gemma-4-12b": "gpu-vision",
"crew-auto": "no replacement alias (64K crew cap removed)",
}
raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
for section_path, section_dict in [
("model", model), ("compression", comp),
("auxiliary.vision", aux.get("vision", {})),
@@ -192,11 +197,17 @@ def audit(path):
("delegation", deleg),
]:
m = section_dict.get("model", "")
if m in raw_names:
if m in retired_names:
check(
False,
"Rule 7/8",
f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}",
)
elif m in raw_names:
warn(
"Rule 7/8",
f"{section_path}.model = {m!r} — raw or retired model name, use a live stable alias "
f"instead (gpu-vision, gpu-dense, strix-moe, syslog-auto)",
f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead "
f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)",
)
# --- Report ---
+4 -2
View File
@@ -95,7 +95,7 @@ auxiliary:
threshold: 0.65
target_ratio: 0.3
provider: harness
model: syslog-auto # or gpu-vision
model: syslog-auto # Rule 7: compression must be syslog-auto
base_url: http://192.168.68.116/litellm/v1
api_key_env: LITELLM_API_KEY
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
@@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension.
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
**models.json** — Must only list models authorized for the agent's LiteLLM key.
Key is injected via `infisical run --` wrapper at PM2 startup:
`/v1/models` is key-scoped and the live registry is CT 116
`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read
the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup:
```json
{
"providers": {
+6 -2
View File
@@ -175,7 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup.
- **Max budget**: $100 per key (config default).
```yaml
# In litellm_config.yaml — ensures all future keys inherit these defaults:
# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block
# today, and a key generated with no explicit models comes back with an EMPTY models list. This is
# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped
# `/v1/models` view), so re-read the live registry at CT 116
# /opt/inference-harness/litellm_config.yaml before applying.
litellm_settings:
default_key_generate_params:
models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
@@ -292,7 +296,7 @@ auxiliary:
api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1
model: gpu-vision
model: syslog-auto
provider: harness
```
+2
View File
@@ -96,6 +96,8 @@ call enable-prompt-caching
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
-- Phase 5: Verify end-to-end latency
-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is
-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use.
call verify-latency
host: 192.168.68.116
+18
View File
@@ -87,3 +87,21 @@ def test_retired_gemma_is_rejected(tmp_path):
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_delegation_is_rejected(tmp_path):
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
cfg = tmp_path / "delegation-gpu-light.yaml"
cfg.write_text(
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: gpu-light",
)
)
proc = subprocess.run(
[sys.executable, str(AUDIT), str(cfg)],
capture_output=True, text=True,
)
assert proc.returncode == 1, proc.stdout
assert "delegation.model = 'gpu-light' is retired" in proc.stdout
assert "RESULT: FAIL" in proc.stdout