no-mistakes(review): Fix compression example alias; make retired aliases fail audit
This commit is contained in:
+18
-7
@@ -180,10 +180,15 @@ def audit(path):
|
||||
)
|
||||
|
||||
# --- No raw or retired model names (Rule 7/8 spirit) ---
|
||||
# Retired names are rejected by the alias rules above; a config that names them directly is
|
||||
# flagged here too. gpu-light/gemma-4-12b/crew-auto were retired 2026-09-12.
|
||||
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b",
|
||||
"gpu-light", "crew-auto"}
|
||||
# Retired names are a hard failure in every audited model-bearing section: a config that
|
||||
# names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were
|
||||
# retired 2026-09-12. Raw-but-live names are a warning only.
|
||||
retired_names = {
|
||||
"gpu-light": "gpu-vision",
|
||||
"gemma-4-12b": "gpu-vision",
|
||||
"crew-auto": "no replacement alias (64K crew cap removed)",
|
||||
}
|
||||
raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
|
||||
for section_path, section_dict in [
|
||||
("model", model), ("compression", comp),
|
||||
("auxiliary.vision", aux.get("vision", {})),
|
||||
@@ -192,11 +197,17 @@ def audit(path):
|
||||
("delegation", deleg),
|
||||
]:
|
||||
m = section_dict.get("model", "")
|
||||
if m in raw_names:
|
||||
if m in retired_names:
|
||||
check(
|
||||
False,
|
||||
"Rule 7/8",
|
||||
f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}",
|
||||
)
|
||||
elif m in raw_names:
|
||||
warn(
|
||||
"Rule 7/8",
|
||||
f"{section_path}.model = {m!r} — raw or retired model name, use a live stable alias "
|
||||
f"instead (gpu-vision, gpu-dense, strix-moe, syslog-auto)",
|
||||
f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead "
|
||||
f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)",
|
||||
)
|
||||
|
||||
# --- Report ---
|
||||
|
||||
@@ -95,7 +95,7 @@ auxiliary:
|
||||
threshold: 0.65
|
||||
target_ratio: 0.3
|
||||
provider: harness
|
||||
model: syslog-auto # or gpu-vision
|
||||
model: syslog-auto # Rule 7: compression must be syslog-auto
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
api_key_env: LITELLM_API_KEY
|
||||
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
||||
@@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension.
|
||||
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
|
||||
|
||||
**models.json** — Must only list models authorized for the agent's LiteLLM key.
|
||||
Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||
`/v1/models` is key-scoped and the live registry is CT 116
|
||||
`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read
|
||||
the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||
```json
|
||||
{
|
||||
"providers": {
|
||||
|
||||
@@ -175,7 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup.
|
||||
- **Max budget**: $100 per key (config default).
|
||||
|
||||
```yaml
|
||||
# In litellm_config.yaml — ensures all future keys inherit these defaults:
|
||||
# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block
|
||||
# today, and a key generated with no explicit models comes back with an EMPTY models list. This is
|
||||
# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped
|
||||
# `/v1/models` view), so re-read the live registry at CT 116
|
||||
# /opt/inference-harness/litellm_config.yaml before applying.
|
||||
litellm_settings:
|
||||
default_key_generate_params:
|
||||
models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
|
||||
@@ -292,7 +296,7 @@ auxiliary:
|
||||
api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
model: gpu-vision
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
```
|
||||
|
||||
|
||||
@@ -96,6 +96,8 @@ call enable-prompt-caching
|
||||
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
|
||||
|
||||
-- Phase 5: Verify end-to-end latency
|
||||
-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is
|
||||
-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use.
|
||||
|
||||
call verify-latency
|
||||
host: 192.168.68.116
|
||||
|
||||
@@ -87,3 +87,21 @@ def test_retired_gemma_is_rejected(tmp_path):
|
||||
assert code == 1, out
|
||||
assert "auxiliary.vision.model must be gpu-vision" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_delegation_is_rejected(tmp_path):
|
||||
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
|
||||
cfg = tmp_path / "delegation-gpu-light.yaml"
|
||||
cfg.write_text(
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
"delegation:\n provider: harness",
|
||||
"delegation:\n provider: harness\n model: gpu-light",
|
||||
)
|
||||
)
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(AUDIT), str(cfg)],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
assert proc.returncode == 1, proc.stdout
|
||||
assert "delegation.model = 'gpu-light' is retired" in proc.stdout
|
||||
assert "RESULT: FAIL" in proc.stdout
|
||||
|
||||
Reference in New Issue
Block a user