From 9cac3589cf82500ac58917b3283a2cb6afb58c9f Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:19:43 +0000 Subject: [PATCH] no-mistakes(review): Fix compression example alias; make retired aliases fail audit --- audit-hermes-config.py | 25 ++++++++++++++++++------- hermes-agent-baseline.prose.md | 6 ++++-- hermes-key-enforcement.prose.md | 8 ++++++-- inference-optimization.prose.md | 2 ++ tests/test_audit_hermes_config_alias.py | 18 ++++++++++++++++++ 5 files changed, 48 insertions(+), 11 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index b0f3323..4d14234 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -180,10 +180,15 @@ def audit(path): ) # --- No raw or retired model names (Rule 7/8 spirit) --- - # Retired names are rejected by the alias rules above; a config that names them directly is - # flagged here too. gpu-light/gemma-4-12b/crew-auto were retired 2026-09-12. - raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b", - "gpu-light", "crew-auto"} + # Retired names are a hard failure in every audited model-bearing section: a config that + # names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were + # retired 2026-09-12. Raw-but-live names are a warning only. + retired_names = { + "gpu-light": "gpu-vision", + "gemma-4-12b": "gpu-vision", + "crew-auto": "no replacement alias (64K crew cap removed)", + } + raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} for section_path, section_dict in [ ("model", model), ("compression", comp), ("auxiliary.vision", aux.get("vision", {})), @@ -192,11 +197,17 @@ def audit(path): ("delegation", deleg), ]: m = section_dict.get("model", "") - if m in raw_names: + if m in retired_names: + check( + False, + "Rule 7/8", + f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}", + ) + elif m in raw_names: warn( "Rule 7/8", - f"{section_path}.model = {m!r} — raw or retired model name, use a live stable alias " - f"instead (gpu-vision, gpu-dense, strix-moe, syslog-auto)", + f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead " + f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)", ) # --- Report --- diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index c9513fc..b77044c 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -95,7 +95,7 @@ auxiliary: threshold: 0.65 target_ratio: 0.3 provider: harness - model: syslog-auto # or gpu-vision + model: syslog-auto # Rule 7: compression must be syslog-auto base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround @@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension. Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`. **models.json** — Must only list models authorized for the agent's LiteLLM key. -Key is injected via `infisical run --` wrapper at PM2 startup: +`/v1/models` is key-scoped and the live registry is CT 116 +`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read +the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup: ```json { "providers": { diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index 45d0296..fb7886c 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -175,7 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup. - **Max budget**: $100 per key (config default). ```yaml -# In litellm_config.yaml — ensures all future keys inherit these defaults: +# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block +# today, and a key generated with no explicit models comes back with an EMPTY models list. This is +# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped +# `/v1/models` view), so re-read the live registry at CT 116 +# /opt/inference-harness/litellm_config.yaml before applying. litellm_settings: default_key_generate_params: models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] @@ -292,7 +296,7 @@ auxiliary: api_key: sk- # ← workaround (same as above) api_key_env: LITELLM_API_KEY base_url: http://192.168.68.116/litellm/v1 - model: gpu-vision + model: syslog-auto provider: harness ``` diff --git a/inference-optimization.prose.md b/inference-optimization.prose.md index bd5dbb2..beb18a2 100644 --- a/inference-optimization.prose.md +++ b/inference-optimization.prose.md @@ -96,6 +96,8 @@ call enable-prompt-caching hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110] -- Phase 5: Verify end-to-end latency +-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is +-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use. call verify-latency host: 192.168.68.116 diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index 4416318..f14016a 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -87,3 +87,21 @@ def test_retired_gemma_is_rejected(tmp_path): assert code == 1, out assert "auxiliary.vision.model must be gpu-vision" in out assert "RESULT: FAIL" in out + + +def test_retired_alias_in_delegation_is_rejected(tmp_path): + """delegation.model has no dedicated value rule, so a retired name there used to PASS.""" + cfg = tmp_path / "delegation-gpu-light.yaml" + cfg.write_text( + BASE.format(alias="gpu-vision").replace( + "delegation:\n provider: harness", + "delegation:\n provider: harness\n model: gpu-light", + ) + ) + proc = subprocess.run( + [sys.executable, str(AUDIT), str(cfg)], + capture_output=True, text=True, + ) + assert proc.returncode == 1, proc.stdout + assert "delegation.model = 'gpu-light' is retired" in proc.stdout + assert "RESULT: FAIL" in proc.stdout