diff --git a/audit-hermes-config.py b/audit-hermes-config.py index 4d14234..a55442c 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -36,6 +36,43 @@ def warn(rule, message): WARNINGS.append(f"[{rule}] {message}") +def _provider_model_fields(label, value): + """Model-bearing fields from a dict-shaped or list-shaped provider section.""" + fields = [] + if isinstance(value, dict): + fields.append((f"{label}.model", value.get("model"))) + elif isinstance(value, list): + for i, item in enumerate(value): + if isinstance(item, dict): + fields.append((f"{label}[{i}].model", item.get("model"))) + return fields + + +def _model_name_fields(cfg): + """Every model-name-bearing field in an agent config, as (path, value) pairs.""" + fields = [] + model = cfg.get("model") or {} + if isinstance(model, dict): + for key, value in model.items(): + if key == "default" or "model" in key: + fields.append((f"model.{key}", value)) + comp = cfg.get("compression") or {} + if isinstance(comp, dict): + fields.append(("compression.model", comp.get("model"))) + aux = cfg.get("auxiliary") or {} + if isinstance(aux, dict): + for name in ("vision", "web_extract", "compression"): + section = aux.get(name) or {} + if isinstance(section, dict): + fields.append((f"auxiliary.{name}.model", section.get("model"))) + deleg = cfg.get("delegation") or {} + if isinstance(deleg, dict): + fields.append(("delegation.model", deleg.get("model"))) + fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers"))) + fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers"))) + return fields + + def audit(path): with open(path) as f: cfg = yaml.safe_load(f) @@ -180,34 +217,23 @@ def audit(path): ) # --- No raw or retired model names (Rule 7/8 spirit) --- - # Retired names are a hard failure in every audited model-bearing section: a config that - # names them gets 400 Invalid model name at runtime. gpu-light/gemma-4-12b/crew-auto were - # retired 2026-09-12. Raw-but-live names are a warning only. - retired_names = { - "gpu-light": "gpu-vision", + # Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets + # 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ + # crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. + bad_model_names = { "gemma-4-12b": "gpu-vision", - "crew-auto": "no replacement alias (64K crew cap removed)", + "gpu-light": "gpu-vision", + "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", + "qwen3.6-27B-code": "gpu-dense", + "qwen3.6-35B-udq4": "strix-moe", + "ornith-1.0-35b": "strix-moe", } - raw_names = {"qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} - for section_path, section_dict in [ - ("model", model), ("compression", comp), - ("auxiliary.vision", aux.get("vision", {})), - ("auxiliary.web_extract", aux.get("web_extract", {})), - ("auxiliary.compression", aux.get("compression", {})), - ("delegation", deleg), - ]: - m = section_dict.get("model", "") - if m in retired_names: + for field_path, value in _model_name_fields(cfg): + if value in bad_model_names: check( False, "Rule 7/8", - f"{section_path}.model = {m!r} is retired (2026-09-12) — use {retired_names[m]}", - ) - elif m in raw_names: - warn( - "Rule 7/8", - f"{section_path}.model = {m!r} — raw model name, use a live stable alias instead " - f"(gpu-vision, gpu-dense, strix-moe, syslog-auto)", + f"{field_path} = {value!r} is retired/raw (2026-09-12 sweep) — use {bad_model_names[value]}", ) # --- Report --- diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index b77044c..71ab79f 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -80,7 +80,7 @@ custom_providers: auxiliary: vision: provider: harness - model: gpu-vision # RTX 5070 stable alias (or syslog-auto) + model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux) base_url: http://192.168.68.116/litellm/v1 api_key_env: LITELLM_API_KEY api_key: # ← MANDATORY workaround diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index fb7886c..b712c0e 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -175,11 +175,11 @@ The agent picks up the new key via `infisical run --` at gateway startup. - **Max budget**: $100 per key (config default). ```yaml -# SNAPSHOT, not a mirror — CT 116 litellm_config.yaml has NO default_key_generate_params block -# today, and a key generated with no explicit models comes back with an EMPTY models list. This is -# a value to ADD. `models` is a literal key-generation parameter (authoritative in the key-scoped -# `/v1/models` view), so re-read the live registry at CT 116 -# /opt/inference-harness/litellm_config.yaml before applying. +# NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no +# default_key_generate_params block today, and a key generated with no explicit models comes back +# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to +# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and +# re-verify before applying. litellm_settings: default_key_generate_params: models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"] diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index f14016a..b489d67 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -56,9 +56,9 @@ custom_providers: """ -def _run(tmp_path, alias): - cfg = tmp_path / f"{alias}.yaml" - cfg.write_text(BASE.format(alias=alias)) +def _run_config(tmp_path, name, text): + cfg = tmp_path / name + cfg.write_text(text) proc = subprocess.run( [sys.executable, str(AUDIT), str(cfg)], capture_output=True, text=True, @@ -66,6 +66,10 @@ def _run(tmp_path, alias): return proc.returncode, proc.stdout +def _run(tmp_path, alias): + return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias)) + + def test_live_canonical_alias_passes(tmp_path): """The RTX 5070 alias that actually resolves must satisfy Rule 8.""" code, out = _run(tmp_path, "gpu-vision") @@ -89,19 +93,53 @@ def test_retired_gemma_is_rejected(tmp_path): assert "RESULT: FAIL" in out +def test_corrected_compression_example_passes(tmp_path): + """The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS.""" + code, out = _run(tmp_path, "gpu-vision") + assert code == 0, out + assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out + assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out + assert "RESULT: PASS" in out + + def test_retired_alias_in_delegation_is_rejected(tmp_path): """delegation.model has no dedicated value rule, so a retired name there used to PASS.""" - cfg = tmp_path / "delegation-gpu-light.yaml" - cfg.write_text( + code, out = _run_config( + tmp_path, + "delegation-gpu-light.yaml", BASE.format(alias="gpu-vision").replace( "delegation:\n provider: harness", "delegation:\n provider: harness\n model: gpu-light", - ) + ), ) - proc = subprocess.run( - [sys.executable, str(AUDIT), str(cfg)], - capture_output=True, text=True, + assert code == 1, out + assert "delegation.model = 'gpu-light' is retired" in out + assert "RESULT: FAIL" in out + + +def test_retired_alias_in_custom_providers_is_rejected(tmp_path): + """custom_providers[*].model is model-bearing; a retired name there must fail.""" + code, out = _run_config( + tmp_path, + "custom-provider-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace( + " - name: harness\n key_env: LITELLM_API_KEY", + " - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY", + ), ) - assert proc.returncode == 1, proc.stdout - assert "delegation.model = 'gpu-light' is retired" in proc.stdout - assert "RESULT: FAIL" in proc.stdout + assert code == 1, out + assert "custom_providers[0].model = 'gpu-light'" in out + assert "RESULT: FAIL" in out + + +def test_raw_alias_is_rejected(tmp_path): + """Raw-but-live model names must fail too, pointing at the stable alias.""" + code, out = _run_config( + tmp_path, + "raw-qwen.yaml", + BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"), + ) + assert code == 1, out + assert "model.default = 'qwen3.6-27B-code'" in out + assert "gpu-dense" in out + assert "RESULT: FAIL" in out