From ce48070f21fdb021ac60fa6bc423dab5d776587f Mon Sep 17 00:00:00 2001 From: root Date: Sat, 12 Sep 2026 16:33:58 +0000 Subject: [PATCH] no-mistakes(review): Derive model fields recursively; fix historical latency and key claims --- audit-hermes-config.py | 86 +++++++++++++++---------- hermes-key-enforcement.prose.md | 2 +- litellm-client-timeouts.prose.md | 5 +- tests/test_audit_hermes_config_alias.py | 43 +++++++++++++ 4 files changed, 97 insertions(+), 39 deletions(-) diff --git a/audit-hermes-config.py b/audit-hermes-config.py index a55442c..561dc99 100644 --- a/audit-hermes-config.py +++ b/audit-hermes-config.py @@ -36,41 +36,57 @@ def warn(rule, message): WARNINGS.append(f"[{rule}] {message}") -def _provider_model_fields(label, value): - """Model-bearing fields from a dict-shaped or list-shaped provider section.""" - fields = [] - if isinstance(value, dict): - fields.append((f"{label}.model", value.get("model"))) - elif isinstance(value, list): - for i, item in enumerate(value): - if isinstance(item, dict): - fields.append((f"{label}[{i}].model", item.get("model"))) - return fields +# Derivation rule: a model name is any scalar under a mapping key named `model` or +# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model +# name lives under `default`/`model`/`model_name` inside that section, so it is descended +# specially. The only other exception is key `models` (litellm key-generation params carry a +# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by +# hand. +MODEL_KEYS = ("model", "model_name") +MODEL_SECTION_KEYS = ("default", "model", "model_name") +MODEL_LIST_KEYS = ("models",) -def _model_name_fields(cfg): - """Every model-name-bearing field in an agent config, as (path, value) pairs.""" - fields = [] - model = cfg.get("model") or {} - if isinstance(model, dict): - for key, value in model.items(): - if key == "default" or "model" in key: - fields.append((f"model.{key}", value)) - comp = cfg.get("compression") or {} - if isinstance(comp, dict): - fields.append(("compression.model", comp.get("model"))) - aux = cfg.get("auxiliary") or {} - if isinstance(aux, dict): - for name in ("vision", "web_extract", "compression"): - section = aux.get(name) or {} - if isinstance(section, dict): - fields.append((f"auxiliary.{name}.model", section.get("model"))) - deleg = cfg.get("delegation") or {} - if isinstance(deleg, dict): - fields.append(("delegation.model", deleg.get("model"))) - fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers"))) - fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers"))) - return fields +def _iter_model_values(node, path=""): + """Yield (path, value) for every model-name-bearing scalar in a config.""" + if isinstance(node, dict): + for key, value in node.items(): + child = f"{path}.{key}" if path else key + if key in MODEL_KEYS: + if isinstance(value, dict): + for subkey in MODEL_SECTION_KEYS: + subvalue = value.get(subkey) + if isinstance(subvalue, str): + yield (f"{child}.{subkey}", subvalue) + for subkey, subvalue in value.items(): + if isinstance(subvalue, (dict, list)): + yield from _iter_model_values(subvalue, f"{child}.{subkey}") + elif isinstance(value, list): + yield from _iter_model_values(value, child) + else: + yield (child, value) + elif key in MODEL_LIST_KEYS: + yield from _iter_model_list(value, child) + elif isinstance(value, (dict, list)): + yield from _iter_model_values(value, child) + elif isinstance(node, list): + for i, item in enumerate(node): + yield from _iter_model_values(item, f"{path}[{i}]") + + +def _iter_model_list(node, path): + """Yield scalars under an allowlisted `models` key (list of names or list of dicts).""" + if isinstance(node, list): + for i, item in enumerate(node): + yield from _iter_model_list(item, f"{path}[{i}]") + elif isinstance(node, dict): + for key, value in node.items(): + if key in MODEL_KEYS and isinstance(value, str): + yield (f"{path}.{key}", value) + elif isinstance(value, (dict, list)): + yield from _iter_model_list(value, f"{path}.{key}") + else: + yield (path, node) def audit(path): @@ -217,7 +233,7 @@ def audit(path): ) # --- No raw or retired model names (Rule 7/8 spirit) --- - # Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets + # Any retired/raw name found by the derivation above is a hard failure: such a config gets # 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/ # crew-auto were retired 2026-09-12; raw model names must use stable aliases instead. bad_model_names = { @@ -228,7 +244,7 @@ def audit(path): "qwen3.6-35B-udq4": "strix-moe", "ornith-1.0-35b": "strix-moe", } - for field_path, value in _model_name_fields(cfg): + for field_path, value in _iter_model_values(cfg): if value in bad_model_names: check( False, diff --git a/hermes-key-enforcement.prose.md b/hermes-key-enforcement.prose.md index b712c0e..96b76d3 100644 --- a/hermes-key-enforcement.prose.md +++ b/hermes-key-enforcement.prose.md @@ -169,7 +169,7 @@ The agent picks up the new key via `infisical run --` at gateway startup. **Keys are permanent and use bare agent name aliases.** -- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`. +- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.) - **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity. - **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven. - **Max budget**: $100 per key (config default). diff --git a/litellm-client-timeouts.prose.md b/litellm-client-timeouts.prose.md index a7d5aca..7987ac7 100644 --- a/litellm-client-timeouts.prose.md +++ b/litellm-client-timeouts.prose.md @@ -28,7 +28,7 @@ description: > | syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load | | qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto | | strix-moe | 7.5s | — | Strix Halo, healthy | -| gpu-vision | 2.6s | — | RTX 5070, healthy | +| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy | Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged backend), full recovery 07:00-08:00 with ZERO client failures once requests @@ -53,8 +53,7 @@ proxy queuing. ### 2. Auxiliary tasks — keep template timeouts, one correction -- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s; - these are fine. +- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`. - compression: 300s (keep — this was already raised from 60 per gpu-fleet). - **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090 (qwen3.6-27B-code backend, 23.0s avg) is the same speed class as diff --git a/tests/test_audit_hermes_config_alias.py b/tests/test_audit_hermes_config_alias.py index b489d67..f4db4d9 100644 --- a/tests/test_audit_hermes_config_alias.py +++ b/tests/test_audit_hermes_config_alias.py @@ -143,3 +143,46 @@ def test_raw_alias_is_rejected(tmp_path): assert "model.default = 'qwen3.6-27B-code'" in out assert "gpu-dense" in out assert "RESULT: FAIL" in out + + +def test_retired_alias_in_fallback_providers_is_rejected(tmp_path): + """fallback_providers.model is model-bearing; a retired name there must fail.""" + code, out = _run_config( + tmp_path, + "fallback-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"), + ) + assert code == 1, out + assert "fallback_providers.model = 'gpu-light'" in out + assert "RESULT: FAIL" in out + + +def test_retired_alias_in_x_search_is_rejected(tmp_path): + """x_search.model was previously not enumerated; the derivation must catch it.""" + code, out = _run_config( + tmp_path, + "x-search-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace( + "delegation:\n provider: harness", + "delegation:\n provider: harness\nx_search:\n model: gpu-light", + ), + ) + assert code == 1, out + assert "x_search.model = 'gpu-light'" in out + assert "RESULT: FAIL" in out + + +def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path): + """A nested auxiliary sub-block outside the named three must still be derived.""" + code, out = _run_config( + tmp_path, + "nested-aux-gpu-light.yaml", + BASE.format(alias="gpu-vision").replace( + " compression:\n model: syslog-auto\n provider: harness\ndelegation:", + " compression:\n model: syslog-auto\n provider: harness\n" + " tasks:\n summarize:\n model: gpu-light\ndelegation:", + ), + ) + assert code == 1, out + assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out + assert "RESULT: FAIL" in out