no-mistakes(review): Derive model fields recursively; fix historical latency and key claims

This commit is contained in:
root
2026-09-12 16:33:58 +00:00
parent 9e581ab203
commit ce48070f21
4 changed files with 97 additions and 39 deletions
+51 -35
View File
@@ -36,41 +36,57 @@ def warn(rule, message):
WARNINGS.append(f"[{rule}] {message}")
def _provider_model_fields(label, value):
"""Model-bearing fields from a dict-shaped or list-shaped provider section."""
fields = []
if isinstance(value, dict):
fields.append((f"{label}.model", value.get("model")))
elif isinstance(value, list):
for i, item in enumerate(value):
if isinstance(item, dict):
fields.append((f"{label}[{i}].model", item.get("model")))
return fields
# Derivation rule: a model name is any scalar under a mapping key named `model` or
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
# specially. The only other exception is key `models` (litellm key-generation params carry a
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
# hand.
MODEL_KEYS = ("model", "model_name")
MODEL_SECTION_KEYS = ("default", "model", "model_name")
MODEL_LIST_KEYS = ("models",)
def _model_name_fields(cfg):
"""Every model-name-bearing field in an agent config, as (path, value) pairs."""
fields = []
model = cfg.get("model") or {}
if isinstance(model, dict):
for key, value in model.items():
if key == "default" or "model" in key:
fields.append((f"model.{key}", value))
comp = cfg.get("compression") or {}
if isinstance(comp, dict):
fields.append(("compression.model", comp.get("model")))
aux = cfg.get("auxiliary") or {}
if isinstance(aux, dict):
for name in ("vision", "web_extract", "compression"):
section = aux.get(name) or {}
if isinstance(section, dict):
fields.append((f"auxiliary.{name}.model", section.get("model")))
deleg = cfg.get("delegation") or {}
if isinstance(deleg, dict):
fields.append(("delegation.model", deleg.get("model")))
fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers")))
fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers")))
return fields
def _iter_model_values(node, path=""):
"""Yield (path, value) for every model-name-bearing scalar in a config."""
if isinstance(node, dict):
for key, value in node.items():
child = f"{path}.{key}" if path else key
if key in MODEL_KEYS:
if isinstance(value, dict):
for subkey in MODEL_SECTION_KEYS:
subvalue = value.get(subkey)
if isinstance(subvalue, str):
yield (f"{child}.{subkey}", subvalue)
for subkey, subvalue in value.items():
if isinstance(subvalue, (dict, list)):
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
elif isinstance(value, list):
yield from _iter_model_values(value, child)
else:
yield (child, value)
elif key in MODEL_LIST_KEYS:
yield from _iter_model_list(value, child)
elif isinstance(value, (dict, list)):
yield from _iter_model_values(value, child)
elif isinstance(node, list):
for i, item in enumerate(node):
yield from _iter_model_values(item, f"{path}[{i}]")
def _iter_model_list(node, path):
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
if isinstance(node, list):
for i, item in enumerate(node):
yield from _iter_model_list(item, f"{path}[{i}]")
elif isinstance(node, dict):
for key, value in node.items():
if key in MODEL_KEYS and isinstance(value, str):
yield (f"{path}.{key}", value)
elif isinstance(value, (dict, list)):
yield from _iter_model_list(value, f"{path}.{key}")
else:
yield (path, node)
def audit(path):
@@ -217,7 +233,7 @@ def audit(path):
)
# --- No raw or retired model names (Rule 7/8 spirit) ---
# Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets
# Any retired/raw name found by the derivation above is a hard failure: such a config gets
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead.
bad_model_names = {
@@ -228,7 +244,7 @@ def audit(path):
"qwen3.6-35B-udq4": "strix-moe",
"ornith-1.0-35b": "strix-moe",
}
for field_path, value in _model_name_fields(cfg):
for field_path, value in _iter_model_values(cfg):
if value in bad_model_names:
check(
False,
+1 -1
View File
@@ -169,7 +169,7 @@ The agent picks up the new key via `infisical run --` at gateway startup.
**Keys are permanent and use bare agent name aliases.**
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`.
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
- **Max budget**: $100 per key (config default).
+2 -3
View File
@@ -28,7 +28,7 @@ description: >
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
| strix-moe | 7.5s | — | Strix Halo, healthy |
| gpu-vision | 2.6s | — | RTX 5070, healthy |
| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
backend), full recovery 07:00-08:00 with ZERO client failures once requests
@@ -53,8 +53,7 @@ proxy queuing.
### 2. Auxiliary tasks — keep template timeouts, one correction
- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s;
these are fine.
- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
+43
View File
@@ -143,3 +143,46 @@ def test_raw_alias_is_rejected(tmp_path):
assert "model.default = 'qwen3.6-27B-code'" in out
assert "gpu-dense" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
"""fallback_providers.model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"fallback-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
)
assert code == 1, out
assert "fallback_providers.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_x_search_is_rejected(tmp_path):
"""x_search.model was previously not enumerated; the derivation must catch it."""
code, out = _run_config(
tmp_path,
"x-search-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
),
)
assert code == 1, out
assert "x_search.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
"""A nested auxiliary sub-block outside the named three must still be derived."""
code, out = _run_config(
tmp_path,
"nested-aux-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
" compression:\n model: syslog-auto\n provider: harness\n"
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
),
)
assert code == 1, out
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out