no-mistakes(review): Derive model fields recursively; fix historical latency and key claims
This commit is contained in:
+51
-35
@@ -36,41 +36,57 @@ def warn(rule, message):
|
||||
WARNINGS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def _provider_model_fields(label, value):
|
||||
"""Model-bearing fields from a dict-shaped or list-shaped provider section."""
|
||||
fields = []
|
||||
if isinstance(value, dict):
|
||||
fields.append((f"{label}.model", value.get("model")))
|
||||
elif isinstance(value, list):
|
||||
for i, item in enumerate(value):
|
||||
if isinstance(item, dict):
|
||||
fields.append((f"{label}[{i}].model", item.get("model")))
|
||||
return fields
|
||||
# Derivation rule: a model name is any scalar under a mapping key named `model` or
|
||||
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
|
||||
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
|
||||
# specially. The only other exception is key `models` (litellm key-generation params carry a
|
||||
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
|
||||
# hand.
|
||||
MODEL_KEYS = ("model", "model_name")
|
||||
MODEL_SECTION_KEYS = ("default", "model", "model_name")
|
||||
MODEL_LIST_KEYS = ("models",)
|
||||
|
||||
|
||||
def _model_name_fields(cfg):
|
||||
"""Every model-name-bearing field in an agent config, as (path, value) pairs."""
|
||||
fields = []
|
||||
model = cfg.get("model") or {}
|
||||
if isinstance(model, dict):
|
||||
for key, value in model.items():
|
||||
if key == "default" or "model" in key:
|
||||
fields.append((f"model.{key}", value))
|
||||
comp = cfg.get("compression") or {}
|
||||
if isinstance(comp, dict):
|
||||
fields.append(("compression.model", comp.get("model")))
|
||||
aux = cfg.get("auxiliary") or {}
|
||||
if isinstance(aux, dict):
|
||||
for name in ("vision", "web_extract", "compression"):
|
||||
section = aux.get(name) or {}
|
||||
if isinstance(section, dict):
|
||||
fields.append((f"auxiliary.{name}.model", section.get("model")))
|
||||
deleg = cfg.get("delegation") or {}
|
||||
if isinstance(deleg, dict):
|
||||
fields.append(("delegation.model", deleg.get("model")))
|
||||
fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers")))
|
||||
fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers")))
|
||||
return fields
|
||||
def _iter_model_values(node, path=""):
|
||||
"""Yield (path, value) for every model-name-bearing scalar in a config."""
|
||||
if isinstance(node, dict):
|
||||
for key, value in node.items():
|
||||
child = f"{path}.{key}" if path else key
|
||||
if key in MODEL_KEYS:
|
||||
if isinstance(value, dict):
|
||||
for subkey in MODEL_SECTION_KEYS:
|
||||
subvalue = value.get(subkey)
|
||||
if isinstance(subvalue, str):
|
||||
yield (f"{child}.{subkey}", subvalue)
|
||||
for subkey, subvalue in value.items():
|
||||
if isinstance(subvalue, (dict, list)):
|
||||
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
|
||||
elif isinstance(value, list):
|
||||
yield from _iter_model_values(value, child)
|
||||
else:
|
||||
yield (child, value)
|
||||
elif key in MODEL_LIST_KEYS:
|
||||
yield from _iter_model_list(value, child)
|
||||
elif isinstance(value, (dict, list)):
|
||||
yield from _iter_model_values(value, child)
|
||||
elif isinstance(node, list):
|
||||
for i, item in enumerate(node):
|
||||
yield from _iter_model_values(item, f"{path}[{i}]")
|
||||
|
||||
|
||||
def _iter_model_list(node, path):
|
||||
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
|
||||
if isinstance(node, list):
|
||||
for i, item in enumerate(node):
|
||||
yield from _iter_model_list(item, f"{path}[{i}]")
|
||||
elif isinstance(node, dict):
|
||||
for key, value in node.items():
|
||||
if key in MODEL_KEYS and isinstance(value, str):
|
||||
yield (f"{path}.{key}", value)
|
||||
elif isinstance(value, (dict, list)):
|
||||
yield from _iter_model_list(value, f"{path}.{key}")
|
||||
else:
|
||||
yield (path, node)
|
||||
|
||||
|
||||
def audit(path):
|
||||
@@ -217,7 +233,7 @@ def audit(path):
|
||||
)
|
||||
|
||||
# --- No raw or retired model names (Rule 7/8 spirit) ---
|
||||
# Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets
|
||||
# Any retired/raw name found by the derivation above is a hard failure: such a config gets
|
||||
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/
|
||||
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead.
|
||||
bad_model_names = {
|
||||
@@ -228,7 +244,7 @@ def audit(path):
|
||||
"qwen3.6-35B-udq4": "strix-moe",
|
||||
"ornith-1.0-35b": "strix-moe",
|
||||
}
|
||||
for field_path, value in _model_name_fields(cfg):
|
||||
for field_path, value in _iter_model_values(cfg):
|
||||
if value in bad_model_names:
|
||||
check(
|
||||
False,
|
||||
|
||||
@@ -169,7 +169,7 @@ The agent picks up the new key via `infisical run --` at gateway startup.
|
||||
|
||||
**Keys are permanent and use bare agent name aliases.**
|
||||
|
||||
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`.
|
||||
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
|
||||
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
|
||||
- **Max budget**: $100 per key (config default).
|
||||
|
||||
@@ -28,7 +28,7 @@ description: >
|
||||
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
|
||||
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
|
||||
| strix-moe | 7.5s | — | Strix Halo, healthy |
|
||||
| gpu-vision | 2.6s | — | RTX 5070, healthy |
|
||||
| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
|
||||
|
||||
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
|
||||
backend), full recovery 07:00-08:00 with ZERO client failures once requests
|
||||
@@ -53,8 +53,7 @@ proxy queuing.
|
||||
|
||||
### 2. Auxiliary tasks — keep template timeouts, one correction
|
||||
|
||||
- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s;
|
||||
these are fine.
|
||||
- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
|
||||
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
|
||||
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
|
||||
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
|
||||
|
||||
@@ -143,3 +143,46 @@ def test_raw_alias_is_rejected(tmp_path):
|
||||
assert "model.default = 'qwen3.6-27B-code'" in out
|
||||
assert "gpu-dense" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
||||
"""fallback_providers.model is model-bearing; a retired name there must fail."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"fallback-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "fallback_providers.model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_x_search_is_rejected(tmp_path):
|
||||
"""x_search.model was previously not enumerated; the derivation must catch it."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"x-search-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
"delegation:\n provider: harness",
|
||||
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
|
||||
),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "x_search.model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
|
||||
"""A nested auxiliary sub-block outside the named three must still be derived."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"nested-aux-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
|
||||
" compression:\n model: syslog-auto\n provider: harness\n"
|
||||
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
|
||||
),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
Reference in New Issue
Block a user