no-mistakes(review): Derive model fields recursively; fix historical latency and key claims
This commit is contained in:
+51
-35
@@ -36,41 +36,57 @@ def warn(rule, message):
|
|||||||
WARNINGS.append(f"[{rule}] {message}")
|
WARNINGS.append(f"[{rule}] {message}")
|
||||||
|
|
||||||
|
|
||||||
def _provider_model_fields(label, value):
|
# Derivation rule: a model name is any scalar under a mapping key named `model` or
|
||||||
"""Model-bearing fields from a dict-shaped or list-shaped provider section."""
|
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
|
||||||
fields = []
|
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
|
||||||
if isinstance(value, dict):
|
# specially. The only other exception is key `models` (litellm key-generation params carry a
|
||||||
fields.append((f"{label}.model", value.get("model")))
|
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
|
||||||
elif isinstance(value, list):
|
# hand.
|
||||||
for i, item in enumerate(value):
|
MODEL_KEYS = ("model", "model_name")
|
||||||
if isinstance(item, dict):
|
MODEL_SECTION_KEYS = ("default", "model", "model_name")
|
||||||
fields.append((f"{label}[{i}].model", item.get("model")))
|
MODEL_LIST_KEYS = ("models",)
|
||||||
return fields
|
|
||||||
|
|
||||||
|
|
||||||
def _model_name_fields(cfg):
|
def _iter_model_values(node, path=""):
|
||||||
"""Every model-name-bearing field in an agent config, as (path, value) pairs."""
|
"""Yield (path, value) for every model-name-bearing scalar in a config."""
|
||||||
fields = []
|
if isinstance(node, dict):
|
||||||
model = cfg.get("model") or {}
|
for key, value in node.items():
|
||||||
if isinstance(model, dict):
|
child = f"{path}.{key}" if path else key
|
||||||
for key, value in model.items():
|
if key in MODEL_KEYS:
|
||||||
if key == "default" or "model" in key:
|
if isinstance(value, dict):
|
||||||
fields.append((f"model.{key}", value))
|
for subkey in MODEL_SECTION_KEYS:
|
||||||
comp = cfg.get("compression") or {}
|
subvalue = value.get(subkey)
|
||||||
if isinstance(comp, dict):
|
if isinstance(subvalue, str):
|
||||||
fields.append(("compression.model", comp.get("model")))
|
yield (f"{child}.{subkey}", subvalue)
|
||||||
aux = cfg.get("auxiliary") or {}
|
for subkey, subvalue in value.items():
|
||||||
if isinstance(aux, dict):
|
if isinstance(subvalue, (dict, list)):
|
||||||
for name in ("vision", "web_extract", "compression"):
|
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
|
||||||
section = aux.get(name) or {}
|
elif isinstance(value, list):
|
||||||
if isinstance(section, dict):
|
yield from _iter_model_values(value, child)
|
||||||
fields.append((f"auxiliary.{name}.model", section.get("model")))
|
else:
|
||||||
deleg = cfg.get("delegation") or {}
|
yield (child, value)
|
||||||
if isinstance(deleg, dict):
|
elif key in MODEL_LIST_KEYS:
|
||||||
fields.append(("delegation.model", deleg.get("model")))
|
yield from _iter_model_list(value, child)
|
||||||
fields.extend(_provider_model_fields("fallback_providers", cfg.get("fallback_providers")))
|
elif isinstance(value, (dict, list)):
|
||||||
fields.extend(_provider_model_fields("custom_providers", cfg.get("custom_providers")))
|
yield from _iter_model_values(value, child)
|
||||||
return fields
|
elif isinstance(node, list):
|
||||||
|
for i, item in enumerate(node):
|
||||||
|
yield from _iter_model_values(item, f"{path}[{i}]")
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_model_list(node, path):
|
||||||
|
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
|
||||||
|
if isinstance(node, list):
|
||||||
|
for i, item in enumerate(node):
|
||||||
|
yield from _iter_model_list(item, f"{path}[{i}]")
|
||||||
|
elif isinstance(node, dict):
|
||||||
|
for key, value in node.items():
|
||||||
|
if key in MODEL_KEYS and isinstance(value, str):
|
||||||
|
yield (f"{path}.{key}", value)
|
||||||
|
elif isinstance(value, (dict, list)):
|
||||||
|
yield from _iter_model_list(value, f"{path}.{key}")
|
||||||
|
else:
|
||||||
|
yield (path, node)
|
||||||
|
|
||||||
|
|
||||||
def audit(path):
|
def audit(path):
|
||||||
@@ -217,7 +233,7 @@ def audit(path):
|
|||||||
)
|
)
|
||||||
|
|
||||||
# --- No raw or retired model names (Rule 7/8 spirit) ---
|
# --- No raw or retired model names (Rule 7/8 spirit) ---
|
||||||
# Any retired/raw name in ANY model-bearing field is a hard failure: such a config gets
|
# Any retired/raw name found by the derivation above is a hard failure: such a config gets
|
||||||
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/
|
# 400 Invalid model name (or lands in the wrong pool) at runtime. gpu-light/gemma-4-12b/
|
||||||
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead.
|
# crew-auto were retired 2026-09-12; raw model names must use stable aliases instead.
|
||||||
bad_model_names = {
|
bad_model_names = {
|
||||||
@@ -228,7 +244,7 @@ def audit(path):
|
|||||||
"qwen3.6-35B-udq4": "strix-moe",
|
"qwen3.6-35B-udq4": "strix-moe",
|
||||||
"ornith-1.0-35b": "strix-moe",
|
"ornith-1.0-35b": "strix-moe",
|
||||||
}
|
}
|
||||||
for field_path, value in _model_name_fields(cfg):
|
for field_path, value in _iter_model_values(cfg):
|
||||||
if value in bad_model_names:
|
if value in bad_model_names:
|
||||||
check(
|
check(
|
||||||
False,
|
False,
|
||||||
|
|||||||
@@ -169,7 +169,7 @@ The agent picks up the new key via `infisical run --` at gateway startup.
|
|||||||
|
|
||||||
**Keys are permanent and use bare agent name aliases.**
|
**Keys are permanent and use bare agent name aliases.**
|
||||||
|
|
||||||
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`.
|
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||||
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
|
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
|
||||||
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
|
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
|
||||||
- **Max budget**: $100 per key (config default).
|
- **Max budget**: $100 per key (config default).
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ description: >
|
|||||||
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
|
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
|
||||||
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
|
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
|
||||||
| strix-moe | 7.5s | — | Strix Halo, healthy |
|
| strix-moe | 7.5s | — | Strix Halo, healthy |
|
||||||
| gpu-vision | 2.6s | — | RTX 5070, healthy |
|
| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
|
||||||
|
|
||||||
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
|
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
|
||||||
backend), full recovery 07:00-08:00 with ZERO client failures once requests
|
backend), full recovery 07:00-08:00 with ZERO client failures once requests
|
||||||
@@ -53,8 +53,7 @@ proxy queuing.
|
|||||||
|
|
||||||
### 2. Auxiliary tasks — keep template timeouts, one correction
|
### 2. Auxiliary tasks — keep template timeouts, one correction
|
||||||
|
|
||||||
- vision: 60s (keep), web_extract: 30s (keep) — gpu-vision averages 2.6s;
|
- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
|
||||||
these are fine.
|
|
||||||
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
|
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
|
||||||
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
|
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
|
||||||
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
|
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
|
||||||
|
|||||||
@@ -143,3 +143,46 @@ def test_raw_alias_is_rejected(tmp_path):
|
|||||||
assert "model.default = 'qwen3.6-27B-code'" in out
|
assert "model.default = 'qwen3.6-27B-code'" in out
|
||||||
assert "gpu-dense" in out
|
assert "gpu-dense" in out
|
||||||
assert "RESULT: FAIL" in out
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
||||||
|
"""fallback_providers.model is model-bearing; a retired name there must fail."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"fallback-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "fallback_providers.model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_x_search_is_rejected(tmp_path):
|
||||||
|
"""x_search.model was previously not enumerated; the derivation must catch it."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"x-search-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
"delegation:\n provider: harness",
|
||||||
|
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "x_search.model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
|
||||||
|
"""A nested auxiliary sub-block outside the named three must still be derived."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"nested-aux-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
|
||||||
|
" compression:\n model: syslog-auto\n provider: harness\n"
|
||||||
|
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|||||||
Reference in New Issue
Block a user