PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 1s
- Remove retired names (qwen3.6-27B-code, qwen3.6-35B-udq4) from live alias claims - Update Strix Halo model to Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf (strix-moe, 256K ctx) - Fix litellm-health step 7 probe to gpu-vision (monitor key scoped) - Move qwen3.6-27B-code/35B-udq4 from raw-but-live to non-resolving in audit - Fold in pm2-self-heal: remove spoton-service (live PM2 set is 4/4) - Update hermes templates, key enforcement, timeout tables to live names
298 lines
11 KiB
Python
298 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Hermes Config Audit — validates a live config.yaml against the prose contract rules.
|
|
|
|
Usage:
|
|
python3 audit-hermes-config.py <config.yaml>
|
|
python3 audit-hermes-config.py /root/.hermes/config.yaml
|
|
|
|
Exit codes:
|
|
0 = all checks pass
|
|
1 = one or more contract violations found
|
|
|
|
This script encodes every rule from hermes-config-template.prose.md so config
|
|
changes can be verified before and after application. It is the single automated
|
|
enforcement layer for the prose contract.
|
|
|
|
Contract: /root/prose-contracts/hermes-config-template.prose.md
|
|
"""
|
|
|
|
import sys
|
|
import yaml
|
|
|
|
VIOLATIONS = []
|
|
WARNINGS = []
|
|
PASSES = []
|
|
|
|
|
|
def check(condition, rule, message):
|
|
if condition:
|
|
PASSES.append(f"[{rule}] {message}")
|
|
else:
|
|
VIOLATIONS.append(f"[{rule}] {message}")
|
|
|
|
|
|
def warn(rule, message):
|
|
WARNINGS.append(f"[{rule}] {message}")
|
|
|
|
|
|
# Derivation rule: a model name is any scalar under a mapping key named `model` or
|
|
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
|
|
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
|
|
# specially. The only other exception is key `models` (litellm key-generation params carry a
|
|
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
|
|
# hand.
|
|
MODEL_KEYS = ("model", "model_name")
|
|
MODEL_SECTION_KEYS = ("default", "model", "model_name")
|
|
MODEL_LIST_KEYS = ("models",)
|
|
|
|
|
|
def _iter_model_values(node, path=""):
|
|
"""Yield (path, value) for every model-name-bearing scalar in a config."""
|
|
if isinstance(node, dict):
|
|
for key, value in node.items():
|
|
child = f"{path}.{key}" if path else key
|
|
if key in MODEL_KEYS:
|
|
if isinstance(value, dict):
|
|
for subkey in MODEL_SECTION_KEYS:
|
|
subvalue = value.get(subkey)
|
|
if isinstance(subvalue, str):
|
|
yield (f"{child}.{subkey}", subvalue)
|
|
for subkey, subvalue in value.items():
|
|
if isinstance(subvalue, (dict, list)):
|
|
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
|
|
elif isinstance(value, list):
|
|
yield from _iter_model_values(value, child)
|
|
else:
|
|
yield (child, value)
|
|
elif key in MODEL_LIST_KEYS:
|
|
yield from _iter_model_list(value, child)
|
|
elif isinstance(value, (dict, list)):
|
|
yield from _iter_model_values(value, child)
|
|
elif isinstance(node, list):
|
|
for i, item in enumerate(node):
|
|
yield from _iter_model_values(item, f"{path}[{i}]")
|
|
|
|
|
|
def _iter_model_list(node, path):
|
|
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
|
|
if isinstance(node, list):
|
|
for i, item in enumerate(node):
|
|
yield from _iter_model_list(item, f"{path}[{i}]")
|
|
elif isinstance(node, dict):
|
|
for key, value in node.items():
|
|
if key in MODEL_KEYS and isinstance(value, str):
|
|
yield (f"{path}.{key}", value)
|
|
elif isinstance(value, (dict, list)):
|
|
yield from _iter_model_list(value, f"{path}.{key}")
|
|
else:
|
|
yield (path, node)
|
|
|
|
|
|
def audit(path):
|
|
with open(path) as f:
|
|
cfg = yaml.safe_load(f)
|
|
|
|
model = cfg.get("model", {})
|
|
fb = cfg.get("fallback_providers", {})
|
|
comp = cfg.get("compression", {})
|
|
aux = cfg.get("auxiliary", {})
|
|
deleg = cfg.get("delegation", {})
|
|
cps = cfg.get("custom_providers", [])
|
|
cp = cps[0] if cps else {}
|
|
|
|
# --- Rule 3: API Keys via Environment ---
|
|
check(
|
|
model.get("api_key") in ("", None),
|
|
"Rule 3",
|
|
f"model.api_key must be empty (got {model.get('api_key')!r}) — keys via env var, not hardcoded",
|
|
)
|
|
check(
|
|
model.get("api_key_env") == "LITELLM_API_KEY",
|
|
"Rule 3",
|
|
f"model.api_key_env must be LITELLM_API_KEY (got {model.get('api_key_env')!r})",
|
|
)
|
|
|
|
# --- Rule 5: Main Config Base URL ---
|
|
expected_base = "http://192.168.68.116/v1"
|
|
check(
|
|
model.get("base_url") == expected_base,
|
|
"Rule 5",
|
|
f"model.base_url must be {expected_base} (got {model.get('base_url')!r}) — /v1 not /litellm/v1",
|
|
)
|
|
|
|
# --- Rule 6: max_tokens Is Required ---
|
|
check(
|
|
isinstance(model.get("max_tokens"), int) and model.get("max_tokens") <= 8192,
|
|
"Rule 6",
|
|
f"model.max_tokens must be set and <= 8192 (got {model.get('max_tokens')!r}) — thermal safety",
|
|
)
|
|
|
|
# --- Rule 7: Auxiliary Model Consistency ---
|
|
check(
|
|
comp.get("model") == "syslog-auto",
|
|
"Rule 7",
|
|
f"compression.model must be syslog-auto (got {comp.get('model')!r}) — auto-routing to prevent Strix Halo overload",
|
|
)
|
|
aux_comp = aux.get("compression", {})
|
|
check(
|
|
aux_comp.get("model") == "syslog-auto",
|
|
"Rule 7",
|
|
f"auxiliary.compression.model must be syslog-auto (got {aux_comp.get('model')!r}) — must match compression.model",
|
|
)
|
|
|
|
# --- Rule 8: GPU Workload Distribution ---
|
|
# gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision.
|
|
check(
|
|
aux.get("vision", {}).get("model") == "gpu-vision",
|
|
"Rule 8",
|
|
f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
|
)
|
|
check(
|
|
aux.get("web_extract", {}).get("model") == "gpu-vision",
|
|
"Rule 8",
|
|
f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
|
)
|
|
|
|
# --- Rule 9: Compression Threshold ---
|
|
check(
|
|
comp.get("threshold") == 0.65,
|
|
"Rule 9",
|
|
f"compression.threshold must be 0.65 for 128K models (got {comp.get('threshold')!r})",
|
|
)
|
|
check(
|
|
comp.get("max_context_window") == 131072,
|
|
"Rule 9",
|
|
f"compression.max_context_window must be 131072 (got {comp.get('max_context_window')!r}) — matches 128K GPU capacity",
|
|
)
|
|
|
|
# --- Rule 10: Default Model Must Be syslog-auto ---
|
|
check(
|
|
model.get("default") == "syslog-auto",
|
|
"Rule 10",
|
|
f"model.default must be syslog-auto (got {model.get('default')!r}) — auto-routing default",
|
|
)
|
|
|
|
# --- Rule 14: Provider Name Must Match custom_providers Name ---
|
|
check(
|
|
model.get("provider") == "harness",
|
|
"Rule 14",
|
|
f"model.provider must be 'harness' (got {model.get('provider')!r}) — NOT 'custom'. "
|
|
f"provider: custom causes generic resolution path that ignores key_env → 'no-key-required' → 401",
|
|
)
|
|
check(
|
|
comp.get("provider") == "harness",
|
|
"Rule 14",
|
|
f"compression.provider must be 'harness' (got {comp.get('provider')!r})",
|
|
)
|
|
for aux_name in ("vision", "web_extract", "compression"):
|
|
aux_provider = aux.get(aux_name, {}).get("provider")
|
|
check(
|
|
aux_provider == "harness",
|
|
"Rule 14",
|
|
f"auxiliary.{aux_name}.provider must be 'harness' (got {aux_provider!r})",
|
|
)
|
|
check(
|
|
deleg.get("provider") == "harness",
|
|
"Rule 14",
|
|
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
|
)
|
|
check(
|
|
fb.get("provider") == "deepseek",
|
|
"Rule 14",
|
|
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
|
f"true fallback diversity, not same endpoint as primary",
|
|
)
|
|
check(
|
|
fb.get("model") == "deepseek-v4-flash",
|
|
"Rule 14",
|
|
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
|
)
|
|
check(
|
|
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
|
"Rule 14",
|
|
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
|
)
|
|
|
|
# --- custom_providers sanity ---
|
|
check(
|
|
cp.get("name") == "harness",
|
|
"custom_providers",
|
|
f"custom_providers[0].name must be 'harness' (got {cp.get('name')!r})",
|
|
)
|
|
check(
|
|
cp.get("key_env") == "LITELLM_API_KEY" or cp.get("api_key_env") == "LITELLM_API_KEY",
|
|
"custom_providers",
|
|
f"custom_providers[0] must have key_env or api_key_env = LITELLM_API_KEY "
|
|
f"(got key_env={cp.get('key_env')!r}, api_key_env={cp.get('api_key_env')!r})",
|
|
)
|
|
check(
|
|
cp.get("base_url", "").endswith("/v1"),
|
|
"custom_providers",
|
|
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
|
)
|
|
|
|
# --- Retired/raw model names (Rule 7/8 spirit) ---
|
|
# The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
|
|
# NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
|
|
# gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
|
|
# crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
|
|
# RESOLVING names (verified 200) are discouraged but working, so they only WARN:
|
|
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
|
|
# Failing a working alias would reject valid configs - the exact defect this change fixes.
|
|
non_resolving = {
|
|
"gpu-light": "gpu-vision",
|
|
"gemma-4-12b": "gpu-vision",
|
|
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
|
|
"ornith-1.0-35b": "strix-moe",
|
|
"qwen3.6-27B-code": "gpu-dense",
|
|
"qwen3.6-35B-udq4": "strix-moe",
|
|
}
|
|
raw_but_live = {}
|
|
for field_path, value in _iter_model_values(cfg):
|
|
if value in non_resolving:
|
|
check(
|
|
False,
|
|
"Rule 7/8",
|
|
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
|
|
)
|
|
elif value in raw_but_live:
|
|
warn(
|
|
"Rule 7/8",
|
|
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
|
|
)
|
|
|
|
# --- Report ---
|
|
print(f"{'=' * 60}")
|
|
print(f"Hermes Config Audit: {path}")
|
|
print(f"{'=' * 60}")
|
|
print(f"\n✅ PASSED ({len(PASSES)}):")
|
|
for p in PASSES:
|
|
print(f" ✅ {p}")
|
|
|
|
if WARNINGS:
|
|
print(f"\n⚠️ WARNINGS ({len(WARNINGS)}):")
|
|
for w in WARNINGS:
|
|
print(f" ⚠️ {w}")
|
|
|
|
if VIOLATIONS:
|
|
print(f"\n❌ VIOLATIONS ({len(VIOLATIONS)}):")
|
|
for v in VIOLATIONS:
|
|
print(f" ❌ {v}")
|
|
print(f"\n{'=' * 60}")
|
|
print(f"RESULT: FAIL — {len(VIOLATIONS)} violation(s) must be fixed")
|
|
print(f"{'=' * 60}")
|
|
return 1
|
|
else:
|
|
print(f"\n{'=' * 60}")
|
|
print(f"RESULT: PASS — all contract rules satisfied")
|
|
print(f"{'=' * 60}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
if len(sys.argv) < 2:
|
|
print("Usage: python3 audit-hermes-config.py <config.yaml>")
|
|
sys.exit(2)
|
|
sys.exit(audit(sys.argv[1]))
|