PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 7s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 17s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 0s
The audit assumed fallback_providers was always a dict (single provider).
Two live agents (koby, koonimo) carry it as a LIST of dicts (one entry per
fallback), so the script crashed with:
File "audit-hermes-config.py", line 211, in audit
fb.get("provider") == "deepseek",
AttributeError: 'list' object has no attribute 'get'
Both are REAL agent configs, so this is not a malformed-input case — the
script simply could not audit two of the four agents it exists to audit.
Fix:
- Normalize fallback_providers to a list of entries (dict → [dict], list → list)
- Apply the existing checks to each entry
- A malformed entry (not a mapping) produces a reported VIOLATION naming the
offending entry, NOT an uncaught exception
Adds regression test using the real failing shape (list of dicts) and proves
it bites against the pre-fix revision.
Real audit results after fix:
- mumuni: FAIL — 7 violations
- tanko: FAIL — 21 violations
- koby: FAIL — 16 violations (previously crashed)
- koonimo: FAIL — 10 violations (previously crashed)
No agent configs were changed. No existing rules were relaxed.
365 lines
14 KiB
Python
365 lines
14 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Hermes Config Audit — validates a live config.yaml against the prose contract rules.
|
|
|
|
Usage:
|
|
python3 audit-hermes-config.py <config.yaml>
|
|
python3 audit-hermes-config.py /root/.hermes/config.yaml
|
|
|
|
Exit codes:
|
|
0 = all checks pass
|
|
1 = one or more contract violations found
|
|
|
|
This script encodes every rule from hermes-config-template.prose.md so config
|
|
changes can be verified before and after application. It is the single automated
|
|
enforcement layer for the prose contract.
|
|
|
|
Contract: /root/prose-contracts/hermes-config-template.prose.md
|
|
"""
|
|
|
|
import sys
|
|
import yaml
|
|
|
|
VIOLATIONS = []
|
|
WARNINGS = []
|
|
PASSES = []
|
|
|
|
|
|
def check(condition, rule, message):
|
|
if condition:
|
|
PASSES.append(f"[{rule}] {message}")
|
|
else:
|
|
VIOLATIONS.append(f"[{rule}] {message}")
|
|
|
|
|
|
def warn(rule, message):
|
|
WARNINGS.append(f"[{rule}] {message}")
|
|
|
|
|
|
# Derivation rule: a model name is any scalar under a mapping key named `model` or
|
|
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
|
|
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
|
|
# specially. The only other exception is key `models` (litellm key-generation params carry a
|
|
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
|
|
# hand.
|
|
MODEL_KEYS = ("model", "model_name")
|
|
MODEL_SECTION_KEYS = ("default", "model", "model_name")
|
|
MODEL_LIST_KEYS = ("models",)
|
|
|
|
|
|
def _iter_model_values(node, path=""):
|
|
"""Yield (path, value) for every model-name-bearing scalar in a config."""
|
|
if isinstance(node, dict):
|
|
for key, value in node.items():
|
|
child = f"{path}.{key}" if path else key
|
|
if key in MODEL_KEYS:
|
|
if isinstance(value, dict):
|
|
for subkey in MODEL_SECTION_KEYS:
|
|
subvalue = value.get(subkey)
|
|
if isinstance(subvalue, str):
|
|
yield (f"{child}.{subkey}", subvalue)
|
|
for subkey, subvalue in value.items():
|
|
if isinstance(subvalue, (dict, list)):
|
|
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
|
|
elif isinstance(value, list):
|
|
yield from _iter_model_values(value, child)
|
|
else:
|
|
yield (child, value)
|
|
elif key in MODEL_LIST_KEYS:
|
|
yield from _iter_model_list(value, child)
|
|
elif isinstance(value, (dict, list)):
|
|
yield from _iter_model_values(value, child)
|
|
elif isinstance(node, list):
|
|
for i, item in enumerate(node):
|
|
yield from _iter_model_values(item, f"{path}[{i}]")
|
|
|
|
|
|
def _iter_model_list(node, path):
|
|
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
|
|
if isinstance(node, list):
|
|
for i, item in enumerate(node):
|
|
yield from _iter_model_list(item, f"{path}[{i}]")
|
|
elif isinstance(node, dict):
|
|
for key, value in node.items():
|
|
if key in MODEL_KEYS and isinstance(value, str):
|
|
yield (f"{path}.{key}", value)
|
|
elif isinstance(value, (dict, list)):
|
|
yield from _iter_model_list(value, f"{path}.{key}")
|
|
else:
|
|
yield (path, node)
|
|
|
|
|
|
def audit(path):
|
|
with open(path) as f:
|
|
cfg = yaml.safe_load(f)
|
|
|
|
model = cfg.get("model", {})
|
|
fb_raw = cfg.get("fallback_providers", {})
|
|
# Normalize: fallback_providers may be a dict (single provider) or a list of dicts
|
|
# (one entry per fallback). Both shapes are valid; we must handle both without crashing.
|
|
if isinstance(fb_raw, dict):
|
|
fb_entries = [fb_raw]
|
|
elif isinstance(fb_raw, list):
|
|
fb_entries = fb_raw
|
|
else:
|
|
fb_entries = [fb_raw] # Let it fail the check below as malformed
|
|
fb = fb_entries[0] if fb_entries else {}
|
|
comp = cfg.get("compression", {})
|
|
aux = cfg.get("auxiliary", {})
|
|
deleg = cfg.get("delegation", {})
|
|
cps = cfg.get("custom_providers", [])
|
|
cp = cps[0] if cps else {}
|
|
|
|
# --- Rule 3: API Keys via Environment ---
|
|
check(
|
|
model.get("api_key") in ("", None),
|
|
"Rule 3",
|
|
f"model.api_key must be empty (got {model.get('api_key')!r}) — keys via env var, not hardcoded",
|
|
)
|
|
check(
|
|
model.get("api_key_env") == "LITELLM_API_KEY",
|
|
"Rule 3",
|
|
f"model.api_key_env must be LITELLM_API_KEY (got {model.get('api_key_env')!r})",
|
|
)
|
|
|
|
# --- Rule 5: Main Config Base URL ---
|
|
# Canonical internal base (hermes-key-enforcement.prose.md:19/38/57/84/91/96)
|
|
# and public base (serves /v1 only, per 2026-09-19 probe from CT 116).
|
|
# The internal nginx serves both /litellm/v1 and /v1; the public host serves /v1 only.
|
|
# FAIL anything else (do not widen to accept any path ending in /v1).
|
|
# Internal /v1 is non-canonical but working (authenticated via nginx), so WARN not FAIL.
|
|
canonical_internal = "http://192.168.68.116/litellm/v1"
|
|
public_host = "https://litellm.sysloggh.net/v1"
|
|
non_canonical_internal = "http://192.168.68.116/v1"
|
|
allowed_bases = (canonical_internal, public_host)
|
|
actual_base = model.get("base_url")
|
|
if actual_base in allowed_bases:
|
|
check(True, "Rule 5", f"model.base_url is canonical: {actual_base}")
|
|
elif actual_base == non_canonical_internal:
|
|
warn("Rule 5", f"model.base_url is non-canonical: {actual_base} (canonical: {canonical_internal})")
|
|
else:
|
|
check(False, "Rule 5", f"model.base_url must be one of {allowed_bases} (got {actual_base!r})")
|
|
|
|
# --- Rule 6: max_tokens Is Required ---
|
|
check(
|
|
isinstance(model.get("max_tokens"), int) and model.get("max_tokens") <= 8192,
|
|
"Rule 6",
|
|
f"model.max_tokens must be set and <= 8192 (got {model.get('max_tokens')!r}) — thermal safety",
|
|
)
|
|
|
|
# --- Rule 7: Auxiliary Model Consistency ---
|
|
check(
|
|
comp.get("model") == "syslog-auto",
|
|
"Rule 7",
|
|
f"compression.model must be syslog-auto (got {comp.get('model')!r}) — auto-routing to prevent Strix Halo overload",
|
|
)
|
|
aux_comp = aux.get("compression", {})
|
|
check(
|
|
aux_comp.get("model") == "syslog-auto",
|
|
"Rule 7",
|
|
f"auxiliary.compression.model must be syslog-auto (got {aux_comp.get('model')!r}) — must match compression.model",
|
|
)
|
|
|
|
# --- Rule 8: GPU Workload Distribution ---
|
|
# gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision.
|
|
check(
|
|
aux.get("vision", {}).get("model") == "gpu-vision",
|
|
"Rule 8",
|
|
f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
|
)
|
|
check(
|
|
aux.get("web_extract", {}).get("model") == "gpu-vision",
|
|
"Rule 8",
|
|
f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
|
)
|
|
|
|
# --- Rule 9: Compression Threshold ---
|
|
check(
|
|
comp.get("threshold") == 0.65,
|
|
"Rule 9",
|
|
f"compression.threshold must be 0.65 for 128K models (got {comp.get('threshold')!r})",
|
|
)
|
|
check(
|
|
comp.get("max_context_window") == 131072,
|
|
"Rule 9",
|
|
f"compression.max_context_window must be 131072 (got {comp.get('max_context_window')!r}) — syslog-auto pool floor (NVIDIA hosts 128K; Strix Halo 256K)",
|
|
)
|
|
|
|
# --- Rule 10: Default Model Must Be syslog-auto ---
|
|
check(
|
|
model.get("default") == "syslog-auto",
|
|
"Rule 10",
|
|
f"model.default must be syslog-auto (got {model.get('default')!r}) — auto-routing default",
|
|
)
|
|
|
|
# --- Rule 14: Provider Name Must Match custom_providers Name ---
|
|
check(
|
|
model.get("provider") == "harness",
|
|
"Rule 14",
|
|
f"model.provider must be 'harness' (got {model.get('provider')!r}) — NOT 'custom'. "
|
|
f"provider: custom causes generic resolution path that ignores key_env → 'no-key-required' → 401",
|
|
)
|
|
check(
|
|
comp.get("provider") == "harness",
|
|
"Rule 14",
|
|
f"compression.provider must be 'harness' (got {comp.get('provider')!r})",
|
|
)
|
|
for aux_name in ("vision", "web_extract", "compression"):
|
|
aux_provider = aux.get(aux_name, {}).get("provider")
|
|
check(
|
|
aux_provider == "harness",
|
|
"Rule 14",
|
|
f"auxiliary.{aux_name}.provider must be 'harness' (got {aux_provider!r})",
|
|
)
|
|
check(
|
|
deleg.get("provider") == "harness",
|
|
"Rule 14",
|
|
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
|
)
|
|
# Check each fallback entry. A malformed entry (not a mapping) is a VIOLATION, not a crash.
|
|
for idx, entry in enumerate(fb_entries):
|
|
prefix = f"fallback_providers[{idx}]"
|
|
if not isinstance(entry, dict):
|
|
check(
|
|
False,
|
|
"Rule 14",
|
|
f"{prefix} must be a mapping (got {type(entry).__name__})",
|
|
)
|
|
continue
|
|
check(
|
|
entry.get("provider") == "deepseek",
|
|
"Rule 14",
|
|
f"{prefix}.provider must be 'deepseek' (got {entry.get('provider')!r}) — "
|
|
f"true fallback diversity, not same endpoint as primary",
|
|
)
|
|
check(
|
|
entry.get("model") == "deepseek-v4-flash",
|
|
"Rule 14",
|
|
f"{prefix}.model must be 'deepseek-v4-flash' (got {entry.get('model')!r})",
|
|
)
|
|
check(
|
|
entry.get("api_key_env") == "DEEPSEEK_API_KEY",
|
|
"Rule 14",
|
|
f"{prefix}.api_key_env must be DEEPSEEK_API_KEY (got {entry.get('api_key_env')!r})",
|
|
)
|
|
|
|
# --- custom_providers sanity ---
|
|
check(
|
|
cp.get("name") == "harness",
|
|
"custom_providers",
|
|
f"custom_providers[0].name must be 'harness' (got {cp.get('name')!r})",
|
|
)
|
|
check(
|
|
cp.get("key_env") == "LITELLM_API_KEY" or cp.get("api_key_env") == "LITELLM_API_KEY",
|
|
"custom_providers",
|
|
f"custom_providers[0] must have key_env or api_key_env = LITELLM_API_KEY "
|
|
f"(got key_env={cp.get('key_env')!r}, api_key_env={cp.get('api_key_env')!r})",
|
|
)
|
|
check(
|
|
cp.get("base_url", "").endswith("/v1"),
|
|
"custom_providers",
|
|
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
|
)
|
|
|
|
# --- Retired/raw model names (Rule 7/8 spirit) ---
|
|
# The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
|
|
# NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
|
|
# gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
|
|
# crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
|
|
# RESOLVING names (verified 200) are discouraged but working, so they only WARN:
|
|
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
|
|
# Failing a working alias would reject valid configs - the exact defect this change fixes.
|
|
non_resolving = {
|
|
"gpu-light": "gpu-vision",
|
|
"gemma-4-12b": "gpu-vision",
|
|
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
|
|
"ornith-1.0-35b": "strix-moe",
|
|
"qwen3.6-27B-code": "gpu-dense",
|
|
"qwen3.6-35B-udq4": "strix-moe",
|
|
}
|
|
raw_but_live = {}
|
|
for field_path, value in _iter_model_values(cfg):
|
|
if value in non_resolving:
|
|
check(
|
|
False,
|
|
"Rule 7/8",
|
|
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
|
|
)
|
|
elif value in raw_but_live:
|
|
warn(
|
|
"Rule 7/8",
|
|
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
|
|
)
|
|
|
|
# --- MCP Server Checks (Rule 15) ---
|
|
# Valid MCP server endpoints
|
|
VALID_MCP_ENDPOINTS = {
|
|
'ra-h-os': 'http://192.168.68.65:3100/mcp',
|
|
'litellm': 'https://litellm.sysloggh.net/mcp',
|
|
}
|
|
|
|
# Check MCP servers if they exist
|
|
mcp_servers = cfg.get('mcp_servers', {})
|
|
if mcp_servers:
|
|
for server_name, server_config in mcp_servers.items():
|
|
url = server_config.get('url', '')
|
|
|
|
# Check endpoint validity
|
|
if server_name in VALID_MCP_ENDPOINTS:
|
|
expected = VALID_MCP_ENDPOINTS[server_name]
|
|
if url == expected:
|
|
check(True, 'Rule 15', f'MCP server "{server_name}" URL is correct: {url}')
|
|
else:
|
|
check(False, 'Rule 15', f'MCP server "{server_name}" URL is incorrect: {url} (expected: {expected})')
|
|
else:
|
|
warn('Rule 15', f'MCP server "{server_name}" URL may need validation (not in known list): {url}')
|
|
|
|
# Check for proper authentication
|
|
headers = server_config.get('headers', {})
|
|
has_auth = False
|
|
for key, value in headers.items():
|
|
if 'key' in key.lower() or 'auth' in key.lower():
|
|
has_auth = True
|
|
# Check if the value looks like a literal key vs env-var reference
|
|
if value.startswith('Bearer ') and value[7:].startswith('sk-'):
|
|
check(True, 'Rule 15', f'MCP server "{server_name}" has valid auth header: {key}')
|
|
else:
|
|
warn('Rule 15', f'MCP server "{server_name}" header may use env-var instead of literal key: {key} = {value}')
|
|
break
|
|
if not has_auth:
|
|
warn('Rule 15', f'MCP server "{server_name}" has no authentication header')
|
|
|
|
# --- Report ---
|
|
print(f"{'=' * 60}")
|
|
print(f"Hermes Config Audit: {path}")
|
|
print(f"{'=' * 60}")
|
|
print(f"\n✅ PASSED ({len(PASSES)}):")
|
|
for p in PASSES:
|
|
print(f" ✅ {p}")
|
|
|
|
if WARNINGS:
|
|
print(f"\n⚠️ WARNINGS ({len(WARNINGS)}):")
|
|
for w in WARNINGS:
|
|
print(f" ⚠️ {w}")
|
|
|
|
if VIOLATIONS:
|
|
print(f"\n❌ VIOLATIONS ({len(VIOLATIONS)}):")
|
|
for v in VIOLATIONS:
|
|
print(f" ❌ {v}")
|
|
print(f"\n{'=' * 60}")
|
|
print(f"RESULT: FAIL — {len(VIOLATIONS)} violation(s) must be fixed")
|
|
print(f"{'=' * 60}")
|
|
return 1
|
|
else:
|
|
print(f"\n{'=' * 60}")
|
|
print(f"RESULT: PASS — all contract rules satisfied")
|
|
print(f"{'=' * 60}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
if len(sys.argv) < 2:
|
|
print("Usage: python3 audit-hermes-config.py <config.yaml>")
|
|
sys.exit(2)
|
|
sys.exit(audit(sys.argv[1]))
|