#!/usr/bin/env python3 """ Hermes Config Audit — validates a live config.yaml against the prose contract rules. Usage: python3 audit-hermes-config.py python3 audit-hermes-config.py /root/.hermes/config.yaml Exit codes: 0 = all checks pass 1 = one or more contract violations found This script encodes every rule from hermes-config-template.prose.md so config changes can be verified before and after application. It is the single automated enforcement layer for the prose contract. Contract: /root/prose-contracts/hermes-config-template.prose.md """ import sys import yaml VIOLATIONS = [] WARNINGS = [] PASSES = [] def check(condition, rule, message): if condition: PASSES.append(f"[{rule}] {message}") else: VIOLATIONS.append(f"[{rule}] {message}") def warn(rule, message): WARNINGS.append(f"[{rule}] {message}") # Derivation rule: a model name is any scalar under a mapping key named `model` or # `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model # name lives under `default`/`model`/`model_name` inside that section, so it is descended # specially. The only other exception is key `models` (litellm key-generation params carry a # list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by # hand. MODEL_KEYS = ("model", "model_name") MODEL_SECTION_KEYS = ("default", "model", "model_name") MODEL_LIST_KEYS = ("models",) def _iter_model_values(node, path=""): """Yield (path, value) for every model-name-bearing scalar in a config.""" if isinstance(node, dict): for key, value in node.items(): child = f"{path}.{key}" if path else key if key in MODEL_KEYS: if isinstance(value, dict): for subkey in MODEL_SECTION_KEYS: subvalue = value.get(subkey) if isinstance(subvalue, str): yield (f"{child}.{subkey}", subvalue) for subkey, subvalue in value.items(): if isinstance(subvalue, (dict, list)): yield from _iter_model_values(subvalue, f"{child}.{subkey}") elif isinstance(value, list): yield from _iter_model_values(value, child) else: yield (child, value) elif key in MODEL_LIST_KEYS: yield from _iter_model_list(value, child) elif isinstance(value, (dict, list)): yield from _iter_model_values(value, child) elif isinstance(node, list): for i, item in enumerate(node): yield from _iter_model_values(item, f"{path}[{i}]") def _iter_model_list(node, path): """Yield scalars under an allowlisted `models` key (list of names or list of dicts).""" if isinstance(node, list): for i, item in enumerate(node): yield from _iter_model_list(item, f"{path}[{i}]") elif isinstance(node, dict): for key, value in node.items(): if key in MODEL_KEYS and isinstance(value, str): yield (f"{path}.{key}", value) elif isinstance(value, (dict, list)): yield from _iter_model_list(value, f"{path}.{key}") else: yield (path, node) def audit(path): with open(path) as f: cfg = yaml.safe_load(f) model = cfg.get("model", {}) fb = cfg.get("fallback_providers", {}) comp = cfg.get("compression", {}) aux = cfg.get("auxiliary", {}) deleg = cfg.get("delegation", {}) cps = cfg.get("custom_providers", []) cp = cps[0] if cps else {} # --- Rule 3: API Keys via Environment --- check( model.get("api_key") in ("", None), "Rule 3", f"model.api_key must be empty (got {model.get('api_key')!r}) — keys via env var, not hardcoded", ) check( model.get("api_key_env") == "LITELLM_API_KEY", "Rule 3", f"model.api_key_env must be LITELLM_API_KEY (got {model.get('api_key_env')!r})", ) # --- Rule 5: Main Config Base URL --- expected_base = "http://192.168.68.116/v1" check( model.get("base_url") == expected_base, "Rule 5", f"model.base_url must be {expected_base} (got {model.get('base_url')!r}) — /v1 not /litellm/v1", ) # --- Rule 6: max_tokens Is Required --- check( isinstance(model.get("max_tokens"), int) and model.get("max_tokens") <= 8192, "Rule 6", f"model.max_tokens must be set and <= 8192 (got {model.get('max_tokens')!r}) — thermal safety", ) # --- Rule 7: Auxiliary Model Consistency --- check( comp.get("model") == "syslog-auto", "Rule 7", f"compression.model must be syslog-auto (got {comp.get('model')!r}) — auto-routing to prevent Strix Halo overload", ) aux_comp = aux.get("compression", {}) check( aux_comp.get("model") == "syslog-auto", "Rule 7", f"auxiliary.compression.model must be syslog-auto (got {aux_comp.get('model')!r}) — must match compression.model", ) # --- Rule 8: GPU Workload Distribution --- # gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision. check( aux.get("vision", {}).get("model") == "gpu-vision", "Rule 8", f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias", ) check( aux.get("web_extract", {}).get("model") == "gpu-vision", "Rule 8", f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias", ) # --- Rule 9: Compression Threshold --- check( comp.get("threshold") == 0.65, "Rule 9", f"compression.threshold must be 0.65 for 128K models (got {comp.get('threshold')!r})", ) check( comp.get("max_context_window") == 131072, "Rule 9", f"compression.max_context_window must be 131072 (got {comp.get('max_context_window')!r}) — syslog-auto pool floor (NVIDIA hosts 128K; Strix Halo 256K)", ) # --- Rule 10: Default Model Must Be syslog-auto --- check( model.get("default") == "syslog-auto", "Rule 10", f"model.default must be syslog-auto (got {model.get('default')!r}) — auto-routing default", ) # --- Rule 14: Provider Name Must Match custom_providers Name --- check( model.get("provider") == "harness", "Rule 14", f"model.provider must be 'harness' (got {model.get('provider')!r}) — NOT 'custom'. " f"provider: custom causes generic resolution path that ignores key_env → 'no-key-required' → 401", ) check( comp.get("provider") == "harness", "Rule 14", f"compression.provider must be 'harness' (got {comp.get('provider')!r})", ) for aux_name in ("vision", "web_extract", "compression"): aux_provider = aux.get(aux_name, {}).get("provider") check( aux_provider == "harness", "Rule 14", f"auxiliary.{aux_name}.provider must be 'harness' (got {aux_provider!r})", ) check( deleg.get("provider") == "harness", "Rule 14", f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})", ) check( fb.get("provider") == "deepseek", "Rule 14", f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — " f"true fallback diversity, not same endpoint as primary", ) check( fb.get("model") == "deepseek-v4-flash", "Rule 14", f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})", ) check( fb.get("api_key_env") == "DEEPSEEK_API_KEY", "Rule 14", f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})", ) # --- custom_providers sanity --- check( cp.get("name") == "harness", "custom_providers", f"custom_providers[0].name must be 'harness' (got {cp.get('name')!r})", ) check( cp.get("key_env") == "LITELLM_API_KEY" or cp.get("api_key_env") == "LITELLM_API_KEY", "custom_providers", f"custom_providers[0] must have key_env or api_key_env = LITELLM_API_KEY " f"(got key_env={cp.get('key_env')!r}, api_key_env={cp.get('api_key_env')!r})", ) check( cp.get("base_url", "").endswith("/v1"), "custom_providers", f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", ) # --- Retired/raw model names (Rule 7/8 spirit) --- # The audit's job is to catch configs that are BROKEN, not to enforce a style preference. # NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL: # gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision # crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe # RESOLVING names (verified 200) are discouraged but working, so they only WARN: # qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe # Failing a working alias would reject valid configs - the exact defect this change fixes. non_resolving = { "gpu-light": "gpu-vision", "gemma-4-12b": "gpu-vision", "crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)", "ornith-1.0-35b": "strix-moe", "qwen3.6-27B-code": "gpu-dense", "qwen3.6-35B-udq4": "strix-moe", } raw_but_live = {} for field_path, value in _iter_model_values(cfg): if value in non_resolving: check( False, "Rule 7/8", f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}", ) elif value in raw_but_live: warn( "Rule 7/8", f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}", ) # --- MCP Server Checks (Rule 15) --- # Valid MCP server endpoints VALID_MCP_ENDPOINTS = { 'ra-h-os': 'http://192.168.68.65:3100/mcp', 'litellm': 'https://litellm.sysloggh.net/mcp', } # Check MCP servers if they exist mcp_servers = cfg.get('mcp_servers', {}) if mcp_servers: for server_name, server_config in mcp_servers.items(): url = server_config.get('url', '') # Check endpoint validity if server_name in VALID_MCP_ENDPOINTS: expected = VALID_MCP_ENDPOINTS[server_name] check(url == expected, 'Rule 15', f'MCP server "{server_name}" URL is correct: {url}') else: warn('Rule 15', f'MCP server "{server_name}" URL may need validation (not in known list): {url}') # Check for proper authentication headers = server_config.get('headers', {}) has_auth = False for key, value in headers.items(): if 'key' in key.lower() or 'auth' in key.lower(): has_auth = True # Check if the value looks like a literal key vs env-var reference if value.startswith('Bearer ') and value[7:].startswith('sk-'): check(True, 'Rule 15', f'MCP server "{server_name}" has valid auth header: {key}') else: warn('Rule 15', f'MCP server "{server_name}" header may use env-var instead of literal key: {key} = {value}') break if not has_auth: warn('Rule 15', f'MCP server "{server_name}" has no authentication header') # --- Report --- print(f"{'=' * 60}") print(f"Hermes Config Audit: {path}") print(f"{'=' * 60}") print(f"\n✅ PASSED ({len(PASSES)}):") for p in PASSES: print(f" ✅ {p}") if WARNINGS: print(f"\n⚠️ WARNINGS ({len(WARNINGS)}):") for w in WARNINGS: print(f" ⚠️ {w}") if VIOLATIONS: print(f"\n❌ VIOLATIONS ({len(VIOLATIONS)}):") for v in VIOLATIONS: print(f" ❌ {v}") print(f"\n{'=' * 60}") print(f"RESULT: FAIL — {len(VIOLATIONS)} violation(s) must be fixed") print(f"{'=' * 60}") return 1 else: print(f"\n{'=' * 60}") print(f"RESULT: PASS — all contract rules satisfied") print(f"{'=' * 60}") return 0 if __name__ == "__main__": if len(sys.argv) < 2: print("Usage: python3 audit-hermes-config.py ") sys.exit(2) sys.exit(audit(sys.argv[1]))