Files
prose-contracts/tests/test_audit_hermes_config_alias.py
T

146 lines
4.7 KiB
Python

"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
observable behaviour — exit code and the emitted rule message — for the live alias and
for both retired names. No network, vault, or SSH access is required.
"""
from __future__ import annotations
import pathlib
import subprocess
import sys
ROOT = pathlib.Path(__file__).resolve().parent.parent
AUDIT = ROOT / "audit-hermes-config.py"
BASE = """
model:
api_key: ""
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/v1
max_tokens: 4096
default: syslog-auto
provider: harness
fallback_providers:
provider: deepseek
model: deepseek-v4-flash
api_key_env: DEEPSEEK_API_KEY
compression:
model: syslog-auto
provider: harness
threshold: 0.65
max_context_window: 131072
auxiliary:
vision:
model: {alias}
provider: harness
web_extract:
model: {alias}
provider: harness
compression:
model: syslog-auto
provider: harness
delegation:
provider: harness
custom_providers:
- name: harness
key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/v1
"""
def _run_config(tmp_path, name, text):
cfg = tmp_path / name
cfg.write_text(text)
proc = subprocess.run(
[sys.executable, str(AUDIT), str(cfg)],
capture_output=True, text=True,
)
return proc.returncode, proc.stdout
def _run(tmp_path, alias):
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
def test_live_canonical_alias_passes(tmp_path):
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "RESULT: PASS" in out
def test_retired_gpu_light_is_rejected(tmp_path):
"""A config pinned to the retired alias must fail, not pass."""
code, out = _run(tmp_path, "gpu-light")
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_retired_gemma_is_rejected(tmp_path):
"""The retired raw model name must fail Rule 8 as well."""
code, out = _run(tmp_path, "gemma-4-12b")
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_corrected_compression_example_passes(tmp_path):
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "RESULT: PASS" in out
def test_retired_alias_in_delegation_is_rejected(tmp_path):
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
code, out = _run_config(
tmp_path,
"delegation-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: gpu-light",
),
)
assert code == 1, out
assert "delegation.model = 'gpu-light' is retired" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"custom-provider-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" - name: harness\n key_env: LITELLM_API_KEY",
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
),
)
assert code == 1, out
assert "custom_providers[0].model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_raw_alias_is_rejected(tmp_path):
"""Raw-but-live model names must fail too, pointing at the stable alias."""
code, out = _run_config(
tmp_path,
"raw-qwen.yaml",
BASE.format(alias="gpu-vision").replace("default: syslog-auto", "default: qwen3.6-27B-code"),
)
assert code == 1, out
assert "model.default = 'qwen3.6-27B-code'" in out
assert "gpu-dense" in out
assert "RESULT: FAIL" in out