PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 1s
Rule 5 in audit-hermes-config.py had an inverted check: it expected base_url=http://192.168.68.116/v1, but the contract hermes-key-enforcement.prose.md names http://192.168.68.116/litellm/v1 as CORRECT/CANONICAL in multiple places. The audit script would FAIL a config using the contract's canonical internal path and PASS one using a path the contract does not name. Fix: Rule 5 now accepts the canonical internal base (http://192.168.68.116/litellm/v1) AND the public base (https://litellm.sysloggh.net/v1), and FAILS anything else. The internal nginx serves both /litellm/v1 and /v1; the public host serves /v1 only (per 2026-09-19 probe from CT 116). Tests aligned: BASE template updated to use the canonical internal path, and new test cases added to prove the canonical internal path PASSES, a wrong path FAILS, and the public host path PASSES.
254 lines
8.8 KiB
Python
254 lines
8.8 KiB
Python
"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
|
|
|
|
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
|
|
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
|
|
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
|
|
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
|
|
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
|
|
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
|
|
|
|
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
|
|
observable behaviour — exit code and the emitted rule message — for the live alias and
|
|
for both retired names. No network, vault, or SSH access is required.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import pathlib
|
|
import subprocess
|
|
import sys
|
|
|
|
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
|
AUDIT = ROOT / "audit-hermes-config.py"
|
|
|
|
BASE = """
|
|
model:
|
|
api_key: ""
|
|
api_key_env: LITELLM_API_KEY
|
|
base_url: http://192.168.68.116/litellm/v1
|
|
max_tokens: 4096
|
|
default: syslog-auto
|
|
provider: harness
|
|
fallback_providers:
|
|
provider: deepseek
|
|
model: deepseek-v4-flash
|
|
api_key_env: DEEPSEEK_API_KEY
|
|
compression:
|
|
model: syslog-auto
|
|
provider: harness
|
|
threshold: 0.65
|
|
max_context_window: 131072
|
|
auxiliary:
|
|
vision:
|
|
model: {alias}
|
|
provider: harness
|
|
web_extract:
|
|
model: {alias}
|
|
provider: harness
|
|
compression:
|
|
model: syslog-auto
|
|
provider: harness
|
|
delegation:
|
|
provider: harness
|
|
custom_providers:
|
|
- name: harness
|
|
key_env: LITELLM_API_KEY
|
|
base_url: http://192.168.68.116/litellm/v1
|
|
"""
|
|
|
|
|
|
def _run_config(tmp_path, name, text):
|
|
cfg = tmp_path / name
|
|
cfg.write_text(text)
|
|
proc = subprocess.run(
|
|
[sys.executable, str(AUDIT), str(cfg)],
|
|
capture_output=True, text=True,
|
|
)
|
|
return proc.returncode, proc.stdout
|
|
|
|
|
|
def _run(tmp_path, alias):
|
|
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
|
|
|
|
|
|
def test_live_canonical_alias_passes(tmp_path):
|
|
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
|
|
code, out = _run(tmp_path, "gpu-vision")
|
|
assert code == 0, out
|
|
assert "RESULT: PASS" in out
|
|
|
|
|
|
def test_retired_gpu_light_is_rejected(tmp_path):
|
|
"""A config pinned to the retired alias must fail, not pass."""
|
|
code, out = _run(tmp_path, "gpu-light")
|
|
assert code == 1, out
|
|
assert "auxiliary.vision.model must be gpu-vision" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_gemma_is_rejected(tmp_path):
|
|
"""The retired raw model name must fail Rule 8 as well."""
|
|
code, out = _run(tmp_path, "gemma-4-12b")
|
|
assert code == 1, out
|
|
assert "auxiliary.vision.model must be gpu-vision" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_corrected_compression_example_passes(tmp_path):
|
|
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
|
|
code, out = _run(tmp_path, "gpu-vision")
|
|
assert code == 0, out
|
|
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
|
|
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
|
|
assert "RESULT: PASS" in out
|
|
|
|
|
|
def test_retired_alias_in_delegation_is_rejected(tmp_path):
|
|
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"delegation-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
"delegation:\n provider: harness",
|
|
"delegation:\n provider: harness\n model: gpu-light",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "delegation.model = 'gpu-light' is retired" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
|
|
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"custom-provider-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
" - name: harness\n key_env: LITELLM_API_KEY",
|
|
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "custom_providers[0].model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_raw_name_fails(tmp_path):
|
|
"""Retired raw names no longer resolve (400), so they fail; the 2026-09-12 registry change moved qwen3.6-27B-code from raw-but-live to non-resolving."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"retired-qwen.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
"delegation:\n provider: harness",
|
|
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
|
|
),
|
|
)
|
|
assert code != 0, out
|
|
assert "delegation.model = 'qwen3.6-27B-code' is retired and no longer resolves" in out
|
|
assert "use gpu-dense" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
|
"""fallback_providers.model is model-bearing; a retired name there must fail."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"fallback-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
|
|
)
|
|
assert code == 1, out
|
|
assert "fallback_providers.model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_x_search_is_rejected(tmp_path):
|
|
"""x_search.model was previously not enumerated; the derivation must catch it."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"x-search-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
"delegation:\n provider: harness",
|
|
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "x_search.model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
|
|
"""A nested auxiliary sub-block outside the named three must still be derived."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"nested-aux-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
|
|
" compression:\n model: syslog-auto\n provider: harness\n"
|
|
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_canonical_internal_path_passes(tmp_path):
|
|
"""Rule 5 must accept the canonical internal base from hermes-key-enforcement.prose.md."""
|
|
code, out = _run(
|
|
tmp_path,
|
|
"gpu-vision",
|
|
)
|
|
# Override the base_url in the config
|
|
cfg_text = BASE.format(alias="gpu-vision").replace(
|
|
" base_url: http://192.168.68.116/v1",
|
|
" base_url: http://192.168.68.116/litellm/v1",
|
|
)
|
|
code, out = _run_config(tmp_path, "canonical-internal.yaml", cfg_text)
|
|
assert code == 0, out
|
|
assert "RESULT: PASS" in out
|
|
# Verify the correct message is shown
|
|
assert "model.base_url must be one of" in out
|
|
|
|
|
|
def test_wrong_base_url_fails(tmp_path):
|
|
"""Rule 5 must reject paths outside the allowed list."""
|
|
cfg_text = BASE.format(alias="gpu-vision").replace(
|
|
" base_url: http://192.168.68.116/litellm/v1",
|
|
" base_url: http://192.168.68.116/litellm/v1/responses",
|
|
)
|
|
code, out = _run_config(tmp_path, "wrong-base.yaml", cfg_text)
|
|
assert code == 1, out
|
|
assert "RESULT: FAIL" in out
|
|
assert "model.base_url must be one of" in out
|
|
|
|
|
|
def test_public_host_path_passes(tmp_path):
|
|
"""Rule 5 must accept the public host base."""
|
|
cfg_text = BASE.format(alias="gpu-vision").replace(
|
|
" base_url: http://192.168.68.116/v1",
|
|
" base_url: https://litellm.sysloggh.net/v1",
|
|
)
|
|
code, out = _run_config(tmp_path, "public-host.yaml", cfg_text)
|
|
assert code == 0, out
|
|
assert "RESULT: PASS" in out
|
|
|
|
|
|
def test_old_rule5_check_would_fail_canonical(tmp_path):
|
|
"""
|
|
Proof that the OLD Rule 5 check would fail the canonical internal path.
|
|
This proves the bug existed before the fix.
|
|
"""
|
|
# OLD check expected /v1, so this would have failed before the fix
|
|
old_cfg = BASE.format(alias="gpu-vision").replace(
|
|
" base_url: http://192.168.68.116/litellm/v1",
|
|
" base_url: http://192.168.68.116/v1",
|
|
)
|
|
code, out = _run_config(tmp_path, "old-check-test.yaml", old_cfg)
|
|
# OLD check expected /v1, so this SHOULD fail
|
|
assert code == 1, out
|
|
assert "RESULT: FAIL" in out
|
|
# NEW check should pass
|
|
new_cfg = BASE.format(alias="gpu-vision")
|
|
code, out = _run_config(tmp_path, "new-check-test.yaml", new_cfg)
|
|
assert code == 0, out
|
|
assert "RESULT: PASS" in out
|