PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 1s
- Remove retired names (qwen3.6-27B-code, qwen3.6-35B-udq4) from live alias claims - Update Strix Halo model to Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf (strix-moe, 256K ctx) - Fix litellm-health step 7 probe to gpu-vision (monitor key scoped) - Move qwen3.6-27B-code/35B-udq4 from raw-but-live to non-resolving in audit - Fold in pm2-self-heal: remove spoton-service (live PM2 set is 4/4) - Update hermes templates, key enforcement, timeout tables to live names
192 lines
6.6 KiB
Python
192 lines
6.6 KiB
Python
"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
|
|
|
|
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
|
|
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
|
|
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
|
|
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
|
|
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
|
|
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
|
|
|
|
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
|
|
observable behaviour — exit code and the emitted rule message — for the live alias and
|
|
for both retired names. No network, vault, or SSH access is required.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import pathlib
|
|
import subprocess
|
|
import sys
|
|
|
|
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
|
AUDIT = ROOT / "audit-hermes-config.py"
|
|
|
|
BASE = """
|
|
model:
|
|
api_key: ""
|
|
api_key_env: LITELLM_API_KEY
|
|
base_url: http://192.168.68.116/v1
|
|
max_tokens: 4096
|
|
default: syslog-auto
|
|
provider: harness
|
|
fallback_providers:
|
|
provider: deepseek
|
|
model: deepseek-v4-flash
|
|
api_key_env: DEEPSEEK_API_KEY
|
|
compression:
|
|
model: syslog-auto
|
|
provider: harness
|
|
threshold: 0.65
|
|
max_context_window: 131072
|
|
auxiliary:
|
|
vision:
|
|
model: {alias}
|
|
provider: harness
|
|
web_extract:
|
|
model: {alias}
|
|
provider: harness
|
|
compression:
|
|
model: syslog-auto
|
|
provider: harness
|
|
delegation:
|
|
provider: harness
|
|
custom_providers:
|
|
- name: harness
|
|
key_env: LITELLM_API_KEY
|
|
base_url: http://192.168.68.116/v1
|
|
"""
|
|
|
|
|
|
def _run_config(tmp_path, name, text):
|
|
cfg = tmp_path / name
|
|
cfg.write_text(text)
|
|
proc = subprocess.run(
|
|
[sys.executable, str(AUDIT), str(cfg)],
|
|
capture_output=True, text=True,
|
|
)
|
|
return proc.returncode, proc.stdout
|
|
|
|
|
|
def _run(tmp_path, alias):
|
|
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
|
|
|
|
|
|
def test_live_canonical_alias_passes(tmp_path):
|
|
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
|
|
code, out = _run(tmp_path, "gpu-vision")
|
|
assert code == 0, out
|
|
assert "RESULT: PASS" in out
|
|
|
|
|
|
def test_retired_gpu_light_is_rejected(tmp_path):
|
|
"""A config pinned to the retired alias must fail, not pass."""
|
|
code, out = _run(tmp_path, "gpu-light")
|
|
assert code == 1, out
|
|
assert "auxiliary.vision.model must be gpu-vision" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_gemma_is_rejected(tmp_path):
|
|
"""The retired raw model name must fail Rule 8 as well."""
|
|
code, out = _run(tmp_path, "gemma-4-12b")
|
|
assert code == 1, out
|
|
assert "auxiliary.vision.model must be gpu-vision" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_corrected_compression_example_passes(tmp_path):
|
|
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
|
|
code, out = _run(tmp_path, "gpu-vision")
|
|
assert code == 0, out
|
|
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
|
|
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
|
|
assert "RESULT: PASS" in out
|
|
|
|
|
|
def test_retired_alias_in_delegation_is_rejected(tmp_path):
|
|
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"delegation-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
"delegation:\n provider: harness",
|
|
"delegation:\n provider: harness\n model: gpu-light",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "delegation.model = 'gpu-light' is retired" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
|
|
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"custom-provider-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
" - name: harness\n key_env: LITELLM_API_KEY",
|
|
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "custom_providers[0].model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_raw_name_fails(tmp_path):
|
|
"""Retired raw names no longer resolve (400), so they fail; the 2026-09-12 registry change moved qwen3.6-27B-code from raw-but-live to non-resolving."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"retired-qwen.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
"delegation:\n provider: harness",
|
|
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
|
|
),
|
|
)
|
|
assert code != 0, out
|
|
assert "delegation.model = 'qwen3.6-27B-code' is retired and no longer resolves" in out
|
|
assert "use gpu-dense" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
|
"""fallback_providers.model is model-bearing; a retired name there must fail."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"fallback-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
|
|
)
|
|
assert code == 1, out
|
|
assert "fallback_providers.model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_x_search_is_rejected(tmp_path):
|
|
"""x_search.model was previously not enumerated; the derivation must catch it."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"x-search-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
"delegation:\n provider: harness",
|
|
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "x_search.model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|
|
|
|
|
|
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
|
|
"""A nested auxiliary sub-block outside the named three must still be derived."""
|
|
code, out = _run_config(
|
|
tmp_path,
|
|
"nested-aux-gpu-light.yaml",
|
|
BASE.format(alias="gpu-vision").replace(
|
|
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
|
|
" compression:\n model: syslog-auto\n provider: harness\n"
|
|
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
|
|
),
|
|
)
|
|
assert code == 1, out
|
|
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
|
|
assert "RESULT: FAIL" in out
|