PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 12s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 9s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 17s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
487 lines
18 KiB
Python
487 lines
18 KiB
Python
"""Regression tests for the 2026-09-09/10 probe-drift corrections.
|
|
|
|
WHY THIS FILE EXISTS: the monitoring contracts kept emitting false alarms from
|
|
stale expectations rather than live faults.
|
|
|
|
* agent-health-check v3 reported 6 failures that were all stale expectations:
|
|
abiba (pi-only since the harness purge) was tested as a Hermes host, koby
|
|
(report-only per the captain's 2026-08-17 ruling) was counted as repairable,
|
|
koby's CT 111 was probed on amdpve where it does not exist (it runs on
|
|
storepve .6), and the wrapper infisical check had two bugs — it read only
|
|
the first 20 lines, so koonimo's wrapper (which references /usr/bin/infisical
|
|
past line 20) false-failed, and it treated koby's genuine no-infisical
|
|
(~/.hermes/.env) wrapper as broken.
|
|
* gpu-monitor emitted "DEGRADED — GPU-rtx3090 000, GPU-rtx5070 000" three
|
|
times from probing bare port 80 on GPU hosts while :8080 answered 200.
|
|
* infrastructure-monitoring probed CT 116 for the PVE API (no pveproxy ->
|
|
000) instead of the five real cluster nodes, which answer 401 = alive.
|
|
|
|
These tests execute the health script (with SSH/vault stubbed) and the real
|
|
provenance consumer (scripts/prose-lint.sh), and parse the contracts' executable
|
|
check-health probe blocks into normalized probe sets. No live network, vault, or
|
|
SSH access is required.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import importlib.util
|
|
import json
|
|
import os
|
|
import pathlib
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import textwrap
|
|
|
|
import pytest
|
|
|
|
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
|
AHC = ROOT / "scripts" / "agent-health-check.py"
|
|
LINT = ROOT / "scripts" / "prose-lint.sh"
|
|
GPU = ROOT / "gpu-monitor.prose.md"
|
|
INFRA = ROOT / "infrastructure-monitoring.prose.md"
|
|
|
|
PVE_NODE_IPS = {
|
|
"192.168.68.9",
|
|
"192.168.68.5",
|
|
"192.168.68.15",
|
|
"192.168.68.6",
|
|
"192.168.68.12",
|
|
}
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def ahc():
|
|
"""Import agent-health-check.py without live network/SSH side effects."""
|
|
spec = importlib.util.spec_from_file_location("agent_health_check", AHC)
|
|
assert spec and spec.loader
|
|
module = importlib.util.module_from_spec(spec)
|
|
spec.loader.exec_module(module)
|
|
return module
|
|
|
|
|
|
# ── helpers: execute the health script with SSH/vault stubbed ─────────
|
|
|
|
def _run_main(ahc, monkeypatch, capsys, argv, ssh_result=None):
|
|
ahc.FAIL.clear()
|
|
ahc.REPORT_ONLY.clear()
|
|
monkeypatch.setattr(ahc, "load_agent_keys", lambda: None)
|
|
monkeypatch.setattr(ahc, "ssh", lambda *a, **k: ssh_result)
|
|
monkeypatch.setattr(sys, "argv", ["agent-health-check.py", "--no-deploy", *argv])
|
|
with pytest.raises(SystemExit) as exc:
|
|
ahc.main()
|
|
return exc.value.code, capsys.readouterr().out
|
|
|
|
|
|
def _json_payload(out):
|
|
for line in reversed(out.splitlines()):
|
|
if line.startswith('{"timestamp"'):
|
|
return json.loads(line)
|
|
raise AssertionError(f"no JSON payload in output:\n{out}")
|
|
|
|
|
|
# ── agent-health-check: stale-expectation legs ───────────────────────
|
|
|
|
def test_import_does_not_contact_vault(ahc):
|
|
# Keys are loaded in main() via load_agent_keys(); importing must stay inert.
|
|
assert callable(ahc.load_agent_keys)
|
|
assert all(agent.get("key") is None for agent in ahc.AGENTS.values())
|
|
|
|
|
|
def test_abiba_is_pi_only_runtime(ahc):
|
|
# .24 has run pi-only since the harness purge: no Hermes gateway, config, or
|
|
# wrapper. Probing those legs produced false failures.
|
|
assert ahc.AGENTS["abiba"]["runtime"] == "pi"
|
|
|
|
|
|
def test_koby_is_report_only(ahc):
|
|
# Captain's 2026-08-17 ruling (Rule 17): detect and report, never repair.
|
|
assert ahc.AGENTS["koby"]["report_only"] is True
|
|
|
|
|
|
def test_koby_ct111_is_on_storepve(ahc):
|
|
# Live-verified 2026-09-10: `pct status 111` = running on storepve (.6);
|
|
# amdpve has no lxc/111.conf, which is what false-failed before.
|
|
assert ahc.AGENTS["koby"]["pve"] == "storepve"
|
|
|
|
|
|
def test_tanko_ct112_is_probed_on_minipve(ahc, monkeypatch, capsys):
|
|
# CT 112 (tanko) was live-migrated to minipve (.12) on 2026-09-27; the
|
|
# amdpve mapping made `pct status 112` fail and read as ct-unreachable.
|
|
# Execute the probe and assert the host the script actually contacts.
|
|
probes = []
|
|
monkeypatch.setattr(
|
|
ahc, "ssh",
|
|
lambda host, cmd, user="root": probes.append((host, cmd)) or "status: running",
|
|
)
|
|
ahc.FAIL.clear()
|
|
ahc.REPORT_ONLY.clear()
|
|
try:
|
|
ahc.check_ct_liveness()
|
|
tanko_hosts = [h for h, cmd in probes if cmd == "pct status 112 2>/dev/null"]
|
|
assert tanko_hosts == ["192.168.68.12"]
|
|
finally:
|
|
ahc.FAIL.clear()
|
|
ahc.REPORT_ONLY.clear()
|
|
|
|
|
|
def test_report_only_legs_never_count_as_failures(ahc):
|
|
for agent, report_only in (("koby", True), ("koonimo", False), ("tanko", False)):
|
|
ahc.FAIL.clear()
|
|
ahc.REPORT_ONLY.clear()
|
|
ahc._fail(f"probe:{agent}", agent)
|
|
if report_only:
|
|
assert ahc.FAIL == []
|
|
assert ahc.REPORT_ONLY == [f"probe:{agent}"]
|
|
else:
|
|
assert ahc.FAIL == [f"probe:{agent}"]
|
|
assert ahc.REPORT_ONLY == []
|
|
ahc.FAIL.clear()
|
|
ahc.REPORT_ONLY.clear()
|
|
|
|
|
|
def test_failure_recording_accepts_agentless_keys(ahc):
|
|
ahc.FAIL.clear()
|
|
try:
|
|
ahc._fail("gpu-no-port:gpu-rtx3090 (.8)")
|
|
assert ahc.FAIL == ["gpu-no-port:gpu-rtx3090 (.8)"]
|
|
finally:
|
|
ahc.FAIL.clear()
|
|
|
|
|
|
def test_json_reports_absolute_execution_provenance(ahc, monkeypatch, capsys):
|
|
code, out = _run_main(ahc, monkeypatch, capsys, ["--json"])
|
|
payload = _json_payload(out)
|
|
assert payload["execution_path"] == os.path.abspath(str(AHC))
|
|
assert payload["cwd"] == os.getcwd()
|
|
assert code == 1 # stubbed SSH fails every leg, but provenance is still emitted
|
|
|
|
|
|
def test_quiet_run_still_carries_provenance_on_the_alert_path(ahc, monkeypatch, capsys):
|
|
# The cron runs --quiet; a failure report must still carry provenance. The
|
|
# header line is suppressed in quiet mode, so the ALERT line is the carrier.
|
|
code, out = _run_main(ahc, monkeypatch, capsys, ["--quiet"])
|
|
assert code == 1
|
|
assert "📍 executed from:" not in out
|
|
alerts = [ln for ln in out.splitlines() if ln.startswith("ALERT agent-health:")]
|
|
assert alerts, out
|
|
assert f"script={os.path.abspath(str(AHC))}" in alerts[0]
|
|
assert f"cwd={os.getcwd()}" in alerts[0]
|
|
|
|
|
|
def test_quiet_healthy_run_emits_no_stdout(ahc, monkeypatch, capsys):
|
|
# --quiet is documented as "only output on failure": a run with no fleet
|
|
# failures must produce no stdout at all (the production cron runs --quiet).
|
|
for name in ("check_keys", "check_gpu_ports", "check_agents", "check_ct_liveness",
|
|
"check_config_integrity", "check_wrapper_integrity", "check_vault_secrets"):
|
|
monkeypatch.setattr(ahc, name, lambda: None)
|
|
code, out = _run_main(ahc, monkeypatch, capsys, ["--quiet"])
|
|
assert code == 0
|
|
assert out == ""
|
|
|
|
|
|
def test_json_surfaces_report_only_findings_separately(ahc, monkeypatch, capsys):
|
|
# Koby's down legs are reported but must not count as fleet failures; the
|
|
# --json payload exposes them in their own array (item 1 + f8).
|
|
_, out = _run_main(ahc, monkeypatch, capsys, ["--json"])
|
|
payload = _json_payload(out)
|
|
assert isinstance(payload["report_only"], list)
|
|
assert any(key.startswith(("gateway-down:koby", "ct-unreachable:koby"))
|
|
for key in payload["report_only"])
|
|
assert not any("koby" in key for key in payload["failures"])
|
|
|
|
|
|
# ── agent-health-check: wrapper infisical behavior (f3) ───────────────
|
|
|
|
def _stub_wrapper_ssh(ahc, monkeypatch, wrapper_body, test_x_result="OK", command_v="/usr/local/bin/infisical"):
|
|
def fake_ssh(host, cmd, user="root"):
|
|
if cmd.startswith("cat /root/.local/bin/hermes"):
|
|
return wrapper_body
|
|
if cmd.startswith("ls -la /root/.local/bin/hermes "):
|
|
return "-rwxr-xr-x 1 root root 0 Jan 1 00:00 /root/.local/bin/hermes"
|
|
if cmd.startswith("ls -la /root/.local/bin/hermes-real") or "venv/bin/hermes" in cmd:
|
|
return "-rwxr-xr-x 1 root root 0 Jan 1 00:00 /root/.local/bin/hermes-real"
|
|
if cmd.startswith("grep -c 'LITELLM_API_KEY'"):
|
|
return "1"
|
|
if cmd.startswith("test -x "):
|
|
path = cmd[len("test -x "):].split()[0]
|
|
if isinstance(test_x_result, dict):
|
|
return test_x_result.get(path, "MISS")
|
|
return test_x_result
|
|
if cmd.startswith("command -v infisical"):
|
|
return command_v
|
|
return None
|
|
|
|
monkeypatch.setattr(ahc, "ssh", fake_ssh)
|
|
monkeypatch.setattr(ahc, "AGENTS", {"koonimo": dict(ahc.AGENTS["koonimo"])})
|
|
ahc.FAIL.clear()
|
|
ahc.REPORT_ONLY.clear()
|
|
|
|
|
|
def test_env_based_wrapper_without_infisical_is_not_failed(ahc, monkeypatch, capsys):
|
|
_stub_wrapper_ssh(ahc, monkeypatch,
|
|
"#!/bin/bash\nsource ~/.hermes/.env\nexec hermes-real \"$@\"\n")
|
|
ahc.check_wrapper_integrity()
|
|
out = capsys.readouterr().out
|
|
assert ahc.FAIL == []
|
|
assert "wrapper resolves creds without infisical" in out
|
|
|
|
|
|
def test_dangling_absolute_infisical_path_is_failed(ahc, monkeypatch, capsys):
|
|
# Wrapper hardcodes /usr/bin/infisical, which is absent, while PATH resolves
|
|
# infisical to /usr/local/bin/infisical. The literal path must be verified,
|
|
# not inferred from PATH resolution.
|
|
_stub_wrapper_ssh(ahc, monkeypatch,
|
|
"#!/bin/bash\n/usr/bin/infisical run -- hermes-real \"$@\"\n",
|
|
test_x_result="MISS", command_v="/usr/local/bin/infisical")
|
|
ahc.check_wrapper_integrity()
|
|
assert "wrapper-infisical-path:koonimo" in ahc.FAIL
|
|
|
|
|
|
def test_existing_absolute_infisical_path_passes(ahc, monkeypatch, capsys):
|
|
_stub_wrapper_ssh(ahc, monkeypatch,
|
|
"#!/bin/bash\n/usr/bin/infisical run -- hermes-real \"$@\"\n",
|
|
test_x_result="OK")
|
|
ahc.check_wrapper_integrity()
|
|
out = capsys.readouterr().out
|
|
assert ahc.FAIL == []
|
|
assert "wrapper infisical path OK" in out
|
|
|
|
|
|
def test_comment_mentioning_removed_infisical_path_is_not_failed(ahc, monkeypatch, capsys):
|
|
# litellm-api-keys.prose.md documents `rm -f /usr/local/bin/infisical`; a
|
|
# wrapper comment about that migration must not manufacture a dangling path
|
|
# when the real invocation (/usr/bin/infisical) is present and executable.
|
|
_stub_wrapper_ssh(ahc, monkeypatch,
|
|
"#!/bin/bash\n# migrated from /usr/local/bin/infisical\n"
|
|
"exec /usr/bin/infisical run -- hermes-real \"$@\"\n",
|
|
test_x_result={"/usr/bin/infisical": "OK",
|
|
"/usr/local/bin/infisical": "MISS"})
|
|
ahc.check_wrapper_integrity()
|
|
out = capsys.readouterr().out
|
|
assert ahc.FAIL == []
|
|
assert "wrapper infisical path OK" in out
|
|
|
|
|
|
def test_comment_only_infisical_mention_does_not_reach_path_check(ahc, monkeypatch, capsys):
|
|
# A comment-only mention of a removed infisical path on a healthy .env-based
|
|
# wrapper is not an invocation: it must not fall through to the `command -v`
|
|
# PATH check and false-FAIL `wrapper-no-infisical`.
|
|
_stub_wrapper_ssh(ahc, monkeypatch,
|
|
"#!/bin/bash\n# migrated from /usr/local/bin/infisical\n"
|
|
"source ~/.hermes/.env\nexec hermes-real \"$@\"\n",
|
|
test_x_result="MISS", command_v=None)
|
|
ahc.check_wrapper_integrity()
|
|
out = capsys.readouterr().out
|
|
assert ahc.FAIL == []
|
|
assert "wrapper resolves creds without infisical" in out
|
|
|
|
|
|
# ── item 4: prose-lint enforces report provenance (real consumer) ─────
|
|
|
|
GOOD_CONTRACT = textwrap.dedent("""\
|
|
---
|
|
kind: function
|
|
name: good
|
|
description: fixture with provenance
|
|
---
|
|
|
|
## Parameters
|
|
|
|
- x: y
|
|
|
|
## Returns
|
|
|
|
ok
|
|
|
|
### check-health
|
|
|
|
```bash
|
|
pwd -P
|
|
```
|
|
|
|
**Report format**: Begin with the absolute path the probe executed from.
|
|
""")
|
|
|
|
DECOY_CONTRACT = textwrap.dedent("""\
|
|
---
|
|
kind: function
|
|
name: decoy
|
|
description: fixture with provenance only outside the report format
|
|
---
|
|
|
|
## Parameters
|
|
|
|
- x: y
|
|
|
|
## Returns
|
|
|
|
ok
|
|
|
|
The absolute path of the config is /etc/foo.
|
|
|
|
### check-health
|
|
|
|
```bash
|
|
true
|
|
```
|
|
|
|
**Report format**: Summarize actual results from each probe.
|
|
""")
|
|
|
|
MISSING_CONTRACT = textwrap.dedent("""\
|
|
---
|
|
kind: function
|
|
name: missing
|
|
description: check-health contract with no report format
|
|
---
|
|
|
|
## Parameters
|
|
|
|
- x: y
|
|
|
|
## Returns
|
|
|
|
ok
|
|
|
|
### check-health
|
|
|
|
```bash
|
|
pwd -P
|
|
```
|
|
""")
|
|
|
|
|
|
def _run_lint(tmp_path, text, name):
|
|
(tmp_path / name).write_text(text)
|
|
return subprocess.run(["bash", str(LINT)], cwd=tmp_path,
|
|
capture_output=True, text=True)
|
|
|
|
|
|
def test_prose_lint_accepts_report_format_with_provenance(tmp_path):
|
|
result = _run_lint(tmp_path, GOOD_CONTRACT, "good.prose.md")
|
|
assert result.returncode == 0, result.stdout + result.stderr
|
|
|
|
|
|
def test_prose_lint_rejects_report_format_without_provenance(tmp_path):
|
|
result = _run_lint(tmp_path, DECOY_CONTRACT, "decoy.prose.md")
|
|
assert result.returncode == 1, result.stdout
|
|
assert "lacks execution provenance" in result.stdout
|
|
|
|
|
|
def test_prose_lint_requires_report_format_on_check_health_contract(tmp_path):
|
|
result = _run_lint(tmp_path, MISSING_CONTRACT, "missing.prose.md")
|
|
assert result.returncode == 1, result.stdout
|
|
assert "no **Report format** paragraph" in result.stdout
|
|
|
|
|
|
# ── contracts: parse the executable check-health probe block ─────────
|
|
|
|
def _check_health_block(contract):
|
|
"""Extract the bash probe block under ### check-health (the probe interface)."""
|
|
text = contract.read_text()
|
|
marker = "### check-health"
|
|
assert marker in text, f"{contract.name} has no {marker}"
|
|
after = text.split(marker, 1)[1]
|
|
match = re.search(r"```bash\n(.*?)```", after, re.S)
|
|
assert match, f"{contract.name} check-health has no bash probe block"
|
|
return match.group(1)
|
|
|
|
|
|
def _loop_nodes(block):
|
|
nodes = []
|
|
for line in block.splitlines():
|
|
match = re.match(r"\s*for\s+\w+\s+in\s+(.+?);?\s*do\b", line)
|
|
if match:
|
|
nodes = match.group(1).split()
|
|
return nodes
|
|
|
|
|
|
def _record(url):
|
|
"""Normalize a URL into a probe record: host, port, path, expected status."""
|
|
match = re.match(r"https?://([^/\s\"')]+)(/[^\s\"')]*)?", url)
|
|
assert match, f"unparseable probe URL: {url}"
|
|
hostport = match.group(1)
|
|
if "@" in hostport:
|
|
hostport = hostport.split("@", 1)[1]
|
|
if hostport.startswith("["):
|
|
host, port = hostport[1:hostport.index("]")], None
|
|
elif ":" in hostport:
|
|
host, raw_port = hostport.rsplit(":", 1)
|
|
port = int(raw_port) if raw_port.isdigit() else None
|
|
else:
|
|
host, port = hostport, None
|
|
return {"host": host, "port": port, "path": match.group(2) or "/",
|
|
"expected": None}
|
|
|
|
|
|
def _probes(block):
|
|
"""Parse the executable check-health bash block into a normalized probe model.
|
|
|
|
Comments are not probes; an `# Expected: <status>` comment annotates the
|
|
preceding probe. URLs using the block's shell-loop variable `$node` are
|
|
expanded over the loop's node list.
|
|
"""
|
|
loop_nodes = _loop_nodes(block)
|
|
probes = []
|
|
last = None
|
|
for raw in block.splitlines():
|
|
stripped = raw.strip()
|
|
if stripped.startswith("#"):
|
|
expected = re.search(r"Expected:\s*(\d{3})", stripped, re.I)
|
|
if expected and last is not None:
|
|
last["expected"] = int(expected.group(1))
|
|
continue
|
|
for url in re.findall(r"https?://[^\s\"')]+", raw):
|
|
hosts = loop_nodes if "$node" in url else [None]
|
|
for node in hosts:
|
|
record = _record(url.replace("$node", node) if node else url)
|
|
probes.append(record)
|
|
last = record
|
|
return probes
|
|
|
|
|
|
def test_gpu_monitor_probes_every_gpu_health_on_8080():
|
|
probes = _probes(_check_health_block(GPU))
|
|
targets = {(p["host"], p["port"], p["path"]) for p in probes}
|
|
assert ("192.168.68.8", 8080, "/health") in targets
|
|
assert ("192.168.68.110", 8080, "/health") in targets
|
|
|
|
|
|
def test_gpu_monitor_never_probes_bare_port_80_on_gpu_hosts():
|
|
probes = _probes(_check_health_block(GPU))
|
|
gpu_hosts = {"192.168.68.8", "192.168.68.110", "192.168.68.15"}
|
|
offenders = [p for p in probes
|
|
if p["host"] in gpu_hosts and p["port"] in (None, 80)]
|
|
assert offenders == []
|
|
|
|
|
|
def test_probe_model_flags_explicit_port_80_on_gpu_host():
|
|
# Regression: a bare-port probe may be spelled with an explicit :80.
|
|
block = ("curl -s -o /dev/null -w '%{http_code}' "
|
|
"http://192.168.68.8:80/health\n")
|
|
gpu_hosts = {"192.168.68.8", "192.168.68.110", "192.168.68.15"}
|
|
offenders = [p for p in _probes(block)
|
|
if p["host"] in gpu_hosts and p["port"] in (None, 80)]
|
|
assert offenders and offenders[0]["port"] == 80
|
|
|
|
|
|
def test_gpu_monitor_treats_router_301_as_alive():
|
|
probes = _probes(_check_health_block(GPU))
|
|
unified = [p for p in probes
|
|
if p["host"] == "192.168.68.116" and p["path"] == "/health/unified"]
|
|
assert unified, "router /health/unified probe missing"
|
|
assert unified[0]["expected"] == 301
|
|
|
|
|
|
def test_infra_monitoring_probes_every_real_pve_node():
|
|
probes = _probes(_check_health_block(INFRA))
|
|
pve = {(p["host"], p["port"], p["path"]) for p in probes if p["port"] == 8006}
|
|
assert {host for host, _, _ in pve} == PVE_NODE_IPS
|
|
assert {path for _, _, path in pve} == {"/api2/json/version"}
|
|
|
|
|
|
def test_infra_monitoring_does_not_probe_ct116_for_pve_api():
|
|
probes = _probes(_check_health_block(INFRA))
|
|
assert not any(p["host"] == "192.168.68.116" and p["port"] == 8006
|
|
for p in probes)
|