fix(monitoring): probe-drift round 2 — gpu port, PVE liveness, report-only visibility, report provenance #70

Merged
abiba-bot merged 7 commits from fm/probe-drift-round2-20260909 into master 2026-09-10 02:17:01 +00:00
3 changed files with 48 additions and 29 deletions
Showing only changes of commit bbdf6c1249 - Show all commits
+1 -1
View File
@@ -240,7 +240,7 @@ The health script prints `📍 executed from: script=… cwd=…` and includes
$ pwd -P $ pwd -P
/root/.treehouse/prose-contracts-9ce5f3/3/prose-contracts /root/.treehouse/prose-contracts-9ce5f3/3/prose-contracts
$ python3 -m pytest -q $ python3 -m pytest -q
23 passed 24 passed
$ shellcheck scripts/prose-lint.sh $ shellcheck scripts/prose-lint.sh
(clean) (clean)
``` ```
+33 -25
View File
@@ -46,7 +46,7 @@ Changelog:
a stale-consumer report is distinguishable from a fault at read time. a stale-consumer report is distinguishable from a fault at read time.
""" """
import subprocess, json, sys, os, time, re import subprocess, json, sys, os, time, re, io, contextlib
from datetime import datetime from datetime import datetime
LITELLM = "http://192.168.68.116:80" LITELLM = "http://192.168.68.116:80"
@@ -659,30 +659,7 @@ def deploy_self():
# MAIN # MAIN
# ═══════════════════════════════════════════════════════════════════ # ═══════════════════════════════════════════════════════════════════
def main(): def _run_checks():
quiet = "--quiet" in sys.argv
as_json = "--json" in sys.argv
# Self-deploy to canonical location
if not quiet and "--no-deploy" not in sys.argv:
deploy_self()
# Provenance: a report is only actionable if the reader can tell WHICH copy
# of this script produced it. Emit the absolute execution path (script + cwd)
# in human, cron-alert, and --json output so a stale-consumer report is
# distinguishable from a real fault at read time. The production cron path
# runs --quiet, so provenance must NOT be behind the quiet guard.
script_path = os.path.abspath(__file__)
cwd = os.getcwd()
if not quiet:
print(f"🏥 Agent Health Check v4 — {datetime.now().strftime('%Y-%m-%d %H:%M UTC')}")
print(f"📍 executed from: script={script_path} cwd={cwd}")
if not quiet:
print()
load_agent_keys()
print("🔑 LiteLLM Keys:") print("🔑 LiteLLM Keys:")
check_keys() check_keys()
print() print()
@@ -710,6 +687,37 @@ def main():
print("🔐 Vault Secrets:") print("🔐 Vault Secrets:")
check_vault_secrets() check_vault_secrets()
def main():
quiet = "--quiet" in sys.argv
as_json = "--json" in sys.argv
# Self-deploy to canonical location
if not quiet and "--no-deploy" not in sys.argv:
deploy_self()
# Provenance: a report is only actionable if the reader can tell WHICH copy
# of this script produced it. A normal run carries it in the header, --json
# carries it for machine consumers, and the cron ALERT line carries it on
# failure. --quiet is documented as "only output on failure", so the header
# is emitted only when not quiet and a healthy quiet run stays silent.
script_path = os.path.abspath(__file__)
cwd = os.getcwd()
if quiet:
captured = io.StringIO()
with contextlib.redirect_stdout(captured):
load_agent_keys()
_run_checks()
if FAIL:
sys.stdout.write(captured.getvalue())
else:
print(f"🏥 Agent Health Check v4 — {datetime.now().strftime('%Y-%m-%d %H:%M UTC')}")
print(f"📍 executed from: script={script_path} cwd={cwd}")
print()
load_agent_keys()
_run_checks()
if FAIL: if FAIL:
print(f"\n❌ {len(FAIL)} FAILURE(S): {' | '.join(FAIL)}") print(f"\n❌ {len(FAIL)} FAILURE(S): {' | '.join(FAIL)}")
if quiet: if quiet:
+14 -3
View File
@@ -137,17 +137,28 @@ def test_json_reports_absolute_execution_provenance(ahc, monkeypatch, capsys):
def test_quiet_run_still_carries_provenance_on_the_alert_path(ahc, monkeypatch, capsys): def test_quiet_run_still_carries_provenance_on_the_alert_path(ahc, monkeypatch, capsys):
# The production cron runs --quiet; a report without provenance is # The cron runs --quiet; a failure report must still carry provenance. The
# unactionable (item 4). Both the header line and the ALERT line must carry it. # header line is suppressed in quiet mode, so the ALERT line is the carrier.
code, out = _run_main(ahc, monkeypatch, capsys, ["--quiet"]) code, out = _run_main(ahc, monkeypatch, capsys, ["--quiet"])
assert code == 1 assert code == 1
assert "📍 executed from: script=" in out assert "📍 executed from:" not in out
alerts = [ln for ln in out.splitlines() if ln.startswith("ALERT agent-health:")] alerts = [ln for ln in out.splitlines() if ln.startswith("ALERT agent-health:")]
assert alerts, out assert alerts, out
assert f"script={os.path.abspath(str(AHC))}" in alerts[0] assert f"script={os.path.abspath(str(AHC))}" in alerts[0]
assert f"cwd={os.getcwd()}" in alerts[0] assert f"cwd={os.getcwd()}" in alerts[0]
def test_quiet_healthy_run_emits_no_stdout(ahc, monkeypatch, capsys):
# --quiet is documented as "only output on failure": a run with no fleet
# failures must produce no stdout at all (the production cron runs --quiet).
for name in ("check_keys", "check_gpu_ports", "check_agents", "check_ct_liveness",
"check_config_integrity", "check_wrapper_integrity", "check_vault_secrets"):
monkeypatch.setattr(ahc, name, lambda: None)
code, out = _run_main(ahc, monkeypatch, capsys, ["--quiet"])
assert code == 0
assert out == ""
def test_json_surfaces_report_only_findings_separately(ahc, monkeypatch, capsys): def test_json_surfaces_report_only_findings_separately(ahc, monkeypatch, capsys):
# Koby's down legs are reported but must not count as fleet failures; the # Koby's down legs are reported but must not count as fleet failures; the
# --json payload exposes them in their own array (item 1 + f8). # --json payload exposes them in their own array (item 1 + f8).