From 226f2ad55e1bf8485a10cba9c71632a841203c5b Mon Sep 17 00:00:00 2001 From: root Date: Sat, 26 Sep 2026 14:50:52 +0000 Subject: [PATCH] fix: reconcile the health-logs self-heal contracts, and add a dead-man's-switch relay-785 / self-heal-contract-stale-script-pointers-20260924. gpu/ had been silent since 2026-09-14T18:02:03Z (12 days) and pm2/ had never posted a run. FINDING - the gpu-self-heal executor was never lost, only its schedule was. The brief concluded the mechanism was gone. It is not: on CT 116 /opt/inference-harness/scripts/gpu-self-heal.py exists (mtime 2026-09-21 11:43), is self-posting via gitea-logger.sh, and runs cleanly - I ran it and it posted health-logs/gpu/gpu-self-heal-20260926-144618.json. What was missing was the cron entry, dropped during a CT 116 /etc/cron.d rework on 2026-09-21. So this is a schedule restoration, not a resurrection. DECISIONS * gpu-self-heal -> RESTORE. Schedule re-added as /etc/cron.d/gpu-self-heal on CT 116, matching the observed historical cadence (:02 past 0/6/12/18). Proven with a REAL unattended cron fire (temporary */1 entry, removed after): gpu-self-heal-20260926-144802.json committed 14:48:03Z. The contract now names the executor, schedule, log and posting, which it previously did not. * pm2 health-logs posting -> RETIRE the claim. pm2-self-heal.prose.md required appending to health-logs/pm2/ as a 'hard rule', but scripts/pm2-self-heal.sh contains no Gitea or push code and never did; the directory has held only its init commit since 2026-07-28. Corrected to point at the contract-runner's durable per-run logs and failure note instead of adding a second, redundant posting path. DEAD-MAN'S-SWITCH - scripts/health-log-freshness.py + contract. Absence of logs must raise an alarm, and the producer cannot raise it, so this runs on CT 100, a different host from the producers, and fails when the newest health-logs/gpu entry is older than 12h (litellm 18h). Verified it would have caught the real gap: evaluated at 2026-09-20 the newest gpu/ entry was 126h old against a 12h limit -> STALE. Bugs found and fixed while testing, each caught by a test that bit: * the documented HEALTH_LOG_MAX_AGE_* override was never implemented; * a Gitea password held under GITEA_TOKEN was sent as an API token -> HTTP 401; now every candidate auth is tried and the first that works is used; * the ~/.git-credentials fallback filtered on GITEA_URL's host, which missed whenever GITEA_URL pointed at the internal IP -> zero candidates. Exit codes: 0 fresh, 1 stale-or-unreadable, 2 cannot run. A directory that cannot be read is a failure, never a skip. Evidence: live PASS; stale override names the directory and producer; no credential -> exit 2; whole thing runs green through contract-run.sh. prose-lint: PASSED. --- gpu-self-heal.prose.md | 23 ++++ health-log-freshness.prose.md | 80 +++++++++++ pm2-self-heal.prose.md | 12 +- scripts/contract-run.sh | 5 + scripts/health-log-freshness.py | 227 ++++++++++++++++++++++++++++++++ 5 files changed, 346 insertions(+), 1 deletion(-) create mode 100644 health-log-freshness.prose.md create mode 100755 scripts/health-log-freshness.py diff --git a/gpu-self-heal.prose.md b/gpu-self-heal.prose.md index c7c26ae..dd0963b 100644 --- a/gpu-self-heal.prose.md +++ b/gpu-self-heal.prose.md @@ -174,6 +174,29 @@ Key notes: ## Execution +### Executor, schedule and dead-man's-switch + +This contract is implemented by a real script, scheduled on the inference host: + +| | | +| --- | --- | +| **Executor** | `/opt/inference-harness/scripts/gpu-self-heal.py` on **CT 116** | +| **Schedule** | `/etc/cron.d/gpu-self-heal` on CT 116 — `2 */6 * * *` | +| **Log** | `/var/log/litellm/gpu-self-heal.log` | +| **Posting** | the script calls `gitea-logger.sh gpu {RUN_ID}.json ` → `SyslogSolution/health-logs/gpu/{RUN_ID}.json` | + +**Dead-man's-switch:** absence of logs must raise an alarm, because that is how +this went silent for 12 days. That alarm cannot live on the producer — a stopped +job cannot report that it stopped — so it lives off-host as the +**`health-log-freshness` contract on CT 100**, which fails when +`health-logs/gpu/` is older than 12 h. A failed run is also visible in the log +above, but only the off-host check catches a *missing* run. + +**History (2026-09-26, relay-785):** the executor was never lost — only its +schedule was, dropped during a CT 116 `/etc/cron.d` rework on 2026-09-21. The +`gpu/` log was silent from `2026-09-14T18:02:03Z`. The schedule was restored and +the off-host freshness check added; do not treat either as optional. + ```prose -- Phase 1: Fetch live GPU data let fleet = call gpu-monitor diff --git a/health-log-freshness.prose.md b/health-log-freshness.prose.md new file mode 100644 index 0000000..9bb8bc1 --- /dev/null +++ b/health-log-freshness.prose.md @@ -0,0 +1,80 @@ +--- +kind: function +name: health-log-freshness +description: > + Dead-man's-switch for the health-logs posting jobs. Fails when the newest + commit in a watched SyslogSolution/health-logs directory is older than that + directory's threshold. + + Exists because absence of logs raises no alarm: the gpu/ directory went silent + for 12 days (2026-09-14T18:02:03Z to 2026-09-26) and nothing noticed, because + the only thing that would have noticed was the job that had stopped. The check + therefore runs on CT 100, a DIFFERENT host from the producers on CT 116, so a + dead producer host or a deleted schedule still raises an alarm. + + Watched: gpu/ (12h, producer 2 */6 * * *), litellm/ (18h, producer 0 */6 * * *). + Not watched: pm2/ — that contract's health-logs claim was retired 2026-09-26; + see pm2-self-heal.prose.md step 6. + + Verified 2026-09-26: would have caught the real gap (at 2026-09-20 the newest + gpu/ entry was 126h old against a 12h limit). + +version: 1.0.0 +--- + +## Execution + +Host-scheduled on CT 100 via `/etc/cron.d/contract-runner`: + +``` +20 */4 * * * root CONTRACT_RUN_LOG_DIR=/var/log/contract-runs /bin/bash /opt/contract-runner/scripts/contract-run.sh health-log-freshness >/dev/null 2>&1 || /root/abiba-workspace/bin/fm-inbox.sh note "contract-runner: health-log-freshness FAILED - see /var/log/contract-runs/" >/dev/null 2>&1 +``` + +Every 4 hours, offset to `:20` to avoid the existing `:05`/`:15`/`:35` slots. + +## Output shape + +``` +Health-log freshness — dead-man's-switch +================================================================== +✅ health-logs/gpu/ newest 2026-09-26T14:48:03Z (0.02h old, limit 12.0h) + last commit: gpu: gpu-self-heal-20260926-144802.json + producer: gpu-self-heal.py, CT116 cron 2 */6 * * * +================================================================== +VERDICT: PASS — every watched health-log directory is advancing +``` + +`--json` emits `{checked: {...}, failures: [...]}`. + +## Exit codes + +| exit | meaning | +| --- | --- | +| 0 | every watched directory is advancing | +| 1 | at least one is stale, or could not be read | +| 2 | the check could not run (no Gitea credential) | + +A directory that **cannot be read** is a failure, not a skip: unreadable and +stopped are indistinguishable from the outside. + +## Configuration + +| variable | default | meaning | +| --- | --- | --- | +| `GITEA_URL` | `https://git.sysloggh.net` | Gitea base URL | +| `GITEA_TOKEN` / `GITEA_PAT` | — | API token; falls back to basic auth from `~/.git-credentials` | +| `HEALTH_LOG_MAX_AGE_GPU` | `12` | hours | +| `HEALTH_LOG_MAX_AGE_LITELLM` | `18` | hours | + +## Thresholds + +Sized for the producer cadence plus one missed run, so a single blip does not +page but a genuine stop does: + +* `gpu/` — 6 h cadence, 12 h limit; +* `litellm/` — 6 h cadence, 18 h limit (proven healthy; a looser bound avoids noise). + +## Maintains + +- health-logs-gpu-freshness: { status: "ok|stale", last_check: timestamp } +- health-logs-litellm-freshness: { status: "ok|stale", last_check: timestamp } diff --git a/pm2-self-heal.prose.md b/pm2-self-heal.prose.md index 1336f70..a00646c 100644 --- a/pm2-self-heal.prose.md +++ b/pm2-self-heal.prose.md @@ -76,7 +76,17 @@ monitoring work happens in the host cron job. - If status is "online" → pass - If status is "stopped" or "errored" → apply Rule 1 - If restarts > 5 → alert owner -6. **Log results** — Append to `SyslogSolution/health-logs/pm2/{timestamp}.md` in Gitea (not knowledge graph — hard rule) +6. **Log results** — the durable per-run record is `/var/log/contract-runs/pm2-self-heal-.log` on CT 100, written by `scripts/contract-run.sh` from `/etc/cron.d/contract-runner` every 4 hours, with a firstmate inbox note raised on any non-zero exit. + + **CORRECTED 2026-09-26 (relay-785):** this step previously required appending to + `SyslogSolution/health-logs/pm2/{timestamp}.md` in Gitea and called it a "hard rule". + That posting was **never implemented** — `scripts/pm2-self-heal.sh` contains no + Gitea or git-push code — so the requirement was a coverage claim the executor did + not honour, and `health-logs/pm2/` has held only its init commit since 2026-07-28. + The claim is retired rather than implemented: the contract-runner's per-run logs + plus its failure note already give a durable record and a working alarm, and a + second posting path would add work without adding a signal. `health-logs/pm2/` + is left as historical evidence, not as a live obligation. 7. **Alert** — Send Telegram (primary) or Zulip DM (secondary) to owner if escalation needed (do NOT run pm2 commands during alerting) 8. **Wait 5 min** → repeat from step 1 diff --git a/scripts/contract-run.sh b/scripts/contract-run.sh index 3b6ec8d..783fee8 100755 --- a/scripts/contract-run.sh +++ b/scripts/contract-run.sh @@ -19,6 +19,7 @@ # disk-gc-threat-response -> scripts/disk-gc-scan.py # pm2-self-heal -> scripts/pm2-self-heal.sh # search-stack-visibility -> scripts/search-stack-check.py +# health-log-freshness -> scripts/health-log-freshness.py # # Execution copy: every contract pins the clone this script lives in (see # docs/contract-execution-pinning.md). Before a contract runs, this wrapper @@ -79,6 +80,10 @@ case "$CONTRACT_NAME" in SCRIPT_PATH="${SCRIPTS_DIR}/search-stack-check.py" INTERPRETER="python3" ;; + health-log-freshness) + SCRIPT_PATH="${SCRIPTS_DIR}/health-log-freshness.py" + INTERPRETER="python3" + ;; *) echo "Unknown contract: $CONTRACT_NAME" | tee -a "$LOG_FILE" # Send alert for unknown contract diff --git a/scripts/health-log-freshness.py b/scripts/health-log-freshness.py new file mode 100755 index 0000000..31c805f --- /dev/null +++ b/scripts/health-log-freshness.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""Health-log freshness watchdog (dead-man's-switch). + +Absence of logs raises no alarm. On 2026-09-26 the `gpu/` directory of +SyslogSolution/health-logs went silent for 12 days unnoticed, because the only +thing that would have noticed was the job that had stopped. This check lives on +a DIFFERENT host (CT 100) from the producers, so a producer host that is dead, +or a schedule that was deleted, still raises an alarm. + +It asks the Gitea API for the newest commit touching each watched directory and +fails when that commit is older than the directory's threshold. + +Usage: + health-log-freshness.py # check every watched directory + health-log-freshness.py --json # machine-readable output + +Exit: 0 = all fresh, 1 = at least one stale or unreachable, 2 = check could not run. + +Environment: + GITEA_URL default https://git.sysloggh.net + GITEA_TOKEN API token (falls back to GITEA_PAT, then to the token + embedded in ~/.git-credentials for that host) + HEALTH_LOG_MAX_AGE_H override thresholds, e.g. HEALTH_LOG_MAX_AGE_GPU=12 +""" + +from __future__ import annotations + +import base64 +import json +import os +import re +import sys +import urllib.error +import urllib.request +from datetime import datetime, timezone + +REPO = "SyslogSolution/health-logs" +GITEA_URL = os.environ.get("GITEA_URL", "https://git.sysloggh.net").rstrip("/") + +# dir -> (threshold_hours, why) +# Thresholds are sized for the producer cadence plus one missed run: +# gpu/ runs every 6h -> 12h tolerates one miss, catches a second +# litellm/ runs every 6h -> 18h (it is proven healthy; a looser bound avoids +# noise while still catching a real stop) +WATCHED: dict[str, tuple[float, str]] = { + "gpu": (12.0, "gpu-self-heal.py, CT116 cron 2 */6 * * *"), + "litellm": (18.0, "litellm-health-check.sh, CT116 cron 0 */6 * * *"), +} + + +def _threshold(directory: str, default: float) -> float: + """Allow HEALTH_LOG_MAX_AGE_ to override a threshold. + + Documented override; used both operationally (tighten a bound while + investigating) and in tests (force a stale verdict deterministically). + """ + raw = os.environ.get(f"HEALTH_LOG_MAX_AGE_{directory.upper()}") + if raw is None: + return default + try: + return float(raw) + except ValueError: + print(f"WARN: ignoring non-numeric HEALTH_LOG_MAX_AGE_{directory.upper()}={raw!r}") + return default + +# pm2/ is deliberately NOT watched: pm2-self-heal.prose.md claimed a health-logs +# posting that its executor never implemented, and the claim was retired on +# 2026-09-26 in favour of the contract-runner's per-run logs. The directory is +# left as historical evidence, not as a live obligation. + + +def _auth_candidates() -> list[str]: + """Authorization header values to try, in order. + + Gitea accepts either an API token (``token ``) or HTTP basic auth, and + the value held under ``GITEA_TOKEN`` is not reliably an API token - on this + host it is the git account's *password*, which sent as a token returns HTTP + 401. So return every candidate and let the caller use the first that works, + rather than guessing and failing. + """ + candidates: list[str] = [] + for var in ("GITEA_TOKEN", "GITEA_PAT"): + if os.environ.get(var): + candidates.append(f"token {os.environ[var]}") + candidates.append( + "Basic " + + base64.b64encode( + f"{_git_user()}:{os.environ[var]}".encode() + ).decode() + ) + # ~/.git-credentials may hold this Gitea under several hostnames (the public + # name and the internal IP both appear in this fleet), so accept any of them + # rather than filtering on the configured host - that filter produced zero + # candidates whenever GITEA_URL pointed at the internal address. + try: + with open(os.path.expanduser("~/.git-credentials")) as fh: + for line in fh: + m = re.match(r"https://([^:]+):([^@]+)@", line.strip()) + if m: + raw = f"{m.group(1)}:{m.group(2)}".encode() + candidates.append("Basic " + base64.b64encode(raw).decode()) + except OSError: + pass + seen, out = set(), [] + for c in candidates: + if c not in seen: + seen.add(c) + out.append(c) + return out + + +def _git_user() -> str: + return os.environ.get("GITEA_USER", "abiba-bot") + + +def newest_commit_iso(directory: str, auth: list[str] | None) -> tuple[str | None, str]: + """Return (iso_timestamp, detail) for the newest commit touching `directory`.""" + url = ( + f"{GITEA_URL}/api/v1/repos/{REPO}/commits" + f"?path={directory}&limit=1&stat=false" + ) + req = urllib.request.Request(url, headers={"Accept": "application/json"}) + headers = list(auth or []) + last = "no credential" + for i, hdr in enumerate(headers or [None]): + r = urllib.request.Request(url, headers={"Accept": "application/json"}) + if hdr: + r.add_header("Authorization", hdr) + try: + with urllib.request.urlopen(r, timeout=20) as resp: + data = json.loads(resp.read().decode("utf-8", "replace")) + break + except urllib.error.HTTPError as exc: + last = f"HTTP {exc.code}" + if exc.code not in (401, 403): + return None, last + except Exception as exc: # noqa: BLE001 + return None, repr(exc) + else: + return None, last + + if not data: + return None, "no commits" + commit = data[0].get("commit", {}) + when = ( + (commit.get("committer") or {}).get("date") + or (commit.get("author") or {}).get("date") + ) + msg = (commit.get("message") or "").splitlines()[0][:60] + return when, msg + + +def main() -> int: + as_json = "--json" in sys.argv + auth = _auth_candidates() + now = datetime.now(timezone.utc) + failures: list[str] = [] + report: dict[str, dict] = {} + + if not auth: + print("FAIL: no Gitea credential available (GITEA_TOKEN/GITEA_PAT/~/.git-credentials)") + return 2 + for directory, (default_age_h, why) in WATCHED.items(): + max_age_h = _threshold(directory, default_age_h) + when, detail = newest_commit_iso(directory, auth) + entry: dict = {"directory": directory, "producer": why, "max_age_h": max_age_h} + + if when is None: + entry.update(ok=False, reason=f"could not read newest commit: {detail}") + failures.append( + f"health-logs/{directory}/ could not be read ({detail}) — " + f"a directory that cannot be read is indistinguishable from one that stopped" + ) + else: + try: + ts = datetime.fromisoformat(when.replace("Z", "+00:00")) + except ValueError: + entry.update(ok=False, reason=f"unparseable timestamp {when!r}") + failures.append(f"health-logs/{directory}/ timestamp unparseable: {when!r}") + report[directory] = entry + continue + age_h = (now - ts).total_seconds() / 3600.0 + stale = age_h > max_age_h + entry.update( + ok=not stale, + newest=when, + age_h=round(age_h, 2), + newest_commit=detail, + ) + if stale: + entry["reason"] = f"stale: {age_h:.1f}h > {max_age_h}h" + failures.append( + f"health-logs/{directory}/ is STALE: newest entry {when} " + f"({age_h:.1f}h old, limit {max_age_h}h) from {why}" + ) + report[directory] = entry + + if as_json: + print(json.dumps({"checked": report, "failures": failures}, indent=2)) + return 1 if failures else 0 + + print("Health-log freshness — dead-man's-switch") + print("=" * 66) + for directory, entry in report.items(): + mark = "✅" if entry.get("ok") else "❌" + if entry.get("newest"): + print( + f"{mark} health-logs/{directory}/ newest {entry['newest']} " + f"({entry['age_h']}h old, limit {entry['max_age_h']}h)" + ) + print(f" last commit: {entry.get('newest_commit')}") + else: + print(f"{mark} health-logs/{directory}/ {entry.get('reason')}") + print(f" producer: {entry['producer']}") + + print("=" * 66) + if failures: + print("VERDICT: FAIL") + for f in failures: + print(f" - {f}") + return 1 + print("VERDICT: PASS — every watched health-log directory is advancing") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) -- 2.54.0