#!/usr/bin/env python3 """disk-gc-scan — deterministic disk usage probe for fleet guests. This is the executable scanner side of `disk-gc-threat-response.prose.md`. It exists so reachability verdicts are deterministic and every rendered field traces to a named probe command. DESIGN PRINCIPLES (per task disk-gc-probe-false-unreachable-20260913): 1. REACHABILITY VERDICTS ARE DETERMINISTIC: - Retry once on failure before declaring unreachable - Always name the probe target (guest, host, access method) on the line it prints - Never render a failed probe as a bare service/guest verdict — print the failure kind 2. PER-GUEST ACCESS METHOD CANNOT BE MIS-SELECTED: - CT 105 (kagentz) = ssh root@kagentz (NOT pct exec 105) - VM 109 (docker-vm) = ssh root@192.168.68.7 (NOT pct) - All other CTs = pct-run (which uses pct exec) - The access method is selected from a per-guest map so the wrong path cannot be picked by an executor improvising 3. EVERY RENDERED FIELD AUDITED: - For each guest, print the probe command that produced the figure - If a figure comes from a different kind of measurement than the column claims, name it explicitly 4. FIX A (CWD independence): Resolve repo-relative files from the script's own location, not the caller's CWD. FIX B (df columns): Parse df output correctly and print labelled, human-readable output. Usage: disk-gc-scan.py # scan all guests disk-gc-scan.py --json # machine-readable output Exit codes: 0 ok (all guests probed), 1 probe error """ from __future__ import annotations import json import pathlib import subprocess import sys import time import os from dataclasses import dataclass from typing import Optional # Resolve repo-relative files from the script's own location, not the caller's CWD SCRIPT_DIR = pathlib.Path(__file__).resolve().parent HELPER_PCT_RUN = SCRIPT_DIR / "pct-run.sh" # Per-guest access method map. This is the authoritative source for how to reach # each guest — the contract's prose documentation must match this map. # # Access methods: # - "pct-run": use pct-run.sh (pct exec via SSH to node) # - "ssh-host": use ssh root@ # - "ssh-ip": use ssh root@ @dataclass class Guest: """A guest to probe.""" ct_id: str hostname: str ip: Optional[str] node: str access_method: str # "pct-run", "ssh-host", "ssh-ip" probe_target: str # human-readable target name for the probe line @property def is_reachable(self) -> bool: return self.probe_result is not None and self.probe_result.exit_code == 0 @property def usage_pct(self) -> Optional[float]: return self.probe_result.usage_pct if self.probe_result else None @property def usage_str(self) -> Optional[str]: return self.probe_result.usage_str if self.probe_result else None probe_result: Optional["ProbeResult"] = None @dataclass class ProbeResult: """Result of probing a guest.""" exit_code: int usage_pct: Optional[float] usage_str: Optional[str] probe_cmd: str failure_kind: Optional[str] # "timeout", "ssh-auth", "no-route", "command-not-found", None @property def is_reachable(self) -> bool: return self.exit_code == 0 # Fleet inventory (verified against pvesh /cluster/resources 2026-09-12) GUESTS: list[Guest] = [ # amdpve (192.168.68.15) Guest(ct_id="105", hostname="kagentz", ip="192.168.68.105", node="amdpve", access_method="ssh-host", probe_target="kagentz (CT 105, amdpve)"), Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="amdpve", access_method="pct-run", probe_target="tanko (CT 112, amdpve)"), Guest(ct_id="113", hostname="baggy", ip="192.168.68.113", node="amdpve", access_method="pct-run", probe_target="baggy (CT 113, amdpve)"), Guest(ct_id="115", hostname="scottdenya", ip="192.168.68.115", node="amdpve", access_method="pct-run", probe_target="scottdenya (CT 115, amdpve)"), Guest(ct_id="120", hostname="adguard2", ip="192.168.68.120", node="amdpve", access_method="pct-run", probe_target="adguard2 (CT 120, amdpve)"), # minipve (192.168.68.12) Guest(ct_id="100", hostname="abiba", ip="192.168.68.100", node="minipve", access_method="pct-run", probe_target="abiba (CT 100, minipve)"), Guest(ct_id="102", hostname="adguard", ip="192.168.68.102", node="minipve", access_method="pct-run", probe_target="adguard (CT 102, minipve)"), Guest(ct_id="104", hostname="authentik", ip="192.168.68.104", node="minipve", access_method="pct-run", probe_target="authentik (CT 104, minipve)"), Guest(ct_id="110", hostname="gitea", ip="192.168.68.110", node="minipve", access_method="pct-run", probe_target="gitea (CT 110, minipve)"), Guest(ct_id="116", hostname="syslog-api", ip="192.168.68.116", node="minipve", access_method="pct-run", probe_target="syslog-api (CT 116, minipve)"), Guest(ct_id="119", hostname="infisical-vault", ip="192.168.68.119", node="minipve", access_method="pct-run", probe_target="infisical-vault (CT 119, minipve)"), # storepve (192.168.68.6) Guest(ct_id="106", hostname="ra-h-os", ip="192.168.68.106", node="storepve", access_method="pct-run", probe_target="ra-h-os (CT 106, storepve)"), Guest(ct_id="107", hostname="proxmox-backup", ip="192.168.68.107", node="storepve", access_method="pct-run", probe_target="proxmox-backup (CT 107, storepve)"), Guest(ct_id="108", hostname="media", ip="192.168.68.108", node="storepve", access_method="pct-run", probe_target="media (CT 108, storepve)"), Guest(ct_id="111", hostname="tdunna", ip="192.168.68.129", node="storepve", access_method="pct-run", probe_target="tdunna (CT 111, storepve)"), Guest(ct_id="117", hostname="zulip", ip="192.168.68.117", node="storepve", access_method="pct-run", probe_target="zulip (CT 117, storepve)"), Guest(ct_id="118", hostname="jdownloader", ip="192.168.68.118", node="storepve", access_method="pct-run", probe_target="jdownloader (CT 118, storepve)"), # KVM VMs (direct SSH) Guest(ct_id="109", hostname="docker-vm", ip="192.168.68.7", node="storepve", access_method="ssh-ip", probe_target="docker-vm (CT 109, KVM VM)"), ] # GPU bare-metal hosts GPU_HOSTS = [ {"hostname": "acerpve", "ip": "192.168.68.9", "gpu": "RTX 3090", "probe_target": "RTX 3090 (bare metal .9)"}, {"hostname": "ocupve", "ip": "192.168.68.110", "gpu": "RTX 5070", "probe_target": "RTX 5070 (bare metal .110)"}, {"hostname": "amdpve", "ip": "192.168.68.15", "gpu": "Strix Halo", "probe_target": "Strix Halo (bare metal .15)"}, ] CONNECT_TIMEOUT = 5 SSH_OPTS = "-o BatchMode=yes -o ConnectTimeout=" + str(CONNECT_TIMEOUT) # Host filesystem thresholds (from contract) HOST_THRESHOLDS = { "WARN": 85, "AMBER": 90, "RED": 95, } # State file path (absolute, so execution context doesn't matter) STATE_FILE = pathlib.Path(__file__).resolve().parent.parent / "state" / "host-disk-bands.json" # PVE nodes to probe for host filesystems HOST_NODES = [ {"hostname": "acerpve", "ip": "192.168.68.9"}, {"hostname": "amdpve", "ip": "192.168.68.15"}, {"hostname": "storepve", "ip": "192.168.68.6"}, {"hostname": "minipve", "ip": "192.168.68.12"}, {"hostname": "ocupve", "ip": "192.168.68.5"}, ] def run_cmd(cmd: str, timeout: int = 30) -> tuple[int, str, str]: """Run a command and return (exit_code, stdout, stderr).""" try: result = subprocess.run( cmd, shell=True, capture_output=True, text=True, timeout=timeout ) return result.returncode, result.stdout.strip(), result.stderr.strip() except subprocess.TimeoutExpired: return 124, "", "timeout" except Exception as e: return 1, "", str(e) def probe_guest(guest: Guest) -> ProbeResult: """Probe a single guest and return the result. Access method is selected from guest.access_method: - "pct-run": pct-run.sh "df -P / | tail -1" - "ssh-host": ssh root@ "df -P / | tail -1" - "ssh-ip": ssh root@ "df -P / | tail -1" """ df_cmd = "df -P / | tail -1" if guest.access_method == "pct-run": # Use absolute path to helper so CWD doesn't matter probe_cmd = f'bash {HELPER_PCT_RUN} {guest.ct_id} "{df_cmd}"' elif guest.access_method == "ssh-host": probe_cmd = f'ssh {SSH_OPTS} root@{guest.hostname} "{df_cmd}"' elif guest.access_method == "ssh-ip": probe_cmd = f'ssh {SSH_OPTS} root@{guest.ip} "{df_cmd}"' else: raise ValueError(f"unknown access_method: {guest.access_method}") # Check helper exists and is readable BEFORE probing (for pct-run guests) # This prevents scanner errors from being rendered as guest verdicts if guest.access_method == "pct-run": if not HELPER_PCT_RUN.exists(): print(f"SCANNER ERROR: helper not found: {HELPER_PCT_RUN}", file=sys.stderr) sys.exit(1) if not os.access(str(HELPER_PCT_RUN), os.R_OK): print(f"SCANNER ERROR: helper not readable: {HELPER_PCT_RUN}", file=sys.stderr) sys.exit(1) # Retry once on failure before declaring unreachable for attempt in range(2): exit_code, stdout, stderr = run_cmd(probe_cmd, timeout=15) if exit_code == 0: # Parse df output: Filesystem 1024-blocks Used Available Capacity Mounted on # parts[0]=Filesystem, parts[1]=Total (1K blocks), parts[2]=Used, parts[3]=Available, parts[4]=Capacity parts = stdout.split() if len(parts) >= 5: capacity_str = parts[4] # e.g., "34%" usage_pct = float(capacity_str.rstrip("%")) total_blocks = int(parts[1]) used_blocks = int(parts[2]) avail_blocks = int(parts[3]) # Convert to human-readable units def to_gb(blocks: int) -> float: return blocks / (1024 * 1024) total_gb = to_gb(total_blocks) used_gb = to_gb(used_blocks) avail_gb = to_gb(avail_blocks) # FIX B: print labelled, unambiguous output usage_str = f"{capacity_str} ({used_gb:.1f}G used of {total_gb:.1f}G total, {avail_gb:.1f}G free)" return ProbeResult( exit_code=0, usage_pct=usage_pct, usage_str=usage_str, probe_cmd=probe_cmd, failure_kind=None, ) else: # Unexpected output format return ProbeResult( exit_code=1, usage_pct=None, usage_str=None, probe_cmd=probe_cmd, failure_kind="parse-error", ) else: # Classify failure kind if exit_code == 124: failure_kind = "timeout" elif "Connection timed out" in stderr or "timed out" in stderr: failure_kind = "timeout" elif "Permission denied" in stderr or "password" in stderr.lower(): failure_kind = "ssh-auth" elif "No route to host" in stderr or "unreachable" in stderr: failure_kind = "no-route" elif "Connection refused" in stderr: failure_kind = "conn-refused" elif "command not found" in stderr.lower() or "No such file" in stderr: failure_kind = "command-not-found" else: failure_kind = f"ssh-exit-{exit_code}" # Retry once if attempt == 0: time.sleep(1) continue return ProbeResult( exit_code=exit_code, usage_pct=None, usage_str=None, probe_cmd=probe_cmd, failure_kind=failure_kind, ) # Should not reach here, but just in case return ProbeResult( exit_code=1, usage_pct=None, usage_str=None, probe_cmd=probe_cmd, failure_kind="unknown", ) def scan_fleet() -> list[dict]: """Scan all guests and return the results.""" results = [] for guest in GUESTS: probe_result = probe_guest(guest) guest.probe_result = probe_result row = { "target": guest.probe_target, "ct_id": guest.ct_id, "hostname": guest.hostname, "node": guest.node, "access_method": guest.access_method, "reachable": probe_result.is_reachable, "usage_pct": probe_result.usage_pct, "usage_str": probe_result.usage_str, "probe_cmd": probe_result.probe_cmd, "failure_kind": probe_result.failure_kind, } results.append(row) return results def classify_band(usage_pct: float) -> str: """Classify a percentage into a band.""" if usage_pct >= HOST_THRESHOLDS["RED"]: return "HOST-RED" elif usage_pct >= HOST_THRESHOLDS["AMBER"]: return "HOST-AMBER" elif usage_pct >= HOST_THRESHOLDS["WARN"]: return "HOST-WARN" else: return "GREEN" def probe_host_filesystems() -> tuple[list[dict], dict[str, str]]: """Probe host filesystems on all PVE nodes. Returns: - List of host filesystem results - Dict of volume_key -> current_band (for state file) """ results = [] current_bands = {} for node in HOST_NODES: ip = node["ip"] hostname = node["hostname"] # Probe df for host filesystems probe_cmd = f'ssh {SSH_OPTS} root@{ip} "df -hP / /media/* tank 2>/dev/null | tail -n +2"' exit_code, stdout, stderr = run_cmd(probe_cmd, timeout=15) if exit_code != 0: results.append({ "target": f"{hostname} ({ip})", "hostname": hostname, "ip": ip, "reachable": False, "volumes": [], "probe_cmd": probe_cmd, "failure_kind": "ssh-error", }) continue # Parse df output and classify each volume volumes = [] for line in stdout.splitlines(): parts = line.split() if len(parts) < 6: continue dev, size, used, avail, pct_str, mount = parts[:6] pct = float(pct_str.rstrip("%")) band = classify_band(pct) # Volume type classification if mount.startswith("/media/"): vol_type = "media" elif mount == "/" or "pve" in dev: vol_type = "host-root" elif mount == "tank" or "tank" in mount: vol_type = "pbs-datastore" else: vol_type = "other" # Volume key for state file (host/volume) volume_key = f"{hostname}/{mount}" current_bands[volume_key] = band volumes.append({ "mount": mount, "device": dev, "size": size, "used": used, "avail": avail, "pct": pct, "band": band, "type": vol_type, }) results.append({ "target": f"{hostname} ({ip})", "hostname": hostname, "ip": ip, "reachable": True, "volumes": volumes, "probe_cmd": probe_cmd, "failure_kind": None, }) return results, current_bands def read_state_file() -> Optional[dict[str, str]]: """Read the state file if it exists.""" if not STATE_FILE.exists(): return None try: with open(STATE_FILE) as f: return json.load(f) except (json.JSONDecodeError, IOError) as e: print(f"⚠️ State file exists but unreadable: {e}", file=sys.stderr) return {} def write_state_file(bands: dict[str, str]) -> None: """Write the state file.""" STATE_FILE.parent.mkdir(parents=True, exist_ok=True) try: with open(STATE_FILE, "w") as f: json.dump(bands, f, indent=2) except IOError as e: print(f"⚠️ State file write failed: {e}", file=sys.stderr) def detect_transitions(current_bands: dict[str, str], prior_bands: Optional[dict[str, str]]) -> list[dict]: """Detect band transitions (current vs. prior).""" if prior_bands is None: # First run — no transitions, just establish baseline return [] transitions = [] # Check for volumes that moved to a higher band (escalation) for volume, current_band in current_bands.items(): prior_band = prior_bands.get(volume, "GREEN") # Band ordering: GREEN < HOST-WARN < HOST-AMBER < HOST-RED band_order = {"GREEN": 0, "HOST-WARN": 1, "HOST-AMBER": 2, "HOST-RED": 3} if band_order[current_band] > band_order[prior_band]: transitions.append({ "type": "escalation", "volume": volume, "from": prior_band, "to": current_band, }) elif band_order[current_band] < band_order[prior_band]: transitions.append({ "type": "recovery", "volume": volume, "from": prior_band, "to": current_band, }) return transitions def render_results(results: list[dict]) -> str: """Render scan results in human-readable format.""" lines = [] lines.append("=== Disk GC Scan ===") lines.append("") for row in results: if row["reachable"]: lines.append(f" ✅ {row['target']}: {row['usage_str']}") lines.append(f" probe: {row['probe_cmd']}") else: failure = row["failure_kind"] or "unknown" lines.append(f" ❌ {row['target']}: UNREACHABLE ({failure})") lines.append(f" probe: {row['probe_cmd']}") return "\n".join(lines) def render_host_results(results: list[dict], transitions: list[dict], prior_bands: Optional[dict[str, str]]) -> str: """Render host filesystem results in human-readable format.""" lines = [] lines.append("") lines.append("=== Host Filesystem Bands ===") lines.append("") # Render transitions first (they're the actionable alerts) if prior_bands is None: lines.append(" (first run — recording baseline, no alerts)") elif not transitions: lines.append(" (no band changes since last scan)") else: for t in transitions: volume, from_band, to_band = t["volume"], t["from"], t["to"] if t["type"] == "escalation": lines.append(f" ⚠️ {volume}: {from_band} → {to_band} (ESCALATION)") else: lines.append(f" ✅ {volume}: {from_band} → {to_band} (RECOVERY)") # Render all volumes with their bands lines.append("") for node_result in results: if not node_result["reachable"]: lines.append(f" ❌ {node_result['target']}: UNREACHABLE ({node_result['failure_kind']})") continue lines.append(f" {node_result['target']}:") for vol in node_result["volumes"]: lines.append(f" {vol['mount']} ({vol['type']}): {vol['pct']}% ({vol['used']}/{vol['size']}, {vol['avail']} free) -> {vol['band']}") return "\n".join(lines) def main() -> int: import argparse ap = argparse.ArgumentParser(description="Deterministic disk usage probe for fleet guests and host filesystems.") ap.add_argument("--json", action="store_true", help="machine-readable output") ap.add_argument("--hosts-only", action="store_true", help="scan host filesystems only") ap.add_argument("--guests-only", action="store_true", help="scan guests only (skip host filesystems)") args = ap.parse_args() # Scan guests (unless --hosts-only) guest_results = [] if not args.hosts_only: guest_results = scan_fleet() # Scan host filesystems (unless --guests-only) host_results = [] current_bands = {} if not args.guests_only: host_results, current_bands = probe_host_filesystems() # Read prior state and detect transitions prior_bands = read_state_file() transitions = detect_transitions(current_bands, prior_bands) # Write new state write_state_file(current_bands) else: prior_bands = None transitions = [] if args.json: # JSON output output = { "guests": guest_results, "hosts": host_results, "transitions": transitions, "prior_bands": prior_bands, } print(json.dumps(output, indent=2)) else: # Human-readable output if guest_results: print(render_results(guest_results)) if host_results: print(render_host_results(host_results, transitions, prior_bands)) # Exit 0 if all probed (reachable or not), 1 if any probe error return 0 if __name__ == "__main__": sys.exit(main())