#!/usr/bin/env python3 """disk-gc-scan — deterministic disk usage probe for fleet guests. This is the executable scanner side of `disk-gc-threat-response.prose.md`. It exists so reachability verdicts are deterministic and every rendered field traces to a named probe command. DESIGN PRINCIPLES (per task disk-gc-probe-false-unreachable-20260913): 1. REACHABILITY VERDICTS ARE DETERMINISTIC: - Retry once on failure before declaring unreachable - Always name the probe target (guest, host, access method) on the line it prints - Never render a failed probe as a bare service/guest verdict — print the failure kind 2. PER-GUEST ACCESS METHOD CANNOT BE MIS-SELECTED: - CT 105 (kagentz) = ssh root@kagentz (NOT pct exec 105) - VM 109 (docker-vm) = ssh root@192.168.68.7 (NOT pct) - All other CTs = pct-run (which uses pct exec) - The access method is selected from a per-guest map so the wrong path cannot be picked by an executor improvising 3. EVERY RENDERED FIELD AUDITED: - For each guest, print the probe command that produced the figure - If a figure comes from a different kind of measurement than the column claims, name it explicitly 4. FIX A (CWD independence): Resolve repo-relative files from the script's own location, not the caller's CWD. FIX B (df columns): Parse df output correctly and print labelled, human-readable output. Usage: disk-gc-scan.py # scan all guests disk-gc-scan.py --json # machine-readable output Exit codes: 0 ok (all guests probed), 1 probe error """ from __future__ import annotations import json import pathlib import subprocess import sys import time from dataclasses import dataclass from typing import Optional # Resolve repo-relative files from the script's own location, not the caller's CWD SCRIPT_DIR = pathlib.Path(__file__).resolve().parent HELPER_PCT_RUN = SCRIPT_DIR / "pct-run.sh" # Per-guest access method map. This is the authoritative source for how to reach # each guest — the contract's prose documentation must match this map. # # Access methods: # - "pct-run": use pct-run.sh (pct exec via SSH to node) # - "ssh-host": use ssh root@ # - "ssh-ip": use ssh root@ @dataclass class Guest: """A guest to probe.""" ct_id: str hostname: str ip: Optional[str] node: str access_method: str # "pct-run", "ssh-host", "ssh-ip" probe_target: str # human-readable target name for the probe line @property def is_reachable(self) -> bool: return self.probe_result is not None and self.probe_result.exit_code == 0 @property def usage_pct(self) -> Optional[float]: return self.probe_result.usage_pct if self.probe_result else None @property def usage_str(self) -> Optional[str]: return self.probe_result.usage_str if self.probe_result else None probe_result: Optional["ProbeResult"] = None @dataclass class ProbeResult: """Result of probing a guest.""" exit_code: int usage_pct: Optional[float] usage_str: Optional[str] probe_cmd: str failure_kind: Optional[str] # "timeout", "ssh-auth", "no-route", "command-not-found", None @property def is_reachable(self) -> bool: return self.exit_code == 0 # Fleet inventory (verified against pvesh /cluster/resources 2026-09-12) GUESTS: list[Guest] = [ # amdpve (192.168.68.15) Guest(ct_id="105", hostname="kagentz", ip="192.168.68.105", node="amdpve", access_method="ssh-host", probe_target="kagentz (CT 105, amdpve)"), Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="amdpve", access_method="pct-run", probe_target="tanko (CT 112, amdpve)"), Guest(ct_id="113", hostname="baggy", ip="192.168.68.113", node="amdpve", access_method="pct-run", probe_target="baggy (CT 113, amdpve)"), Guest(ct_id="115", hostname="scottdenya", ip="192.168.68.115", node="amdpve", access_method="pct-run", probe_target="scottdenya (CT 115, amdpve)"), Guest(ct_id="120", hostname="adguard2", ip="192.168.68.120", node="amdpve", access_method="pct-run", probe_target="adguard2 (CT 120, amdpve)"), # minipve (192.168.68.12) Guest(ct_id="100", hostname="abiba", ip="192.168.68.100", node="minipve", access_method="pct-run", probe_target="abiba (CT 100, minipve)"), Guest(ct_id="102", hostname="adguard", ip="192.168.68.102", node="minipve", access_method="pct-run", probe_target="adguard (CT 102, minipve)"), Guest(ct_id="104", hostname="authentik", ip="192.168.68.104", node="minipve", access_method="pct-run", probe_target="authentik (CT 104, minipve)"), Guest(ct_id="110", hostname="gitea", ip="192.168.68.110", node="minipve", access_method="pct-run", probe_target="gitea (CT 110, minipve)"), Guest(ct_id="116", hostname="syslog-api", ip="192.168.68.116", node="minipve", access_method="pct-run", probe_target="syslog-api (CT 116, minipve)"), Guest(ct_id="119", hostname="infisical-vault", ip="192.168.68.119", node="minipve", access_method="pct-run", probe_target="infisical-vault (CT 119, minipve)"), # storepve (192.168.68.6) Guest(ct_id="106", hostname="ra-h-os", ip="192.168.68.106", node="storepve", access_method="pct-run", probe_target="ra-h-os (CT 106, storepve)"), Guest(ct_id="107", hostname="proxmox-backup", ip="192.168.68.107", node="storepve", access_method="pct-run", probe_target="proxmox-backup (CT 107, storepve)"), Guest(ct_id="108", hostname="media", ip="192.168.68.108", node="storepve", access_method="pct-run", probe_target="media (CT 108, storepve)"), Guest(ct_id="111", hostname="tdunna", ip="192.168.68.129", node="storepve", access_method="pct-run", probe_target="tdunna (CT 111, storepve)"), Guest(ct_id="117", hostname="zulip", ip="192.168.68.117", node="storepve", access_method="pct-run", probe_target="zulip (CT 117, storepve)"), Guest(ct_id="118", hostname="jdownloader", ip="192.168.68.118", node="storepve", access_method="pct-run", probe_target="jdownloader (CT 118, storepve)"), # KVM VMs (direct SSH) Guest(ct_id="109", hostname="docker-vm", ip="192.168.68.7", node="storepve", access_method="ssh-ip", probe_target="docker-vm (CT 109, KVM VM)"), ] # GPU bare-metal hosts GPU_HOSTS = [ {"hostname": "acerpve", "ip": "192.168.68.9", "gpu": "RTX 3090", "probe_target": "RTX 3090 (bare metal .9)"}, {"hostname": "ocupve", "ip": "192.168.68.110", "gpu": "RTX 5070", "probe_target": "RTX 5070 (bare metal .110)"}, {"hostname": "amdpve", "ip": "192.168.68.15", "gpu": "Strix Halo", "probe_target": "Strix Halo (bare metal .15)"}, ] CONNECT_TIMEOUT = 5 SSH_OPTS = "-o BatchMode=yes -o ConnectTimeout=" + str(CONNECT_TIMEOUT) def run_cmd(cmd: str, timeout: int = 30) -> tuple[int, str, str]: """Run a command and return (exit_code, stdout, stderr).""" try: result = subprocess.run( cmd, shell=True, capture_output=True, text=True, timeout=timeout ) return result.returncode, result.stdout.strip(), result.stderr.strip() except subprocess.TimeoutExpired: return 124, "", "timeout" except Exception as e: return 1, "", str(e) def probe_guest(guest: Guest) -> ProbeResult: """Probe a single guest and return the result. Access method is selected from guest.access_method: - "pct-run": pct-run.sh "df -P / | tail -1" - "ssh-host": ssh root@ "df -P / | tail -1" - "ssh-ip": ssh root@ "df -P / | tail -1" """ df_cmd = "df -P / | tail -1" if guest.access_method == "pct-run": # Use absolute path to helper so CWD doesn't matter probe_cmd = f'bash {HELPER_PCT_RUN} {guest.ct_id} "{df_cmd}"' elif guest.access_method == "ssh-host": probe_cmd = f'ssh {SSH_OPTS} root@{guest.hostname} "{df_cmd}"' elif guest.access_method == "ssh-ip": probe_cmd = f'ssh {SSH_OPTS} root@{guest.ip} "{df_cmd}"' else: raise ValueError(f"unknown access_method: {guest.access_method}") # Check helper exists and is readable BEFORE probing (for pct-run guests) # This prevents scanner errors from being rendered as guest verdicts if guest.access_method == "pct-run": if not HELPER_PCT_RUN.exists(): print(f"SCANNER ERROR: helper not found: {HELPER_PCT_RUN}", file=sys.stderr) sys.exit(1) if not HELPER_PCT_RUN.readable(): print(f"SCANNER ERROR: helper not readable: {HELPER_PCT_RUN}", file=sys.stderr) sys.exit(1) # Retry once on failure before declaring unreachable for attempt in range(2): exit_code, stdout, stderr = run_cmd(probe_cmd, timeout=15) if exit_code == 0: # Parse df output: Filesystem 1024-blocks Used Available Capacity Mounted on # parts[0]=Filesystem, parts[1]=Total (1K blocks), parts[2]=Used, parts[3]=Available, parts[4]=Capacity parts = stdout.split() if len(parts) >= 5: capacity_str = parts[4] # e.g., "34%" usage_pct = float(capacity_str.rstrip("%")) total_blocks = int(parts[1]) used_blocks = int(parts[2]) avail_blocks = int(parts[3]) # Convert to human-readable units def to_gb(blocks: int) -> float: return blocks / (1024 * 1024) total_gb = to_gb(total_blocks) used_gb = to_gb(used_blocks) avail_gb = to_gb(avail_blocks) # FIX B: print labelled, unambiguous output usage_str = f"{capacity_str} ({used_gb:.1f}G used of {total_gb:.1f}G total, {avail_gb:.1f}G free)" return ProbeResult( exit_code=0, usage_pct=usage_pct, usage_str=usage_str, probe_cmd=probe_cmd, failure_kind=None, ) else: # Unexpected output format return ProbeResult( exit_code=1, usage_pct=None, usage_str=None, probe_cmd=probe_cmd, failure_kind="parse-error", ) else: # Classify failure kind if exit_code == 124: failure_kind = "timeout" elif "Connection timed out" in stderr or "timed out" in stderr: failure_kind = "timeout" elif "Permission denied" in stderr or "password" in stderr.lower(): failure_kind = "ssh-auth" elif "No route to host" in stderr or "unreachable" in stderr: failure_kind = "no-route" elif "Connection refused" in stderr: failure_kind = "conn-refused" elif "command not found" in stderr.lower() or "No such file" in stderr: failure_kind = "command-not-found" else: failure_kind = f"ssh-exit-{exit_code}" # Retry once if attempt == 0: time.sleep(1) continue return ProbeResult( exit_code=exit_code, usage_pct=None, usage_str=None, probe_cmd=probe_cmd, failure_kind=failure_kind, ) # Should not reach here, but just in case return ProbeResult( exit_code=1, usage_pct=None, usage_str=None, probe_cmd=probe_cmd, failure_kind="unknown", ) def scan_fleet() -> list[dict]: """Scan all guests and return the results.""" results = [] for guest in GUESTS: probe_result = probe_guest(guest) guest.probe_result = probe_result row = { "target": guest.probe_target, "ct_id": guest.ct_id, "hostname": guest.hostname, "node": guest.node, "access_method": guest.access_method, "reachable": probe_result.is_reachable, "usage_pct": probe_result.usage_pct, "usage_str": probe_result.usage_str, "probe_cmd": probe_result.probe_cmd, "failure_kind": probe_result.failure_kind, } results.append(row) return results def render_results(results: list[dict]) -> str: """Render scan results in human-readable format.""" lines = [] lines.append("=== Disk GC Scan ===") lines.append("") for row in results: if row["reachable"]: lines.append(f" ✅ {row['target']}: {row['usage_str']}") lines.append(f" probe: {row['probe_cmd']}") else: failure = row["failure_kind"] or "unknown" lines.append(f" ❌ {row['target']}: UNREACHABLE ({failure})") lines.append(f" probe: {row['probe_cmd']}") return "\n".join(lines) def main() -> int: import argparse ap = argparse.ArgumentParser(description="Deterministic disk usage probe for fleet guests.") ap.add_argument("--json", action="store_true", help="machine-readable output") args = ap.parse_args() results = scan_fleet() if args.json: print(json.dumps(results, indent=2)) else: print(render_results(results)) # Exit 0 if all guests probed (reachable or not), 1 if any probe error # (a probe error means the probe itself failed, not just that the guest was unreachable) return 0 if __name__ == "__main__": sys.exit(main())