Files
prose-contracts/scripts/disk-gc-scan.py
T
root 96769a103f
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 12s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 9s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 17s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
no-mistakes(review): sync tanko CT 112 mapping to minipve across consumers
2026-09-28 06:47:46 +00:00

570 lines
21 KiB
Python

#!/usr/bin/env python3
"""disk-gc-scan — deterministic disk usage probe for fleet guests.
This is the executable scanner side of `disk-gc-threat-response.prose.md`. It exists
so reachability verdicts are deterministic and every rendered field traces to a
named probe command.
DESIGN PRINCIPLES (per task disk-gc-probe-false-unreachable-20260913):
1. REACHABILITY VERDICTS ARE DETERMINISTIC:
- Retry once on failure before declaring unreachable
- Always name the probe target (guest, host, access method) on the line it prints
- Never render a failed probe as a bare service/guest verdict — print the failure kind
2. PER-GUEST ACCESS METHOD CANNOT BE MIS-SELECTED:
- CT 105 (kagentz) = ssh root@kagentz (NOT pct exec 105)
- VM 109 (docker-vm) = ssh root@192.168.68.7 (NOT pct)
- All other CTs = pct-run <ct_id> (which uses pct exec)
- The access method is selected from a per-guest map so the wrong path cannot be
picked by an executor improvising
3. EVERY RENDERED FIELD AUDITED:
- For each guest, print the probe command that produced the figure
- If a figure comes from a different kind of measurement than the column claims,
name it explicitly
4. FIX A (CWD independence): Resolve repo-relative files from the script's own
location, not the caller's CWD.
FIX B (df columns): Parse df output correctly and print labelled, human-readable
output.
Usage:
disk-gc-scan.py # scan all guests
disk-gc-scan.py --json # machine-readable output
Exit codes: 0 ok (all guests probed), 1 probe error
"""
from __future__ import annotations
import json
import pathlib
import subprocess
import sys
import time
import os
from dataclasses import dataclass
from typing import Optional
# Resolve repo-relative files from the script's own location, not the caller's CWD
SCRIPT_DIR = pathlib.Path(__file__).resolve().parent
HELPER_PCT_RUN = SCRIPT_DIR / "pct-run.sh"
# Per-guest access method map. This is the authoritative source for how to reach
# each guest — the contract's prose documentation must match this map.
#
# Access methods:
# - "pct-run": use pct-run.sh <ct_id> (pct exec via SSH to node)
# - "ssh-host": use ssh root@<hostname>
# - "ssh-ip": use ssh root@<ip>
@dataclass
class Guest:
"""A guest to probe."""
ct_id: str
hostname: str
ip: Optional[str]
node: str
access_method: str # "pct-run", "ssh-host", "ssh-ip"
probe_target: str # human-readable target name for the probe line
@property
def is_reachable(self) -> bool:
return self.probe_result is not None and self.probe_result.exit_code == 0
@property
def usage_pct(self) -> Optional[float]:
return self.probe_result.usage_pct if self.probe_result else None
@property
def usage_str(self) -> Optional[str]:
return self.probe_result.usage_str if self.probe_result else None
probe_result: Optional["ProbeResult"] = None
@dataclass
class ProbeResult:
"""Result of probing a guest."""
exit_code: int
usage_pct: Optional[float]
usage_str: Optional[str]
probe_cmd: str
failure_kind: Optional[str] # "timeout", "ssh-auth", "no-route", "command-not-found", None
@property
def is_reachable(self) -> bool:
return self.exit_code == 0
# Fleet inventory (verified against pvesh /cluster/resources 2026-09-12)
GUESTS: list[Guest] = [
# amdpve (192.168.68.15)
Guest(ct_id="105", hostname="kagentz", ip="192.168.68.105", node="amdpve",
access_method="ssh-host", probe_target="kagentz (CT 105, amdpve)"),
Guest(ct_id="113", hostname="baggy", ip="192.168.68.113", node="amdpve",
access_method="pct-run", probe_target="baggy (CT 113, amdpve)"),
Guest(ct_id="115", hostname="scottdenya", ip="192.168.68.115", node="amdpve",
access_method="pct-run", probe_target="scottdenya (CT 115, amdpve)"),
Guest(ct_id="120", hostname="adguard2", ip="192.168.68.120", node="amdpve",
access_method="pct-run", probe_target="adguard2 (CT 120, amdpve)"),
# minipve (192.168.68.12)
Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="minipve",
access_method="pct-run", probe_target="tanko (CT 112, minipve)"),
Guest(ct_id="100", hostname="abiba", ip="192.168.68.100", node="minipve",
access_method="pct-run", probe_target="abiba (CT 100, minipve)"),
Guest(ct_id="102", hostname="adguard", ip="192.168.68.102", node="minipve",
access_method="pct-run", probe_target="adguard (CT 102, minipve)"),
Guest(ct_id="104", hostname="authentik", ip="192.168.68.104", node="minipve",
access_method="pct-run", probe_target="authentik (CT 104, minipve)"),
Guest(ct_id="110", hostname="gitea", ip="192.168.68.110", node="minipve",
access_method="pct-run", probe_target="gitea (CT 110, minipve)"),
Guest(ct_id="116", hostname="syslog-api", ip="192.168.68.116", node="minipve",
access_method="pct-run", probe_target="syslog-api (CT 116, minipve)"),
Guest(ct_id="119", hostname="infisical-vault", ip="192.168.68.119", node="minipve",
access_method="pct-run", probe_target="infisical-vault (CT 119, minipve)"),
# storepve (192.168.68.6)
Guest(ct_id="106", hostname="ra-h-os", ip="192.168.68.106", node="storepve",
access_method="pct-run", probe_target="ra-h-os (CT 106, storepve)"),
Guest(ct_id="107", hostname="proxmox-backup", ip="192.168.68.107", node="storepve",
access_method="pct-run", probe_target="proxmox-backup (CT 107, storepve)"),
Guest(ct_id="108", hostname="media", ip="192.168.68.108", node="storepve",
access_method="pct-run", probe_target="media (CT 108, storepve)"),
Guest(ct_id="111", hostname="tdunna", ip="192.168.68.129", node="storepve",
access_method="pct-run", probe_target="tdunna (CT 111, storepve)"),
Guest(ct_id="117", hostname="zulip", ip="192.168.68.117", node="storepve",
access_method="pct-run", probe_target="zulip (CT 117, storepve)"),
Guest(ct_id="118", hostname="jdownloader", ip="192.168.68.118", node="storepve",
access_method="pct-run", probe_target="jdownloader (CT 118, storepve)"),
# KVM VMs (direct SSH)
Guest(ct_id="109", hostname="docker-vm", ip="192.168.68.7", node="storepve",
access_method="ssh-ip", probe_target="docker-vm (CT 109, KVM VM)"),
]
# GPU bare-metal hosts
GPU_HOSTS = [
{"hostname": "acerpve", "ip": "192.168.68.9", "gpu": "RTX 3090",
"probe_target": "RTX 3090 (bare metal .9)"},
{"hostname": "ocupve", "ip": "192.168.68.110", "gpu": "RTX 5070",
"probe_target": "RTX 5070 (bare metal .110)"},
{"hostname": "amdpve", "ip": "192.168.68.15", "gpu": "Strix Halo",
"probe_target": "Strix Halo (bare metal .15)"},
]
CONNECT_TIMEOUT = 5
SSH_OPTS = "-o BatchMode=yes -o ConnectTimeout=" + str(CONNECT_TIMEOUT)
# Host filesystem thresholds (from contract)
HOST_THRESHOLDS = {
"WARN": 85,
"AMBER": 90,
"RED": 95,
}
# State file path (absolute, so execution context doesn't matter)
STATE_FILE = pathlib.Path(__file__).resolve().parent.parent / "state" / "host-disk-bands.json"
# PVE nodes to probe for host filesystems
HOST_NODES = [
{"hostname": "acerpve", "ip": "192.168.68.9"},
{"hostname": "amdpve", "ip": "192.168.68.15"},
{"hostname": "storepve", "ip": "192.168.68.6"},
{"hostname": "minipve", "ip": "192.168.68.12"},
{"hostname": "ocupve", "ip": "192.168.68.5"},
]
def run_cmd(cmd: str, timeout: int = 30) -> tuple[int, str, str]:
"""Run a command and return (exit_code, stdout, stderr)."""
try:
result = subprocess.run(
cmd, shell=True, capture_output=True, text=True, timeout=timeout
)
return result.returncode, result.stdout.strip(), result.stderr.strip()
except subprocess.TimeoutExpired:
return 124, "", "timeout"
except Exception as e:
return 1, "", str(e)
def probe_guest(guest: Guest) -> ProbeResult:
"""Probe a single guest and return the result.
Access method is selected from guest.access_method:
- "pct-run": pct-run.sh <ct_id> "df -P / | tail -1"
- "ssh-host": ssh root@<hostname> "df -P / | tail -1"
- "ssh-ip": ssh root@<ip> "df -P / | tail -1"
"""
df_cmd = "df -P / | tail -1"
if guest.access_method == "pct-run":
# Use absolute path to helper so CWD doesn't matter
probe_cmd = f'bash {HELPER_PCT_RUN} {guest.ct_id} "{df_cmd}"'
elif guest.access_method == "ssh-host":
probe_cmd = f'ssh {SSH_OPTS} root@{guest.hostname} "{df_cmd}"'
elif guest.access_method == "ssh-ip":
probe_cmd = f'ssh {SSH_OPTS} root@{guest.ip} "{df_cmd}"'
else:
raise ValueError(f"unknown access_method: {guest.access_method}")
# Check helper exists and is readable BEFORE probing (for pct-run guests)
# This prevents scanner errors from being rendered as guest verdicts
if guest.access_method == "pct-run":
if not HELPER_PCT_RUN.exists():
print(f"SCANNER ERROR: helper not found: {HELPER_PCT_RUN}", file=sys.stderr)
sys.exit(1)
if not os.access(str(HELPER_PCT_RUN), os.R_OK):
print(f"SCANNER ERROR: helper not readable: {HELPER_PCT_RUN}", file=sys.stderr)
sys.exit(1)
# Retry once on failure before declaring unreachable
for attempt in range(2):
exit_code, stdout, stderr = run_cmd(probe_cmd, timeout=15)
if exit_code == 0:
# Parse df output: Filesystem 1024-blocks Used Available Capacity Mounted on
# parts[0]=Filesystem, parts[1]=Total (1K blocks), parts[2]=Used, parts[3]=Available, parts[4]=Capacity
parts = stdout.split()
if len(parts) >= 5:
capacity_str = parts[4] # e.g., "34%"
usage_pct = float(capacity_str.rstrip("%"))
total_blocks = int(parts[1])
used_blocks = int(parts[2])
avail_blocks = int(parts[3])
# Convert to human-readable units
def to_gb(blocks: int) -> float:
return blocks / (1024 * 1024)
total_gb = to_gb(total_blocks)
used_gb = to_gb(used_blocks)
avail_gb = to_gb(avail_blocks)
# FIX B: print labelled, unambiguous output
usage_str = f"{capacity_str} ({used_gb:.1f}G used of {total_gb:.1f}G total, {avail_gb:.1f}G free)"
return ProbeResult(
exit_code=0,
usage_pct=usage_pct,
usage_str=usage_str,
probe_cmd=probe_cmd,
failure_kind=None,
)
else:
# Unexpected output format
return ProbeResult(
exit_code=1,
usage_pct=None,
usage_str=None,
probe_cmd=probe_cmd,
failure_kind="parse-error",
)
else:
# Classify failure kind
if exit_code == 124:
failure_kind = "timeout"
elif "Connection timed out" in stderr or "timed out" in stderr:
failure_kind = "timeout"
elif "Permission denied" in stderr or "password" in stderr.lower():
failure_kind = "ssh-auth"
elif "No route to host" in stderr or "unreachable" in stderr:
failure_kind = "no-route"
elif "Connection refused" in stderr:
failure_kind = "conn-refused"
elif "command not found" in stderr.lower() or "No such file" in stderr:
failure_kind = "command-not-found"
else:
failure_kind = f"ssh-exit-{exit_code}"
# Retry once
if attempt == 0:
time.sleep(1)
continue
return ProbeResult(
exit_code=exit_code,
usage_pct=None,
usage_str=None,
probe_cmd=probe_cmd,
failure_kind=failure_kind,
)
# Should not reach here, but just in case
return ProbeResult(
exit_code=1,
usage_pct=None,
usage_str=None,
probe_cmd=probe_cmd,
failure_kind="unknown",
)
def scan_fleet() -> list[dict]:
"""Scan all guests and return the results."""
results = []
for guest in GUESTS:
probe_result = probe_guest(guest)
guest.probe_result = probe_result
row = {
"target": guest.probe_target,
"ct_id": guest.ct_id,
"hostname": guest.hostname,
"node": guest.node,
"access_method": guest.access_method,
"reachable": probe_result.is_reachable,
"usage_pct": probe_result.usage_pct,
"usage_str": probe_result.usage_str,
"probe_cmd": probe_result.probe_cmd,
"failure_kind": probe_result.failure_kind,
}
results.append(row)
return results
def classify_band(usage_pct: float) -> str:
"""Classify a percentage into a band."""
if usage_pct >= HOST_THRESHOLDS["RED"]:
return "HOST-RED"
elif usage_pct >= HOST_THRESHOLDS["AMBER"]:
return "HOST-AMBER"
elif usage_pct >= HOST_THRESHOLDS["WARN"]:
return "HOST-WARN"
else:
return "GREEN"
def probe_host_filesystems() -> tuple[list[dict], dict[str, str]]:
"""Probe host filesystems on all PVE nodes.
Returns:
- List of host filesystem results
- Dict of volume_key -> current_band (for state file)
"""
results = []
current_bands = {}
for node in HOST_NODES:
ip = node["ip"]
hostname = node["hostname"]
# Probe df for host filesystems
probe_cmd = f'ssh {SSH_OPTS} root@{ip} "df -hP / /media/* tank 2>/dev/null | tail -n +2"'
exit_code, stdout, stderr = run_cmd(probe_cmd, timeout=15)
if exit_code != 0:
results.append({
"target": f"{hostname} ({ip})",
"hostname": hostname,
"ip": ip,
"reachable": False,
"volumes": [],
"probe_cmd": probe_cmd,
"failure_kind": "ssh-error",
})
continue
# Parse df output and classify each volume
volumes = []
for line in stdout.splitlines():
parts = line.split()
if len(parts) < 6:
continue
dev, size, used, avail, pct_str, mount = parts[:6]
pct = float(pct_str.rstrip("%"))
band = classify_band(pct)
# Volume type classification
if mount.startswith("/media/"):
vol_type = "media"
elif mount == "/" or "pve" in dev:
vol_type = "host-root"
elif mount == "tank" or "tank" in mount:
vol_type = "pbs-datastore"
else:
vol_type = "other"
# Volume key for state file (host/volume)
volume_key = f"{hostname}/{mount}"
current_bands[volume_key] = band
volumes.append({
"mount": mount,
"device": dev,
"size": size,
"used": used,
"avail": avail,
"pct": pct,
"band": band,
"type": vol_type,
})
results.append({
"target": f"{hostname} ({ip})",
"hostname": hostname,
"ip": ip,
"reachable": True,
"volumes": volumes,
"probe_cmd": probe_cmd,
"failure_kind": None,
})
return results, current_bands
def read_state_file() -> Optional[dict[str, str]]:
"""Read the state file if it exists."""
if not STATE_FILE.exists():
return None
try:
with open(STATE_FILE) as f:
return json.load(f)
except (json.JSONDecodeError, IOError) as e:
print(f"⚠️ State file exists but unreadable: {e}", file=sys.stderr)
return {}
def write_state_file(bands: dict[str, str]) -> None:
"""Write the state file."""
STATE_FILE.parent.mkdir(parents=True, exist_ok=True)
try:
with open(STATE_FILE, "w") as f:
json.dump(bands, f, indent=2)
except IOError as e:
print(f"⚠️ State file write failed: {e}", file=sys.stderr)
def detect_transitions(current_bands: dict[str, str], prior_bands: Optional[dict[str, str]]) -> list[dict]:
"""Detect band transitions (current vs. prior)."""
if prior_bands is None:
# First run — no transitions, just establish baseline
return []
transitions = []
# Check for volumes that moved to a higher band (escalation)
for volume, current_band in current_bands.items():
prior_band = prior_bands.get(volume, "GREEN")
# Band ordering: GREEN < HOST-WARN < HOST-AMBER < HOST-RED
band_order = {"GREEN": 0, "HOST-WARN": 1, "HOST-AMBER": 2, "HOST-RED": 3}
if band_order[current_band] > band_order[prior_band]:
transitions.append({
"type": "escalation",
"volume": volume,
"from": prior_band,
"to": current_band,
})
elif band_order[current_band] < band_order[prior_band]:
transitions.append({
"type": "recovery",
"volume": volume,
"from": prior_band,
"to": current_band,
})
return transitions
def render_results(results: list[dict]) -> str:
"""Render scan results in human-readable format."""
lines = []
lines.append("=== Disk GC Scan ===")
lines.append("")
for row in results:
if row["reachable"]:
lines.append(f" ✅ {row['target']}: {row['usage_str']}")
lines.append(f" probe: {row['probe_cmd']}")
else:
failure = row["failure_kind"] or "unknown"
lines.append(f" ❌ {row['target']}: UNREACHABLE ({failure})")
lines.append(f" probe: {row['probe_cmd']}")
return "\n".join(lines)
def render_host_results(results: list[dict], transitions: list[dict], prior_bands: Optional[dict[str, str]]) -> str:
"""Render host filesystem results in human-readable format."""
lines = []
lines.append("")
lines.append("=== Host Filesystem Bands ===")
lines.append("")
# Render transitions first (they're the actionable alerts)
if prior_bands is None:
lines.append(" (first run — recording baseline, no alerts)")
elif not transitions:
lines.append(" (no band changes since last scan)")
else:
for t in transitions:
volume, from_band, to_band = t["volume"], t["from"], t["to"]
if t["type"] == "escalation":
lines.append(f" ⚠️ {volume}: {from_band} → {to_band} (ESCALATION)")
else:
lines.append(f" ✅ {volume}: {from_band} → {to_band} (RECOVERY)")
# Render all volumes with their bands
lines.append("")
for node_result in results:
if not node_result["reachable"]:
lines.append(f" ❌ {node_result['target']}: UNREACHABLE ({node_result['failure_kind']})")
continue
lines.append(f" {node_result['target']}:")
for vol in node_result["volumes"]:
lines.append(f" {vol['mount']} ({vol['type']}): {vol['pct']}% ({vol['used']}/{vol['size']}, {vol['avail']} free) -> {vol['band']}")
return "\n".join(lines)
def main() -> int:
import argparse
ap = argparse.ArgumentParser(description="Deterministic disk usage probe for fleet guests and host filesystems.")
ap.add_argument("--json", action="store_true", help="machine-readable output")
ap.add_argument("--hosts-only", action="store_true", help="scan host filesystems only")
ap.add_argument("--guests-only", action="store_true", help="scan guests only (skip host filesystems)")
args = ap.parse_args()
# Scan guests (unless --hosts-only)
guest_results = []
if not args.hosts_only:
guest_results = scan_fleet()
# Scan host filesystems (unless --guests-only)
host_results = []
current_bands = {}
if not args.guests_only:
host_results, current_bands = probe_host_filesystems()
# Read prior state and detect transitions
prior_bands = read_state_file()
transitions = detect_transitions(current_bands, prior_bands)
# Write new state
write_state_file(current_bands)
else:
prior_bands = None
transitions = []
if args.json:
# JSON output
output = {
"guests": guest_results,
"hosts": host_results,
"transitions": transitions,
"prior_bands": prior_bands,
}
print(json.dumps(output, indent=2))
else:
# Human-readable output
if guest_results:
print(render_results(guest_results))
if host_results:
print(render_host_results(host_results, transitions, prior_bands))
# Exit 0 if all probed (reachable or not), 1 if any probe error
return 0
if __name__ == "__main__":
sys.exit(main())