diff --git a/infrastructure-update.prose.md b/infrastructure-update.prose.md index 01667fe..7913034 100644 --- a/infrastructure-update.prose.md +++ b/infrastructure-update.prose.md @@ -3,15 +3,16 @@ kind: responsibility name: infrastructure-update description: > Autonomous system-wide update contract covering all 5 Proxmox nodes, - 15+ containers/VMs, and 4 Docker ecosystems. Updates apt packages, + 15+ containers/VMs, and 5 Docker ecosystems (docker-vm .7, CT 116 .116, + CT 117, hwpve .11, NetBird VPS 72.61.0.17). Updates apt packages, Docker images, and container stacks in safe waves with health checks and automatic rollback on failure. agent: abiba triggers: - on "infra update" command - - weekly (Sunday 03:00 EDT) via cron + - weekly (Sunday 03:00 America/New_York) via Agent Zero scheduler task "weekly-fleet-docker-update" (qSOOVzsU) — implemented 2026-09-08 - on security advisory relay from Mumuni -version: 1.2.0 +version: 1.3.0 --- ## Maintains @@ -80,7 +81,13 @@ Before ANY update wave: | VM 109 (.7) | Home stack (Pulse, Stirling PDF) — JDownloader moved to CT 118 LXC 2026-08-01 | `cd /opt/home_stack && docker compose pull && docker compose up -d` | | VM 109 (.7) | Audiobookshelf | `cd /opt/audiobookshelf && docker compose pull && docker compose up -d` | | CT 116 (.116) | Inference Harness (LiteLLM, Prometheus, Grafana) | `cd /opt/inference-harness && docker compose pull && docker compose up -d` | -| CT 117 (zulip, storepve) | Zulip | `docker pull zulip/docker-zulip:latest && docker restart zulip-zulip-1` | +| CT 117 (storepve) | Zulip | `pct exec 117 -- bash -c 'cd /opt/zulip && docker compose pull && docker compose up -d'` (from storepve; compose recreates on zulip_default network) | +| CT 117 (storepve) | Jitsi | `pct exec 117 -- bash -c 'cd /opt/jitsi && docker compose pull && docker compose up -d'` (from storepve) | +| hwpve (.11) | Authentik (server, worker, postgres) | `ssh root@192.168.68.11 'cd /root && docker compose pull && docker compose up -d'` | +| NetBird VPS (72.61.0.17) | NetBird (server, dashboard, proxy, traefik, crowdsec) | `ssh root@72.61.0.17 'cd /root && docker compose pull && docker compose up -d'` | +| VM 109 (.7) | Trove test | `cd /opt/trove-test && docker compose pull && docker compose up -d` | +| VM 109 (.7) | docker-stats | `cd /opt/docker-stats && docker compose pull && docker compose up -d` | +| CT 116 (.116) | Monitoring (Grafana, Prometheus, Alertmanager, PVE exporter) | `cd /opt/monitoring && docker compose pull && docker compose up -d` | **Verify after Wave 3:** - All containers healthy: `docker ps` on each host @@ -88,8 +95,12 @@ Before ANY update wave: - MCP integration test: `curl localhost:4000/mcp-rest/tools/list -H "Authorization: Bearer $MASTER_KEY"` → 90 tools (23 RA-H OS + 67 GitHub) - Zulip test: send test message to #agent-hub - Dashboard loading: `curl localhost:3001/` (via CT 116) -- Firecrawl test: `curl :3002/` +- Firecrawl test: `curl -X POST http://192.168.68.7:3002/v1/search -H 'Content-Type: application/json' -d '{"query":"health","limit":1}'` → `"success":true` (GET `/` returns 200) +- Authentik test: `curl http://192.168.68.11:9000/` → 302 redirect to login +- NetBird test: `curl -s -o /dev/null -w '%{http_code}' https://netbird.sysloggh.net/` → 200 +- harness-litellm cold start: allow 3-5 min after recreate — reports unhealthy and :4000 refuses connections while loading config/DB, then recovers to 200 on its own (verified 2026-09-08) - SearXNG test: `curl :8888` +- Digest-pin sweep: `grep -rn '@sha256:' /opt/*/docker-compose.y*` on every host — digest-pinned images are INVISIBLE to `docker compose pull` (the pin re-pulls the same digest forever, so new releases never appear). Flag every pin in the run report and propose un-pinning to a floating tag with user approval before editing. Found 2026-09-10: audiobookshelf was digest-pinned at 2.34.0 (container created 2026-07-18) and silently missed by every sweep; dockhand stack was also pinned (stack removed 2026-09-10, unused). After un-pinning audiobookshelf to :latest it updated to 2.36.0 and verified HTTP 200. ## Wave 4: Proxmox Kernel Reboot @@ -158,6 +169,8 @@ Before Wave 1, snapshot these files: /opt/search-stack/searxng/docker-compose.yml (VM 109 .7) /opt/home_stack/docker-compose.yml (VM 109 .7) /opt/audiobookshelf/docker-compose.yml (VM 109 .7) +/root/compose.yml (hwpve .11 — Authentik server/worker/postgres) +/root/docker-compose.yml (NetBird VPS — netbird server/dashboard/proxy, traefik, crowdsec) /root/.pi/agent/extensions/config.yaml (CT 100 .24) /etc/systemd/system/strix-server.service (amdpve .15 — strix-moe) /etc/systemd/system/llama-server.service (VM 101 .8, VM 103 .110) @@ -220,7 +233,7 @@ When LiteLLM is upgraded to a version supporting per-key MCP grants: - [ ] All 5 PVE nodes updated, no reboot-loop - [ ] All VMs/CTs running post-update -- [ ] All Docker containers healthy (VM 109 + CT 116 + CT 117) +- [ ] All Docker containers healthy (VM 109 + CT 116 + CT 117 + hwpve .11 + NetBird VPS) - [ ] LiteLLM inference passing (syslog-auto test) - [ ] Zulip server + all 3 agents connected - [ ] GPU fleet at full capacity (3/3) diff --git a/scripts/daily-infra-report.py b/scripts/daily-infra-report.py index 08f149e..3386d4d 100755 --- a/scripts/daily-infra-report.py +++ b/scripts/daily-infra-report.py @@ -15,7 +15,7 @@ import smtplib, json, subprocess, os, sys, datetime, re from email.mime.text import MIMEText from email.mime.multipart import MIMEMultipart -PVE = "https://minipve.sysloggh.net" +PVE = "https://192.168.68.12:8006" AUTH = "Authorization: PVEAPIToken=monitoring@pve!mumuni=eafd56c5-93d4-4d40-a41d-e688be0987f3" # ── Shared credentials —─ @@ -38,10 +38,16 @@ TIME_STR = NOW.strftime("%Y-%m-%d %H:%M UTC") # ── Helpers ── def pve_get(path): - cmd = f'curl -sfk --connect-timeout 10 "{PVE}{path}" -H "{AUTH}"' + """Fetch PVE API data. Returns list on success, None on error (to distinguish from empty list).""" + cmd = f'curl -sk --connect-timeout 10 "{PVE}{path}" -H "{AUTH}"' try: - return json.loads(subprocess.check_output(cmd, shell=True))["data"] - except: return [] + r = subprocess.run(cmd, shell=True, capture_output=True, text=True, timeout=12) + if r.returncode != 0: + return None + data = json.loads(r.stdout) + return data.get("data", []) + except: + return None def ssh(host, cmd): try: @@ -97,21 +103,33 @@ def collect(): # ── Proxmox Nodes ── nodes = pve_get("/api2/json/nodes") - report["nodes"] = {n["node"]: { - "cpu_pct": round(n.get('cpu',0)*100, 1), - "ram": f"{n.get('mem',0)//1024//1024}/{n.get('maxmem',0)//1024//1024}MB", - "ram_pct": round(n.get('mem',0)/n.get('maxmem',1)*100, 0), - "disk": f"{n.get('disk',0)//1024//1024//1024}/{n.get('maxdisk',0)//1024//1024//1024}GB", - "disk_pct": round(n.get('disk',0)/n.get('maxdisk',1)*100, 0), - "uptime_h": n.get('uptime',0)//3600, - "status": n["status"] - } for n in nodes} - report["node_count"] = len(nodes) - report["nodes_online"] = sum(1 for n in nodes if n["status"] == "online") + if nodes is None: + report["nodes"] = {} + report["node_count"] = 0 + report["nodes_online"] = 0 + report["pve_probe_status"] = "unreachable" + else: + report["nodes"] = {n["node"]: { + "cpu_pct": round(n.get('cpu',0)*100, 1), + "ram": f"{n.get('mem',0)//1024//1024}/{n.get('maxmem',0)//1024//1024}MB", + "ram_pct": round(n.get('mem',0)/n.get('maxmem',1)*100, 0), + "disk": f"{n.get('disk',0)//1024//1024//1024}/{n.get('maxdisk',0)//1024//1024//1024}GB", + "disk_pct": round(n.get('disk',0)/n.get('maxdisk',1)*100, 0), + "uptime_h": n.get('uptime',0)//3600, + "status": n["status"] + } for n in nodes} + report["node_count"] = len(nodes) + report["nodes_online"] = sum(1 for n in nodes if n["status"] == "online") + report["pve_probe_status"] = "ok" # ── VMs/CTs ── resources = pve_get("/api2/json/cluster/resources") - vms = [r for r in resources if r.get("type") in ("qemu","lxc")] + if resources is None: + vms = [] + report["resources_probe_status"] = "unreachable" + else: + vms = [r for r in resources if r.get("type") in ("qemu","lxc")] + report["resources_probe_status"] = "ok" report["total_vms"] = len(vms) report["running_vms"] = sum(1 for v in vms if v.get("status") == "running") stopped = [v for v in vms if v.get("status") != "running"] @@ -183,7 +201,7 @@ def collect(): ("Authentik", "https://auth.sysloggh.net"), ("Zulip", "https://chat.sysloggh.net"), ("Pulse", "https://pulse.sysloggh.net"), - ("Proxmox", "https://minipve.sysloggh.net"), + ("Proxmox", "https://192.168.68.12:8006"), ("SearXNG", "http://192.168.68.7:8888"), ("Firecrawl", "http://192.168.68.7:3002/health"), ] @@ -247,11 +265,13 @@ def collect(): zulip_health = json.loads(health_body) if health_body else {} except: zulip_health = {} - report["zulip_ext"]["connected"] = zulip_health.get("connected", False) - report["zulip_ext"]["queue_id"] = zulip_health.get("queue_id") - report["zulip_ext"]["last_error"] = zulip_health.get("last_error") - report["zulip_ext"]["messages_processed"] = zulip_health.get("messages_processed", 0) - report["zulip_ext"]["retry_count"] = zulip_health.get("retry_count", 0) + # Live state is nested under 'zulip' key + zulip_state = zulip_health.get("zulip", {}) + report["zulip_ext"]["connected"] = zulip_state.get("connected", False) + report["zulip_ext"]["queue_id"] = zulip_state.get("queue_id") + report["zulip_ext"]["last_error"] = zulip_state.get("last_error") + report["zulip_ext"]["messages_processed"] = zulip_state.get("messages_processed", 0) + report["zulip_ext"]["skipped"] = zulip_state.get("skipped", 0) # Phase 2: PM2 process check pm2_raw = subprocess.check_output( @@ -289,8 +309,8 @@ def collect(): # Abiba (pi) report["agents"]["abiba"] = { "platform": "pi", "ct": 100, "ip": "192.168.68.24", - "zulip_connected": zulip_health.get("connected", False), - "zulip_processed": zulip_health.get("messages_processed", 0), + "zulip_connected": zulip_state.get("connected", False), + "zulip_processed": zulip_state.get("messages_processed", 0), "pm2_status": pm2.get("status", "unknown"), "pm2_restarts": pm2.get("restarts", "?"), "pm2_uptime": pm2.get("uptime", "?"), @@ -406,7 +426,7 @@ th {{ color: #8b949e; font-weight: normal; }}
{status}
-{r['node_count']} PVE nodes · {r['total_vms']} VMs/CTs · {r['running_vms']} running · +Proxmox: {r.get('pve_probe_status', 'ok')} ({r['nodes_online']}/{r['node_count']}) · {r['total_vms']} VMs/CTs · {r['running_vms']} running · {r['docker_vm']['total'] + r['docker_syslog']['total'] + r['docker_netbird']['total']} containers · {len(r['endpoints'])} endpoints · {len(r.get('agents',{}))} agents
@@ -422,8 +442,10 @@ th {{ color: #8b949e; font-weight: normal; }} # ── Quick Stats ── html += '