PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
- Create scripts/infra-monitoring.sh with all targets/ports/paths in code - Add -k flag for PVE API (self-signed certs) - Probe Docker Stats and PVE Exporter via SSH (bind to 127.0.0.1) - Non-zero exit with named failures - Add tests/test_infra_monitoring.py for port drift detection - Prove happy path + broken target Fixes: infra-monitoring-probe-targets-drift-20260917
166 lines
5.2 KiB
Python
166 lines
5.2 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
test_infra_monitoring.py — Tests for infrastructure-monitoring.sh
|
|
|
|
Tests:
|
|
1. Port drift detection: asserts each probed port matches the documented value
|
|
2. PVE API probe targets: verifies we probe real PVE nodes, not the monitoring host
|
|
3. TLS failure labeling: verifies we use -k for self-signed certs
|
|
4. Happy path and broken target (already done via shell test)
|
|
"""
|
|
|
|
import subprocess
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
SCRIPT_PATH = "/root/abiba-workspace/projects/prose-contracts/scripts/infra-monitoring.sh"
|
|
|
|
def run_script():
|
|
"""Run the monitoring script and return output + exit code."""
|
|
result = subprocess.run(
|
|
["bash", SCRIPT_PATH],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=60
|
|
)
|
|
return result.stdout, result.stderr, result.returncode
|
|
|
|
def test_happy_path():
|
|
"""Test that all documented targets are probed and healthy."""
|
|
print("=== Test 1: Happy Path ===")
|
|
stdout, stderr, returncode = run_script()
|
|
print(stdout)
|
|
|
|
# Verify exit code is 0
|
|
if returncode != 0:
|
|
print(f"❌ FAILED: Expected exit code 0, got {returncode}")
|
|
return False
|
|
|
|
# Verify all legs passed
|
|
if "All legs OK" not in stdout:
|
|
print(f"❌ FAILED: Expected 'All legs OK' in output")
|
|
return False
|
|
|
|
print("✅ PASSED: Happy path works")
|
|
return True
|
|
|
|
def test_port_drift_detection():
|
|
"""Test that port drift from documented values is detected."""
|
|
print("\n=== Test 2: Port Drift Detection ===")
|
|
|
|
# Create a modified version with wrong port
|
|
import shutil
|
|
test_script = SCRIPT_PATH.replace("scripts/", "scripts/test-drift-")
|
|
shutil.copy(SCRIPT_PATH, test_script)
|
|
|
|
# Change Grafana port from 3001 to 3099
|
|
with open(test_script, 'r') as f:
|
|
content = f.read()
|
|
content = content.replace('GRAFANA_PORT="3001"', 'GRAFANA_PORT="3099"')
|
|
|
|
with open(test_script, 'w') as f:
|
|
f.write(content)
|
|
|
|
# Run the modified script
|
|
stdout, stderr, returncode = run_script()
|
|
|
|
# Clean up
|
|
os.remove(test_script)
|
|
|
|
print(stdout)
|
|
|
|
# Verify:
|
|
# 1. Script exited non-zero
|
|
if returncode == 0:
|
|
print(f"❌ FAILED: Expected non-zero exit code for drift, got {returncode}")
|
|
return False
|
|
|
|
# 2. Output mentions probe-failed for grafana
|
|
if "probe-failed" not in stdout.lower():
|
|
print(f"❌ FAILED: Expected 'probe-failed' in output for Grafana")
|
|
return False
|
|
|
|
# 3. Output mentions the wrong port
|
|
if "192.168.68.116:3099" not in stdout:
|
|
print(f"❌ FAILED: Expected '192.168.68.116:3099' in output")
|
|
return False
|
|
|
|
print("✅ PASSED: Port drift detection works")
|
|
return True
|
|
|
|
def test_pve_api_targets():
|
|
"""Test that PVE API probes the real nodes, not the monitoring host."""
|
|
print("\n=== Test 3: PVE API Targets ===")
|
|
|
|
stdout, stderr, returncode = run_script()
|
|
|
|
# Verify we're NOT probing the monitoring host (192.168.68.116) for PVE API
|
|
if ":116:8006" in stdout or "192.168.68.116:8006" in stdout:
|
|
print(f"❌ FAILED: Should not probe monitoring host (192.168.68.116) for PVE API")
|
|
return False
|
|
|
|
# Verify we're probing real PVE nodes
|
|
expected_nodes = ["192.168.68.9", "192.168.68.12", "192.168.68.6", "192.168.68.15", "192.168.68.5"]
|
|
found_nodes = [node for node in expected_nodes if f"{node}:8006" in stdout]
|
|
|
|
if len(found_nodes) != 5:
|
|
print(f"❌ FAILED: Expected to probe all 5 PVE nodes, found {len(found_nodes)}")
|
|
print(f" Found: {found_nodes}")
|
|
return False
|
|
|
|
print("✅ PASSED: PVE API probes correct targets")
|
|
return True
|
|
|
|
def test_tls_handling():
|
|
"""Test that PVE API uses -k flag for self-signed certs."""
|
|
print("\n=== Test 4: TLS Handling ===")
|
|
|
|
# Check the script for -k flag usage
|
|
with open(SCRIPT_PATH, 'r') as f:
|
|
content = f.read()
|
|
|
|
# Verify -k is used for PVE API
|
|
if 'use_k' not in content or 'PVE_API_USE_K' not in content:
|
|
print(f"❌ FAILED: Script should use -k flag for PVE API (self-signed certs)")
|
|
return False
|
|
|
|
# Verify PVE API probes use -k
|
|
if 'PVE_API_USE_K="1"' not in content:
|
|
print(f"❌ FAILED: PVE_API_USE_K should be set to 1")
|
|
return False
|
|
|
|
print("✅ PASSED: TLS handling configured correctly")
|
|
return True
|
|
|
|
def main():
|
|
"""Run all tests."""
|
|
tests = [
|
|
("Happy Path", test_happy_path),
|
|
("Port Drift Detection", test_port_drift_detection),
|
|
("PVE API Targets", test_pve_api_targets),
|
|
("TLS Handling", test_tls_handling),
|
|
]
|
|
|
|
results = []
|
|
for name, test_func in tests:
|
|
try:
|
|
result = test_func()
|
|
results.append((name, result))
|
|
except Exception as e:
|
|
print(f"\n❌ EXCEPTION in {name}: {e}")
|
|
results.append((name, False))
|
|
|
|
print("\n" + "=" * 60)
|
|
print("SUMMARY")
|
|
print("=" * 60)
|
|
for name, result in results:
|
|
status = "✅ PASSED" if result else "❌ FAILED"
|
|
print(f"{name}: {status}")
|
|
|
|
all_passed = all(r for _, r in results)
|
|
sys.exit(0 if all_passed else 1)
|
|
|
|
if __name__ == "__main__":
|
|
main()
|