fix: move infra-monitoring probes into versioned script
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 3s
- Create scripts/infra-monitoring.sh with all targets/ports/paths in code - Add -k flag for PVE API (self-signed certs) - Probe Docker Stats and PVE Exporter via SSH (bind to 127.0.0.1) - Non-zero exit with named failures - Add tests/test_infra_monitoring.py for port drift detection - Prove happy path + broken target Fixes: infra-monitoring-probe-targets-drift-20260917
This commit is contained in:
@@ -0,0 +1,165 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
test_infra_monitoring.py — Tests for infrastructure-monitoring.sh
|
||||
|
||||
Tests:
|
||||
1. Port drift detection: asserts each probed port matches the documented value
|
||||
2. PVE API probe targets: verifies we probe real PVE nodes, not the monitoring host
|
||||
3. TLS failure labeling: verifies we use -k for self-signed certs
|
||||
4. Happy path and broken target (already done via shell test)
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
SCRIPT_PATH = "/root/abiba-workspace/projects/prose-contracts/scripts/infra-monitoring.sh"
|
||||
|
||||
def run_script():
|
||||
"""Run the monitoring script and return output + exit code."""
|
||||
result = subprocess.run(
|
||||
["bash", SCRIPT_PATH],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60
|
||||
)
|
||||
return result.stdout, result.stderr, result.returncode
|
||||
|
||||
def test_happy_path():
|
||||
"""Test that all documented targets are probed and healthy."""
|
||||
print("=== Test 1: Happy Path ===")
|
||||
stdout, stderr, returncode = run_script()
|
||||
print(stdout)
|
||||
|
||||
# Verify exit code is 0
|
||||
if returncode != 0:
|
||||
print(f"❌ FAILED: Expected exit code 0, got {returncode}")
|
||||
return False
|
||||
|
||||
# Verify all legs passed
|
||||
if "All legs OK" not in stdout:
|
||||
print(f"❌ FAILED: Expected 'All legs OK' in output")
|
||||
return False
|
||||
|
||||
print("✅ PASSED: Happy path works")
|
||||
return True
|
||||
|
||||
def test_port_drift_detection():
|
||||
"""Test that port drift from documented values is detected."""
|
||||
print("\n=== Test 2: Port Drift Detection ===")
|
||||
|
||||
# Create a modified version with wrong port
|
||||
import shutil
|
||||
test_script = SCRIPT_PATH.replace("scripts/", "scripts/test-drift-")
|
||||
shutil.copy(SCRIPT_PATH, test_script)
|
||||
|
||||
# Change Grafana port from 3001 to 3099
|
||||
with open(test_script, 'r') as f:
|
||||
content = f.read()
|
||||
content = content.replace('GRAFANA_PORT="3001"', 'GRAFANA_PORT="3099"')
|
||||
|
||||
with open(test_script, 'w') as f:
|
||||
f.write(content)
|
||||
|
||||
# Run the modified script
|
||||
stdout, stderr, returncode = run_script()
|
||||
|
||||
# Clean up
|
||||
os.remove(test_script)
|
||||
|
||||
print(stdout)
|
||||
|
||||
# Verify:
|
||||
# 1. Script exited non-zero
|
||||
if returncode == 0:
|
||||
print(f"❌ FAILED: Expected non-zero exit code for drift, got {returncode}")
|
||||
return False
|
||||
|
||||
# 2. Output mentions probe-failed for grafana
|
||||
if "probe-failed" not in stdout.lower():
|
||||
print(f"❌ FAILED: Expected 'probe-failed' in output for Grafana")
|
||||
return False
|
||||
|
||||
# 3. Output mentions the wrong port
|
||||
if "192.168.68.116:3099" not in stdout:
|
||||
print(f"❌ FAILED: Expected '192.168.68.116:3099' in output")
|
||||
return False
|
||||
|
||||
print("✅ PASSED: Port drift detection works")
|
||||
return True
|
||||
|
||||
def test_pve_api_targets():
|
||||
"""Test that PVE API probes the real nodes, not the monitoring host."""
|
||||
print("\n=== Test 3: PVE API Targets ===")
|
||||
|
||||
stdout, stderr, returncode = run_script()
|
||||
|
||||
# Verify we're NOT probing the monitoring host (192.168.68.116) for PVE API
|
||||
if ":116:8006" in stdout or "192.168.68.116:8006" in stdout:
|
||||
print(f"❌ FAILED: Should not probe monitoring host (192.168.68.116) for PVE API")
|
||||
return False
|
||||
|
||||
# Verify we're probing real PVE nodes
|
||||
expected_nodes = ["192.168.68.9", "192.168.68.12", "192.168.68.6", "192.168.68.15", "192.168.68.5"]
|
||||
found_nodes = [node for node in expected_nodes if f"{node}:8006" in stdout]
|
||||
|
||||
if len(found_nodes) != 5:
|
||||
print(f"❌ FAILED: Expected to probe all 5 PVE nodes, found {len(found_nodes)}")
|
||||
print(f" Found: {found_nodes}")
|
||||
return False
|
||||
|
||||
print("✅ PASSED: PVE API probes correct targets")
|
||||
return True
|
||||
|
||||
def test_tls_handling():
|
||||
"""Test that PVE API uses -k flag for self-signed certs."""
|
||||
print("\n=== Test 4: TLS Handling ===")
|
||||
|
||||
# Check the script for -k flag usage
|
||||
with open(SCRIPT_PATH, 'r') as f:
|
||||
content = f.read()
|
||||
|
||||
# Verify -k is used for PVE API
|
||||
if 'use_k' not in content or 'PVE_API_USE_K' not in content:
|
||||
print(f"❌ FAILED: Script should use -k flag for PVE API (self-signed certs)")
|
||||
return False
|
||||
|
||||
# Verify PVE API probes use -k
|
||||
if 'PVE_API_USE_K="1"' not in content:
|
||||
print(f"❌ FAILED: PVE_API_USE_K should be set to 1")
|
||||
return False
|
||||
|
||||
print("✅ PASSED: TLS handling configured correctly")
|
||||
return True
|
||||
|
||||
def main():
|
||||
"""Run all tests."""
|
||||
tests = [
|
||||
("Happy Path", test_happy_path),
|
||||
("Port Drift Detection", test_port_drift_detection),
|
||||
("PVE API Targets", test_pve_api_targets),
|
||||
("TLS Handling", test_tls_handling),
|
||||
]
|
||||
|
||||
results = []
|
||||
for name, test_func in tests:
|
||||
try:
|
||||
result = test_func()
|
||||
results.append((name, result))
|
||||
except Exception as e:
|
||||
print(f"\n❌ EXCEPTION in {name}: {e}")
|
||||
results.append((name, False))
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("SUMMARY")
|
||||
print("=" * 60)
|
||||
for name, result in results:
|
||||
status = "✅ PASSED" if result else "❌ FAILED"
|
||||
print(f"{name}: {status}")
|
||||
|
||||
all_passed = all(r for _, r in results)
|
||||
sys.exit(0 if all_passed else 1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user