Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e42b970dec | ||
|
|
eadb927ec1 | ||
|
|
d4e238047d | ||
|
|
73d5097555 | ||
|
|
c460ef905c | ||
|
|
d99b552448 | ||
|
|
fc0cd7a032 | ||
|
|
0a41a2d584 | ||
|
|
de32f54337 |
+36
-17
@@ -94,7 +94,16 @@ def audit(path):
|
||||
cfg = yaml.safe_load(f)
|
||||
|
||||
model = cfg.get("model", {})
|
||||
fb = cfg.get("fallback_providers", {})
|
||||
fb_raw = cfg.get("fallback_providers", {})
|
||||
# Normalize: fallback_providers may be a dict (single provider) or a list of dicts
|
||||
# (one entry per fallback). Both shapes are valid; we must handle both without crashing.
|
||||
if isinstance(fb_raw, dict):
|
||||
fb_entries = [fb_raw]
|
||||
elif isinstance(fb_raw, list):
|
||||
fb_entries = fb_raw
|
||||
else:
|
||||
fb_entries = [fb_raw] # Let it fail the check below as malformed
|
||||
fb = fb_entries[0] if fb_entries else {}
|
||||
comp = cfg.get("compression", {})
|
||||
aux = cfg.get("auxiliary", {})
|
||||
deleg = cfg.get("delegation", {})
|
||||
@@ -207,22 +216,32 @@ def audit(path):
|
||||
"Rule 14",
|
||||
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
fb.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
||||
)
|
||||
# Check each fallback entry. A malformed entry (not a mapping) is a VIOLATION, not a crash.
|
||||
for idx, entry in enumerate(fb_entries):
|
||||
prefix = f"fallback_providers[{idx}]"
|
||||
if not isinstance(entry, dict):
|
||||
check(
|
||||
False,
|
||||
"Rule 14",
|
||||
f"{prefix} must be a mapping (got {type(entry).__name__})",
|
||||
)
|
||||
continue
|
||||
check(
|
||||
entry.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"{prefix}.provider must be 'deepseek' (got {entry.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
entry.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"{prefix}.model must be 'deepseek-v4-flash' (got {entry.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
entry.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"{prefix}.api_key_env must be DEEPSEEK_API_KEY (got {entry.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- custom_providers sanity ---
|
||||
check(
|
||||
|
||||
+14
-10
@@ -1965,17 +1965,21 @@ contracts:
|
||||
verify_commands:
|
||||
- infisical run --env=prod -- python3 scripts/daily-infra-report.py --test-email
|
||||
- python3 -m pytest tests/test_daily_infra_report.py -q
|
||||
email_dependency:
|
||||
transport: smtp.gmail.com:587
|
||||
identity: jtabiri@gmail.com
|
||||
secret: EMAIL_PASSWORD (must be a Google app password)
|
||||
status: DEGRADED as of 2026-09-25 - 534 5.7.9 Application-specific password required
|
||||
note: A delivery failure is a credential dependency, not a code defect. Tracked
|
||||
as daily-digest-mail-transport-20260921.
|
||||
delivery:
|
||||
transport: zulip-dm-attachment
|
||||
recipient_user_id: 9
|
||||
sender: abiba-bot@chat.sysloggh.net
|
||||
key_source: abiba-bot Zulip key already on the execution host, read from the
|
||||
600-mode env file /root/.pi/agent/extensions/zulip/.env
|
||||
key_policy: do NOT add a vault entry - that is a captain decision under the auth-keys charter
|
||||
body: short Markdown pointer; the HTML attachment IS the report
|
||||
artifact: /var/log/daily-infra-report/infra-report-<UTCstamp>.html
|
||||
note: Replaced SMTP/mail on 2026-09-26 by captain decision. Removes the Google
|
||||
dependency entirely; closes daily-digest-mail-transport-20260921.
|
||||
exit_semantics:
|
||||
'1': missing PVE_TOKEN, unreachable Proxmox probe, or failed email send - raises an alert
|
||||
'0': healthy, or a deliberate DEGRADED leg where the email credential is absent
|
||||
and the report is still produced
|
||||
'1': missing PVE_TOKEN, unreachable Proxmox probe, missing/rejected Zulip
|
||||
credential, or a failed upload/post - raises an alert
|
||||
'0': healthy delivery only - there is no degraded delivery leg any more
|
||||
depends_on: []
|
||||
last_run: null
|
||||
last_status: null
|
||||
|
||||
@@ -16,11 +16,12 @@ description: >
|
||||
|
||||
Exit-code semantics (as they actually behave, verified 2026-09-25):
|
||||
* missing PVE_TOKEN, or an unreachable Proxmox probe -> exit 1 + alert
|
||||
* missing EMAIL credential -> deliberate DEGRADED leg, exit 0, report still
|
||||
produced
|
||||
* email send failure -> exit 1 (a delivery fault, not a code defect)
|
||||
* missing or rejected Zulip credential -> exit 1 (delivery is the only
|
||||
output path, so it is a real failure, not a degraded leg)
|
||||
* delivery failure -> exit 1, and the report body is printed AND persisted
|
||||
so the content is never swallowed
|
||||
|
||||
version: 1.0.0
|
||||
version: 2.0.0
|
||||
---
|
||||
|
||||
## Purpose
|
||||
@@ -93,14 +94,13 @@ $ infisical run --env=prod -- python3 scripts/daily-infra-report.py --json
|
||||
EXIT=0
|
||||
```
|
||||
|
||||
and in mail mode:
|
||||
and in delivery mode:
|
||||
|
||||
```
|
||||
Sending email...
|
||||
✅ All legs fully credentialed
|
||||
📋 Summary:
|
||||
Proxmox: 5/5 nodes online
|
||||
VMs/CTs: 22/22 running
|
||||
report ready: 16208 chars of HTML (delivered as a file attachment)
|
||||
Sending to the captain's Zulip DM...
|
||||
✅ Delivered to Zulip DM (user 9), message id 86221, attachment 16208 bytes
|
||||
at /user_uploads/2/45/m1cQesBFV78BGeNY2lN8xkN5/infra-report-20260926-153406.html
|
||||
```
|
||||
|
||||
Healthy means: every probe reports `ok`, `nodes_online == node_count`, and the
|
||||
@@ -115,9 +115,8 @@ Verified on 2026-09-25 by running each case deliberately.
|
||||
| all probes reachable, email sent | 0 | — | healthy |
|
||||
| **missing `PVE_TOKEN`** | **1** | yes | `PROBE FAILURES: proxmox: node list unreachable (PVE_TOKEN missing or API down)`, and `cluster resources unreachable` |
|
||||
| **Proxmox probe unreachable** | **1** | yes | same path as above; `pve_probe_status: unreachable` |
|
||||
| **missing `EMAIL_PASSWORD`** | **0** | no | deliberate **DEGRADED** leg (`credential-missing: EMAIL_PASSWORD`); the report is still produced |
|
||||
| **email send fails** | **1** | yes | e.g. Gmail `534 5.7.9 Application-specific password required` |
|
||||
| degraded legs present (non-email) | 0 | no | logged under `⚠️ Degraded legs` |
|
||||
| **missing/rejected Zulip credential** | **1** | yes | delivery is the only output path; report printed and persisted |
|
||||
| **upload or message post fails** | **1** | yes | report printed and persisted; message names which step failed |
|
||||
|
||||
The distinction is deliberate and must not be flattened:
|
||||
|
||||
@@ -130,22 +129,31 @@ The distinction is deliberate and must not be flattened:
|
||||
`PROBE_FAILURES` and `DEGRADED_LEGS` are separate lists for exactly this
|
||||
reason. Do not merge them.
|
||||
|
||||
## Email-delivery dependency
|
||||
## Delivery: Zulip DM carrying the report as an HTML ATTACHMENT
|
||||
|
||||
Delivery is a **credential dependency, not a code path**. The producer
|
||||
authenticates to `smtp.gmail.com:587` as `jtabiri@gmail.com` with
|
||||
`EMAIL_PASSWORD` from the vault and sends to `jerome@sysloggh.com`.
|
||||
Captain's decision 2026-09-26, clarified the same day: the digest is delivered to
|
||||
his **Zulip DM (user id 9)** from `abiba-bot@chat.sysloggh.net`, as an **HTML
|
||||
FILE** — an attachment, not HTML rendered in the message body and not a Markdown
|
||||
translation of it.
|
||||
|
||||
* Since that Google account has two-step verification, `EMAIL_PASSWORD` must be
|
||||
a Google **app password**, not the account password.
|
||||
* As of 2026-09-25 delivery is **failing** with
|
||||
`534 5.7.9 Application-specific password required`; the fix is for the
|
||||
captain to generate a fresh app password and place it in Infisical
|
||||
(`infrastructure/production`) as `EMAIL_PASSWORD`.
|
||||
* **A delivery failure is not a code defect.** Investigation of a failed send
|
||||
should start at the credential, not the script. Chasing it as a code bug
|
||||
wastes the effort; verify the credential path first with `--test-email`.
|
||||
* Tracked separately as `daily-digest-mail-transport-20260921`.
|
||||
* the styled dashboard is built exactly as before and written to
|
||||
`/var/log/daily-infra-report/infra-report-<UTCstamp>.html`;
|
||||
* it is uploaded through `POST /api/v1/user_uploads`;
|
||||
* the **message body stays short Markdown** — subject line, top-line status
|
||||
(nodes online, guests running, any degraded legs), and a link to the
|
||||
attachment. The attachment IS the report; the body does not reproduce it.
|
||||
|
||||
This removes the Google dependency entirely: **no SMTP, no `EMAIL_PASSWORD`, no
|
||||
app password, nothing to rotate.** `daily-digest-mail-transport-20260921` is
|
||||
closed under this option.
|
||||
|
||||
The **10,000-character message cap does not apply** — it bounds message TEXT
|
||||
only, and the report travels as a file. Do not shrink the report to fit it.
|
||||
|
||||
The credential is abiba-bot's Zulip key already on the execution host at
|
||||
`/root/.pi/agent/extensions/zulip/.env` (`ABIBA_ZULIP_API_KEY`, mode 600,
|
||||
root-readable). **Do not place a new credential in the vault** — under the
|
||||
auth-keys charter that is a captain decision.
|
||||
|
||||
## What counts as a failure
|
||||
|
||||
@@ -153,10 +161,18 @@ A run FAILS (exit 1) when the report cannot be trusted or delivered:
|
||||
|
||||
* any probe is unreachable, so a section would silently be empty;
|
||||
* `PVE_TOKEN` is missing;
|
||||
* the email send fails.
|
||||
* the Zulip credential is missing or rejected, or the upload/post fails.
|
||||
|
||||
A run is DEGRADED (exit 0, report still produced) when a non-load-bearing
|
||||
credential is absent, currently only `EMAIL_PASSWORD`.
|
||||
There is **no degraded delivery leg any more**. Delivery is the only output
|
||||
path, so a missing credential is a failure rather than a survivable degradation —
|
||||
the previous "missing `EMAIL_PASSWORD` still exits 0" rule is retired with the
|
||||
mail transport.
|
||||
|
||||
**A delivery failure must never swallow the report.** On failure the script
|
||||
prints the report body to stdout *and* leaves the HTML artifact on disk, so the
|
||||
content is always recoverable from the run log. That closes the queued defect
|
||||
where a failed send printed only the transport error and the report never
|
||||
surfaced.
|
||||
|
||||
## Failure behaviour
|
||||
|
||||
@@ -175,7 +191,7 @@ infisical run --env=prod -- python3 scripts/daily-infra-report.py --json \
|
||||
| grep -E 'pve_probe_status|node_count|nodes_online'
|
||||
|
||||
# delivery path
|
||||
infisical run --env=prod -- python3 scripts/daily-infra-report.py --test-email
|
||||
infisical run --env=prod -- python3 scripts/daily-infra-report.py --test-zulip
|
||||
```
|
||||
|
||||
Regression tests: `tests/test_daily_infra_report.py` (7 tests). Four of them
|
||||
@@ -183,5 +199,5 @@ fail against the pre-fix script, which is what makes them bite.
|
||||
|
||||
## Maintains
|
||||
|
||||
- daily-infra-dashboard: { status: "degraded", reason: "email credential", last_check: timestamp }
|
||||
- daily-infra-dashboard: { status: "ok|undelivered", transport: zulip-dm-attachment, last_check: timestamp }
|
||||
- pve-probe: { status: "ok|unreachable", last_check: timestamp }
|
||||
|
||||
@@ -5,6 +5,11 @@ description: >
|
||||
Standard Hermes configuration template for Syslog Solution LLC agents.
|
||||
Enforces shared infrastructure setup (Firecrawl, SearXNG, local models,
|
||||
RA-H OS MCP) while keeping agent-specific API keys and model choices.
|
||||
UPDATED 2026-09-27: Clarified the Auxiliary Tasks policy — light aux (vision,
|
||||
web_extract/browsing) -> gpu-vision (RTX 5070); context-heavy aux (compression) ->
|
||||
syslog-auto (2026-07-23 decision, Rule 7). Removed the false "one model for all
|
||||
auxiliary" / "never syslog-auto" claim; stated gpu-dense + strix-moe are the reasoning
|
||||
hosts and aux should not be pinned to them. Now matches audit-hermes-config.py line-for-line.
|
||||
UPDATED 2026-08-07: Added litellm MCP server entry; updated Rule 15 (MCP Validation)
|
||||
to enforce REAL key headers (not env-vars) from the 2026-08-07 keyless-MCP incident.
|
||||
Added Rule 12 (Context-Issue Diagnostic) + Rule 13 (.env fallback enforcement) from the
|
||||
@@ -167,13 +172,16 @@ compression:
|
||||
abort_on_summary_failure: false
|
||||
|
||||
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
||||
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
||||
# model: gpu-vision # stable alias (NOT a raw model name)
|
||||
# Auxiliary tasks split into TWO model classes — do NOT assume one model for all:
|
||||
# Light auxiliary (vision, web_extract/browsing) -> model: gpu-vision # RTX 5070
|
||||
# Keeps the reasoning hosts (gpu-dense / strix-moe) free for agent prompts.
|
||||
# Context-heavy auxiliary (compression) -> model: syslog-auto # weighted pool
|
||||
# Deliberate per the 2026-07-23 OPERATIONAL DECISION in Rule 7: summarization
|
||||
# runs against long histories and must be able to use the pool.
|
||||
# Do NOT pin auxiliary work to the reasoning hosts (gpu-dense / strix-moe).
|
||||
# All auxiliary services share identical ROUTING (base_url + api_key_env), not model:
|
||||
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||
# api_key_env: LITELLM_API_KEY
|
||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||
# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||
# NEVER use retired model names (qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
|
||||
# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
auxiliary:
|
||||
|
||||
+44
-6
@@ -6,13 +6,19 @@ name: memory-fixer
|
||||
description: >
|
||||
Auto-fix low-hanging fruit in the RA-H OS knowledge graph. No judgment calls — only deterministic Level 1 operations.
|
||||
Escalate anything that needs Kwame's input. Executes confirmed Kwame decisions to completion (state + updated_at).
|
||||
version: 2.1.0
|
||||
version: 2.2.0
|
||||
---
|
||||
---
|
||||
|
||||
# Memory Fixer
|
||||
|
||||
> **Canonical copy:** `/root/.hermes/contracts/memory-fixer-v3.md` (used by the `memory-fixer-daily` cron job). This file is the institutional record of the same contract. When the two diverge, treat the v3 source in `/root/.hermes/contracts/` as executable truth.
|
||||
> **Executable copy:** the `okyeame-memory-fixer` cron job on kagentz (`hermes cron list`) holds its instruction
|
||||
> set **inline in `~/.hermes/cron/jobs.json`** (`hermes cron edit <id> --prompt …`; there is no `--prompt-file`, and
|
||||
> `~/.hermes/cron/memory-fixer-prompt.md` is a synced draft, not the live instruction). This file is the institutional
|
||||
> record of the same contract; when the two diverge, the job prompt is what actually runs — diff it against this file
|
||||
> before claiming a prompt change landed.
|
||||
> ⚠️ Corrected 2026-09-26: the previous pointer (`/root/.hermes/contracts/memory-fixer-v3.md`) does not exist on
|
||||
> kagentz — no `/root` access from this container — and was verified unreachable, not merely stale.
|
||||
|
||||
## Purpose
|
||||
Auto-fix low-hanging fruit in the graph. No judgment calls — only deterministic Level 1 operations. Escalate anything that needs Kwame's input. When Kwame replies to an escalation, **execute the decision to completion** (update state and timestamps), never leaving a node in review-pending forever.
|
||||
@@ -125,15 +131,44 @@ updateNode(id, {
|
||||
|
||||
**Archive candidates are identified by the fix 3 query's `suggested_action = 'archive'` branch** (the `ELSE 'archive'` case: anything not an infrastructure/skill/documentation/strategic/audit type).
|
||||
|
||||
### 5. Duplicate-Node Detection (Level 1 — read-only, every run)
|
||||
|
||||
The graph's duplicate problem is rarely an agent mistyping a title: it is **recurring writers creating a new
|
||||
node per run instead of updating one**. This phase detects that class and reports it. It is read-only and
|
||||
**never merges**.
|
||||
|
||||
```bash
|
||||
python3 /home/hermes/.hermes/scripts/memory_dup_detect.py --json
|
||||
```
|
||||
Read-only, ~15s over the whole graph, exit 0. That script is the source of truth for the clustering logic —
|
||||
do not re-implement it in the prompt or hand-count "duplicates" from titles.
|
||||
|
||||
Consume each `items[]` entry's `verdict` field; do not invent your own:
|
||||
|
||||
| `verdict` | Meaning | Required action |
|
||||
|---|---|---|
|
||||
| `WRITER-DEFECT` (`run_family: true`) | ONE scheduled task writes a new node per run | Report the ids, the `agents` (the writer) and `span_days`. **Never merge** — each node is that run's audit record. If the family grew since the last report, say `UNFIXED` and name the writer. |
|
||||
| `SAFE-MERGE` | Bodies identical | Still requires an explicit `merge #A into #B` decision from Kwame. |
|
||||
| `HUMAN-DECISION` | Same subject, bodies differ | Propose **connect (an edge)**, never merge. |
|
||||
|
||||
- **Title overlap alone is not duplication.** Four distinct client workflows of one family (#357-#361) and two
|
||||
different machines' migrations (#1792/#1793) both score high on title tokens while their bodies sit 0.1-0.3
|
||||
apart. Confirm against body similarity before calling anything a duplicate.
|
||||
- Report clusters as **candidates for Kwame's decision**, never as established duplicates — a wrong auto-merge
|
||||
destroys distinct content irrecoverably.
|
||||
- Per-run history nodes are kept deliberately. Bulk-merging a run family destroys the audit trail the family exists for.
|
||||
|
||||
## Level 2 Escalations (Kwame Decision Required)
|
||||
|
||||
1. **Refresh-suggested stale nodes** flagged with `[REVIEW: refresh]` — refresh or keep? (Archive-suggested nodes are auto-archived under fix 4 and are not escalated.)
|
||||
2. **Duplicate Nodes** (same title or >70% title overlap) — Merge or keep?
|
||||
2. **Duplicate Nodes** — as detected by fix 5, by `verdict`, never by raw title overlap. `WRITER-DEFECT` is a writer fix (update one canonical node), not a merge decision; `SAFE-MERGE` and `HUMAN-DECISION` clusters are escalated for merge-or-connect.
|
||||
3. **Orphan Nodes >90 days old** — Archive or connect?
|
||||
|
||||
## Reporting Format
|
||||
|
||||
The fixer reports to Kwame via this Zulip DM:
|
||||
The fixer does **not** send anything. Under the single-egress model (2026-09-21) every report leaves the node
|
||||
through Mumuni's gate (`comms_drop.py` for the queue, `comms_gate.py` to release and read-back verify), so
|
||||
exit 0 means QUEUED, never delivered. A report body is written to a file and handed to the outbox helper:
|
||||
|
||||
```
|
||||
🦅 Memory Fixer — [HH:MM UTC]
|
||||
@@ -147,8 +182,10 @@ Stale nodes needing review (max 10):
|
||||
2. [Node #YYY] Title — Y days stale, SUGGEST: archive
|
||||
...
|
||||
|
||||
Duplicates needing decision:
|
||||
1. [Node #AAA] vs [Node #BBB] — Same title
|
||||
Duplicate clusters (candidates — Kwame decides; the fixer never merges unilaterally):
|
||||
1. [WRITER-DEFECT] #AAA/#BBB/#CCC — writer <agent>, N nodes, span Nd (UNFIXED if it grew since the last report)
|
||||
2. [HUMAN-DECISION] #DDD/#EEE — same subject, bodies differ, SUGGEST: connect
|
||||
3. "none" when the scan returned no clusters
|
||||
|
||||
Orphans >90 days:
|
||||
1. [Node #EEE] Title — X days stale, orphaned
|
||||
@@ -195,6 +232,7 @@ The result must be 0 rows when all decisions are executed. Report what was done.
|
||||
- **State integrity:** archived nodes have `state: archived` + `[ARCHIVED]` prefix; kept nodes are `state: active` without a `[REVIEW:]` tag.
|
||||
- **Auto-archive applied:** no node should ever be left tagged `[REVIEW: archive]` — that tag is retired. Any `[REVIEW: archive]` found means fix 4 was skipped; archive it and report.
|
||||
- **No review-pending forever:** after executing Kwame's decisions, `[REVIEW:%` node count must be 0.
|
||||
- **Duplicate scan ran:** every report carries the fix 5 block (`none` when there were no clusters). A report with no duplicate section means phase 5 was skipped — a silently skipped detection phase is the failure this phase exists to prevent.
|
||||
- **Timestamps:** every executed decision (and every auto-archive) bumps `updated_at`, so the node exits the stale window on the next run.
|
||||
|
||||
## Logging
|
||||
|
||||
+169
-38
@@ -243,7 +243,7 @@ def collect():
|
||||
("Pulse", "https://pulse.sysloggh.net"),
|
||||
("Proxmox", "https://192.168.68.12:8006"),
|
||||
("SearXNG", "http://192.168.68.7:8888"),
|
||||
("Firecrawl", "http://192.168.68.7:3002/health"),
|
||||
("Firecrawl", "http://192.168.68.7:3002/"), # Firecrawl serves no /health - the root is its liveness endpoint
|
||||
]
|
||||
report["endpoints"] = []
|
||||
for name, url in endpoints:
|
||||
@@ -389,6 +389,28 @@ def collect():
|
||||
|
||||
# ── HTML Dashboard ──
|
||||
|
||||
def classify_endpoint(code):
|
||||
"""Classify an endpoint probe per the fleet's probe policy.
|
||||
|
||||
Codified 2026-09-14 in the monitoring contracts: ANY HTTP status proves the
|
||||
service answered, so the service is ALIVE - 200/301/302/401/403/404 alike.
|
||||
Only a failed CONNECTION (000 / timeout / refused) is a failed probe. A 404
|
||||
from a wrong path is not a service fault and must not render as one.
|
||||
|
||||
This replaces a string comparison that was wrong in both directions
|
||||
(`ep["code"] >= "400"`): it rendered 301 as red, 404 as yellow, and a real
|
||||
500 as yellow. 5xx is kept as its own "server error" signal rather than
|
||||
being merged with 4xx.
|
||||
"""
|
||||
if not code or code == "000":
|
||||
return "red", "no connection"
|
||||
if code.startswith("5"):
|
||||
return "yellow", "server error"
|
||||
if code.startswith(("2", "3", "4")):
|
||||
return "green", "alive"
|
||||
return "yellow", f"unexpected {code}"
|
||||
|
||||
|
||||
def build_html(r):
|
||||
issues = []
|
||||
|
||||
@@ -624,7 +646,7 @@ Proxmox: {r.get('pve_probe_status', 'ok')} ({r['nodes_online']}/{r['node_count']
|
||||
# ── Network Endpoints ──
|
||||
html += '<div class="card"><h2>🌐 Network Endpoints</h2><table><tr><th>Service</th><th>Status</th></tr>'
|
||||
for ep in r["endpoints"]:
|
||||
color = "green" if ep["code"] in ("200","302","401") else ("yellow" if ep["code"] >= "400" else "red")
|
||||
color = classify_endpoint(ep["code"])[0]
|
||||
html += f'<tr><td>{ep["name"]}</td><td class="{color}">HTTP {ep["code"]}</td></tr>'
|
||||
html += '</table></div>'
|
||||
|
||||
@@ -688,42 +710,152 @@ Proxmox: {r.get('pve_probe_status', 'ok')} ({r['nodes_online']}/{r['node_count']
|
||||
return html
|
||||
|
||||
|
||||
# ── Send Email ──
|
||||
# ── Delivery: Zulip DM carrying the report as an HTML ATTACHMENT ──
|
||||
#
|
||||
# Captain's decision, clarified 2026-09-26: the report is sent as an HTML FILE,
|
||||
# i.e. an attachment - NOT HTML rendered in the message body, and NOT a Markdown
|
||||
# translation of it. So the styled dashboard is built exactly as before, uploaded
|
||||
# through Zulip's file-upload API, and the message body stays short: subject,
|
||||
# top-line status, and a pointer to the attachment.
|
||||
#
|
||||
# This removes the Google dependency entirely (no SMTP, no EMAIL_PASSWORD).
|
||||
# The 10,000-character message cap does not apply: it bounds message TEXT only,
|
||||
# and the report travels as a file.
|
||||
|
||||
def send_email(html_content, subject_prefix=""):
|
||||
FROM = "abiba@sysloggh.com"
|
||||
TO = "jerome@sysloggh.com"
|
||||
SUBJECT = f"{subject_prefix}{'🏗️ Infrastructure Report — ' + DATE_STR}"
|
||||
|
||||
msg = MIMEMultipart("alternative")
|
||||
msg["From"] = FROM
|
||||
msg["To"] = TO
|
||||
msg["Subject"] = SUBJECT
|
||||
msg.attach(MIMEText("Infrastructure report in HTML format — enable images to view.", "plain"))
|
||||
msg.attach(MIMEText(html_content, "html"))
|
||||
|
||||
ZULIP_SITE = "https://chat.sysloggh.net"
|
||||
ZULIP_BOT_EMAIL = "abiba-bot@chat.sysloggh.net"
|
||||
CAPTAIN_USER_ID = 9
|
||||
ZULIP_KEY_FILE = "/root/.pi/agent/extensions/zulip/.env"
|
||||
REPORT_ARTIFACT_DIR = "/var/log/daily-infra-report"
|
||||
|
||||
|
||||
def zulip_key():
|
||||
"""abiba-bot's Zulip key, from the env or the on-host 600 file."""
|
||||
key = os.environ.get("ABIBA_ZULIP_API_KEY")
|
||||
if key:
|
||||
return key.strip()
|
||||
try:
|
||||
EMAIL_PASSWORD = os.environ.get("EMAIL_PASSWORD") or os.environ.get("SMTP_PASSWORD") or os.environ.get("MAIL_PASSWORD")
|
||||
if not EMAIL_PASSWORD:
|
||||
print(" ⚠️ Degraded leg: credential-missing: EMAIL_PASSWORD (or SMTP_PASSWORD/MAIL_PASSWORD)", file=sys.stderr)
|
||||
DEGRADED_LEGS.append("credential-missing: EMAIL_PASSWORD")
|
||||
return True, "✅ Email leg degraded (no credential) — report still produced"
|
||||
GMAIL_EMAIL = "jtabiri@gmail.com"
|
||||
|
||||
server = smtplib.SMTP("smtp.gmail.com", 587)
|
||||
server.starttls()
|
||||
server.login(GMAIL_EMAIL, EMAIL_PASSWORD)
|
||||
server.sendmail(FROM, [TO], msg.as_string())
|
||||
server.quit()
|
||||
return True, "✅ Email sent to jerome@sysloggh.com"
|
||||
except Exception as e:
|
||||
return False, f"❌ Email failed: {e}"
|
||||
with open(ZULIP_KEY_FILE) as fh:
|
||||
for line in fh:
|
||||
if line.startswith("ABIBA_ZULIP_API_KEY="):
|
||||
return line.split("=", 1)[1].strip()
|
||||
except OSError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def build_summary(r, filename, test=False):
|
||||
"""Short Markdown body: subject, top-line status, pointer to the attachment.
|
||||
|
||||
Deliberately NOT a reproduction of the report - the attachment is the report.
|
||||
"""
|
||||
nodes = f"{r.get('nodes_online', 0)}/{r.get('node_count', 0)} nodes online"
|
||||
guests = f"{r.get('running_vms', 0)}/{r.get('total_vms', 0)} guests running"
|
||||
lines = [
|
||||
("\U0001F9EA **TEST — **" if test else "") + "\U0001F3D7\uFE0F **Infrastructure Report — " + DATE_STR + "**",
|
||||
f"**{nodes}** \u00b7 **{guests}** \u00b7 generated {TIME_STR}",
|
||||
]
|
||||
problems = []
|
||||
if r.get("pve_probe_status") != "ok":
|
||||
problems.append(f"\u274c Proxmox probe: {r.get('pve_probe_status')}")
|
||||
if r.get("resources_probe_status") != "ok":
|
||||
problems.append(f"\u274c Resources probe: {r.get('resources_probe_status')}")
|
||||
lit = r.get("litellm", {}) or {}
|
||||
checks = lit.get("checks", []) or []
|
||||
if checks:
|
||||
passed = sum(1 for c in checks if c.get("status") == "pass")
|
||||
if passed != len(checks):
|
||||
problems.append(f"\u274c LiteLLM: {passed}/{len(checks)} checks pass")
|
||||
if not (r.get("zulip_ext", {}) or {}).get("connected"):
|
||||
problems.append("\u274c Zulip extension: not connected")
|
||||
for leg in DEGRADED_LEGS:
|
||||
problems.append(f"\u26a0\uFE0F degraded: {leg}")
|
||||
|
||||
lines.append("\n".join(problems) if problems else "\u2705 All monitored services healthy")
|
||||
lines.append(f"\U0001F4CE **Full report attached:** `{filename}`")
|
||||
return "\n\n".join(lines)
|
||||
|
||||
|
||||
def _curl(args, timeout=60):
|
||||
r = subprocess.run(["curl", "-s", "-m", str(timeout)] + args,
|
||||
capture_output=True, text=True)
|
||||
try:
|
||||
return json.loads(r.stdout or "{}"), r.stdout
|
||||
except json.JSONDecodeError:
|
||||
return {}, r.stdout
|
||||
|
||||
|
||||
def _curl_json(args, timeout=90):
|
||||
r = subprocess.run(["curl", "-s", "-m", str(timeout)] + args,
|
||||
capture_output=True, text=True)
|
||||
try:
|
||||
return json.loads(r.stdout or "{}"), r.stdout
|
||||
except json.JSONDecodeError:
|
||||
return {}, r.stdout
|
||||
|
||||
|
||||
def send_zulip(html_content, report, test=False):
|
||||
"""Upload the styled HTML and post a short pointer to the captain's DM.
|
||||
|
||||
Returns (ok, message). On ANY failure the report body is also printed to
|
||||
stdout and persisted to disk, so a delivery failure can never swallow the
|
||||
content - the defect this folds in.
|
||||
"""
|
||||
os.makedirs(REPORT_ARTIFACT_DIR, exist_ok=True)
|
||||
stamp = NOW.strftime("%Y%m%d-%H%M%S")
|
||||
filename = f"infra-report-{stamp}.html"
|
||||
html_path = os.path.join(REPORT_ARTIFACT_DIR, filename)
|
||||
try:
|
||||
with open(html_path, "w") as fh:
|
||||
fh.write(html_content)
|
||||
except OSError as e:
|
||||
print(f" \u26a0\uFE0F could not persist report artifact: {e}", file=sys.stderr)
|
||||
|
||||
key = zulip_key()
|
||||
if not key:
|
||||
print(html_content) # never swallow the content
|
||||
return False, ("\u274c Delivery FAILED: no Zulip credential "
|
||||
"(ABIBA_ZULIP_API_KEY unset and "
|
||||
f"{ZULIP_KEY_FILE} unreadable). Report persisted to {html_path}")
|
||||
|
||||
auth = ["-u", f"{ZULIP_BOT_EMAIL}:{key}"]
|
||||
|
||||
# 1. Upload the report as a file.
|
||||
up, up_raw = _curl_json(auth + [
|
||||
"-X", "POST", f"{ZULIP_SITE}/api/v1/user_uploads",
|
||||
"-F", f"file=@{html_path};type=text/html",
|
||||
])
|
||||
if up.get("result") != "success" or not up.get("uri"):
|
||||
print(html_content)
|
||||
return False, (f"\u274c Delivery FAILED at upload: {up.get('msg') or up_raw[:160]} "
|
||||
f"(report persisted to {html_path})")
|
||||
|
||||
uri = up["uri"]
|
||||
size = os.path.getsize(html_path)
|
||||
|
||||
# 2. Post a short message pointing at it.
|
||||
body = build_summary(report, filename, test=test)
|
||||
link = f"[{filename}]({uri})"
|
||||
body = body.replace(f"`{filename}`", link)
|
||||
payload, raw = _curl_json(auth + [
|
||||
"-X", "POST", f"{ZULIP_SITE}/api/v1/messages",
|
||||
"-d", "type=private",
|
||||
"-d", f"to=[{CAPTAIN_USER_ID}]",
|
||||
"--data-urlencode", f"content={body}",
|
||||
])
|
||||
if payload.get("result") == "success":
|
||||
return True, (f"\u2705 Delivered to Zulip DM (user {CAPTAIN_USER_ID}), "
|
||||
f"message id {payload.get('id')}, attachment {size} bytes at {uri}")
|
||||
|
||||
print(html_content)
|
||||
return False, (f"\u274c Delivery FAILED at message post: {payload.get('msg') or raw[:160]} "
|
||||
f"(uploaded {uri}; report persisted to {html_path})")
|
||||
|
||||
|
||||
# ── Main ──
|
||||
|
||||
if __name__ == "__main__":
|
||||
is_test = "--test-email" in sys.argv
|
||||
is_test = ("--test-email" in sys.argv) or ("--test-zulip" in sys.argv)
|
||||
|
||||
print(f"{'🧪 TEST MODE' if is_test else '📊'} Collecting infrastructure data...")
|
||||
report = collect()
|
||||
@@ -738,15 +870,14 @@ if __name__ == "__main__":
|
||||
|
||||
print(" Building dashboard...")
|
||||
html = build_html(report)
|
||||
|
||||
print(f" report ready: {len(html)} chars of HTML (delivered as a file attachment)")
|
||||
|
||||
if is_test:
|
||||
prefix = "🧪 TEST — "
|
||||
print(" Sending test email...")
|
||||
print(" Sending TEST message to the captain's Zulip DM...")
|
||||
else:
|
||||
prefix = ""
|
||||
print(" Sending email...")
|
||||
|
||||
ok, msg = send_email(html, subject_prefix=prefix)
|
||||
print(" Sending to the captain's Zulip DM...")
|
||||
|
||||
ok, msg = send_zulip(html, report, test=is_test)
|
||||
print(f" {msg}")
|
||||
|
||||
# Show summary
|
||||
|
||||
@@ -0,0 +1,236 @@
|
||||
"""Regression test for the fallback_providers list-shape crash in audit-hermes-config.py.
|
||||
|
||||
WHY THIS FILE EXISTS: audit-hermes-config.py assumed `fallback_providers` was always a dict
|
||||
(single provider). Two live agents (koby, koonimo) carry it as a LIST of dicts (one entry per
|
||||
fallback), so the script crashed with:
|
||||
|
||||
File "audit-hermes-config.py", line 211, in audit
|
||||
fb.get("provider") == "deepseek",
|
||||
AttributeError: 'list' object has no attribute 'get'
|
||||
|
||||
Both are REAL agent configs, so this is not a malformed-input case — the script simply could not
|
||||
audit two of the four agents it exists to audit. Until fixed, the key-hygiene check had no
|
||||
coverage for half the fleet while appearing to run.
|
||||
|
||||
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert:
|
||||
1. A config whose `fallback_providers` is a LIST of valid dicts does NOT crash (exit code is 0 or 1,
|
||||
never a traceback/AttributeError).
|
||||
2. A config whose `fallback_providers` contains a MALFORMED entry (a list element that is not a
|
||||
mapping) reports a VIOLATION naming the offending entry, NOT an uncaught exception.
|
||||
3. The dict shape still works (existing tests must stay green).
|
||||
|
||||
No network, vault, or SSH access is required.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||
AUDIT = ROOT / "audit-hermes-config.py"
|
||||
|
||||
# A valid config where fallback_providers is a LIST of dicts (the real koby/koonimo shape).
|
||||
# One entry, well-formed: provider=deepseek, model=deepseek-v4-flash, api_key_env=DEEPSEEK_API_KEY.
|
||||
# This must produce a real verdict (PASS or FAIL) without crashing.
|
||||
LIST_SHAPE_VALID = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# A valid config where fallback_providers is a LIST with TWO entries (multiple fallbacks).
|
||||
# Both entries well-formed. Must not crash and should produce a real verdict.
|
||||
LIST_SHAPE_MULTI = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# A config where fallback_providers is a LIST containing a MALFORMED entry:
|
||||
# one element is a plain string, not a mapping. The checker must report a VIOLATION
|
||||
# naming the offending entry (fallback_providers[1]) and NOT crash.
|
||||
LIST_SHAPE_MALFORMED = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
- "not-a-mapping"
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# The original DICT shape (single provider) must still work — existing behaviour preserved.
|
||||
DICT_SHAPE_VALID = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
|
||||
def _run_config(tmp_path, name, text):
|
||||
cfg = tmp_path / name
|
||||
cfg.write_text(text)
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(AUDIT), str(cfg)],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
return proc.returncode, proc.stdout, proc.stderr
|
||||
|
||||
|
||||
def test_list_shape_single_entry_does_not_crash(tmp_path):
|
||||
"""A LIST with one valid dict must not raise AttributeError; exit 0 (PASS)."""
|
||||
code, out, err = _run_config(tmp_path, "list-single.yaml", LIST_SHAPE_VALID)
|
||||
# Must NOT be a crash (traceback). A clean run exits 0 (PASS) or 1 (FAIL), never 2+ (exception).
|
||||
assert code in (0, 1), f"Expected clean exit 0 or 1, got {code}\nSTDOUT:\n{out}\nSTDERR:\n{err}"
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
# The valid single-entry list should PASS (all rules satisfied).
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}"
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_list_shape_multiple_entries_does_not_crash(tmp_path):
|
||||
"""A LIST with two valid dicts must not raise AttributeError; exit 0 (PASS)."""
|
||||
code, out, err = _run_config(tmp_path, "list-multi.yaml", LIST_SHAPE_MULTI)
|
||||
assert code in (0, 1), f"Expected clean exit 0 or 1, got {code}\nSTDOUT:\n{out}\nSTDERR:\n{err}"
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}"
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_list_shape_malformed_entry_reports_violation_not_crash(tmp_path):
|
||||
"""A LIST containing a non-mapping element must be a reported VIOLATION, not a crash."""
|
||||
code, out, err = _run_config(tmp_path, "list-malformed.yaml", LIST_SHAPE_MALFORMED)
|
||||
# Must NOT be a crash.
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
# Should be a FAIL (exit 1) because the malformed entry is a violation.
|
||||
assert code == 1, f"Expected FAIL (exit 1) but got {code}\n{out}"
|
||||
assert "RESULT: FAIL" in out
|
||||
# The violation must name the offending entry (fallback_providers[1]).
|
||||
assert "fallback_providers[1]" in out, f"Violation did not name the offending entry:\n{out}"
|
||||
|
||||
|
||||
def test_dict_shape_still_passes(tmp_path):
|
||||
"""The original DICT shape (single provider) must still PASS — existing behaviour preserved."""
|
||||
code, out, err = _run_config(tmp_path, "dict-valid.yaml", DICT_SHAPE_VALID)
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}\nSTDERR:\n{err}"
|
||||
assert "RESULT: PASS" in out
|
||||
Reference in New Issue
Block a user