Compare commits
34
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4320369bb9 | ||
|
|
3b74ca28d1 | ||
|
|
af9397d672 | ||
|
|
97dc2d772f | ||
|
|
2238777a2f | ||
|
|
6b2ba1bba5 | ||
|
|
a48b947242 | ||
|
|
ba9d29b4b9 | ||
|
|
d6376e5142 | ||
|
|
2ea6b4fc17 | ||
|
|
c6fd8eece3 | ||
|
|
4c715526ef | ||
|
|
308265e7ce | ||
|
|
88e9243ce4 | ||
|
|
13ac189365 | ||
|
|
2851a0cfc8 | ||
|
|
e0c9852de8 | ||
|
|
bf3a1ba523 | ||
|
|
f230812e3a | ||
|
|
96769a103f | ||
|
|
d697baa7b6 | ||
|
|
fb7f351a2b | ||
|
|
e42b970dec | ||
|
|
eadb927ec1 | ||
|
|
d4e238047d | ||
|
|
73d5097555 | ||
|
|
c460ef905c | ||
|
|
d99b552448 | ||
|
|
fc0cd7a032 | ||
|
|
ba38efcd75 | ||
|
|
0d30091f62 | ||
|
|
0a41a2d584 | ||
|
|
de32f54337 | ||
|
|
042fd3ccc6 |
@@ -24,7 +24,7 @@ Runs every 4 hours (2, 6, 10, 14, 18, 22 UTC at :35) via cron (`35 2,6,10,14,18,
|
||||
## Requires
|
||||
|
||||
- **LiteLLM admin key** for key validation (retrieved from `/root/.pi/agent/env.sh`)
|
||||
- **SSH access** to GPU hosts (.8, .110, .15) and agent CTs (.122, .129, .114, .24)
|
||||
- **SSH access** to GPU hosts — `llmuser` on .8 (owns `llama-server`), `root` on .110 and .15 — and agent CTs (.122, .129, .114, .24)
|
||||
- **Python 3** for script execution
|
||||
- **Network access** to LiteLLM (:4000), GPU exporters (:9400), and gateway endpoints
|
||||
|
||||
|
||||
+36
-17
@@ -94,7 +94,16 @@ def audit(path):
|
||||
cfg = yaml.safe_load(f)
|
||||
|
||||
model = cfg.get("model", {})
|
||||
fb = cfg.get("fallback_providers", {})
|
||||
fb_raw = cfg.get("fallback_providers", {})
|
||||
# Normalize: fallback_providers may be a dict (single provider) or a list of dicts
|
||||
# (one entry per fallback). Both shapes are valid; we must handle both without crashing.
|
||||
if isinstance(fb_raw, dict):
|
||||
fb_entries = [fb_raw]
|
||||
elif isinstance(fb_raw, list):
|
||||
fb_entries = fb_raw
|
||||
else:
|
||||
fb_entries = [fb_raw] # Let it fail the check below as malformed
|
||||
fb = fb_entries[0] if fb_entries else {}
|
||||
comp = cfg.get("compression", {})
|
||||
aux = cfg.get("auxiliary", {})
|
||||
deleg = cfg.get("delegation", {})
|
||||
@@ -207,22 +216,32 @@ def audit(path):
|
||||
"Rule 14",
|
||||
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
fb.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
||||
)
|
||||
# Check each fallback entry. A malformed entry (not a mapping) is a VIOLATION, not a crash.
|
||||
for idx, entry in enumerate(fb_entries):
|
||||
prefix = f"fallback_providers[{idx}]"
|
||||
if not isinstance(entry, dict):
|
||||
check(
|
||||
False,
|
||||
"Rule 14",
|
||||
f"{prefix} must be a mapping (got {type(entry).__name__})",
|
||||
)
|
||||
continue
|
||||
check(
|
||||
entry.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"{prefix}.provider must be 'deepseek' (got {entry.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
entry.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"{prefix}.model must be 'deepseek-v4-flash' (got {entry.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
entry.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"{prefix}.api_key_env must be DEEPSEEK_API_KEY (got {entry.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- custom_providers sanity ---
|
||||
check(
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
# Search ranking policy for the agent-consumption layer.
|
||||
#
|
||||
# Everything here is CONFIG, not code, so it is reviewable and changeable without
|
||||
# touching the module. Read by scripts/search-agent-consume.py.
|
||||
#
|
||||
# Why this file exists: multi-engine aggregation returns results with no
|
||||
# filtering, no dedupe and no reranking. On 2026-09-26 that put a shopping page
|
||||
# and a dictionary definition into "best practices agent context management",
|
||||
# and put four SEO blogs ABOVE the actual Proxmox forum threads on a precise
|
||||
# technical query. Identical queries also ranked differently between runs, which
|
||||
# is the strongest argument for a deterministic layer rather than hoping the
|
||||
# engines behave.
|
||||
|
||||
version: 1
|
||||
|
||||
# ── Non-answers: dropped outright, never returned ────────────────────────────
|
||||
# These are pages that cannot answer a question: navigational homepages,
|
||||
# shopping/product pages, dictionary definitions, and login walls.
|
||||
non_answer:
|
||||
# URL path is empty -> it is a site's front door, not an answer. Still allowed
|
||||
# when the host is explicitly preferred (see prefer_domains), because some
|
||||
# docs/repo front doors ARE the answer.
|
||||
host_root: true
|
||||
path_patterns:
|
||||
- '/dictionary/'
|
||||
- '/dictionary?'
|
||||
- '/wiki/Wiktionary:'
|
||||
- '/search?'
|
||||
- '/cart'
|
||||
- '/checkout'
|
||||
- '/login'
|
||||
- '/signin'
|
||||
- '/sign-in'
|
||||
- '/account/login'
|
||||
- '/shop/'
|
||||
- '/store/'
|
||||
- '/dp/' # Amazon-style product URL
|
||||
- '/gp/product/'
|
||||
- '/add-to-cart'
|
||||
- '/checkout'
|
||||
# NOTE: '/products/' and '/product/' were REMOVED as path patterns. They fired
|
||||
# on docs.digitalocean.com/products/inference/... — a legitimate documentation
|
||||
# page — which the 2026-09-26 before/after run caught. Shopping is caught by
|
||||
# the shopping HOST list instead, which does not have that false positive.
|
||||
# Query strings that betray a search/shopping surface rather than an article.
|
||||
query_keys:
|
||||
- 'q'
|
||||
- 'query'
|
||||
- 's'
|
||||
- 'search'
|
||||
- 'add-to-cart'
|
||||
# Hosts that are shopping/retail and never answer a technical question.
|
||||
hosts:
|
||||
- bestbuy.com
|
||||
- amazon.com
|
||||
- ebay.com
|
||||
- walmart.com
|
||||
- etsy.com
|
||||
- aliexpress.com
|
||||
- merriam-webster.com
|
||||
- dictionary.com
|
||||
- thesaurus.com
|
||||
- vocabulary.com
|
||||
- collinsdictionary.com
|
||||
|
||||
# ── Demotion: ranked below everything else, never dropped ────────────────────
|
||||
# Low-authority content farms / SEO aggregators. Demoted rather than dropped so
|
||||
# a genuinely useful hit is not lost, but it can never outrank a primary source.
|
||||
# Reviewable: add or remove hosts here, no code change required.
|
||||
demote_domains:
|
||||
- medium.com
|
||||
- sparkco.ai
|
||||
- mindstudio.ai
|
||||
- aitechmonk.com
|
||||
- stackai.com
|
||||
- agentic-design.ai
|
||||
- voxfor.com
|
||||
- bigiron.cc
|
||||
- linuxoperatingsystem.net
|
||||
- riparazioneserver.com
|
||||
- rossmanngroup.com
|
||||
- dev.to
|
||||
- hashnode.dev
|
||||
- substack.com
|
||||
- towardsdatascience.com
|
||||
- analyticsvidhya.com
|
||||
- geeksforgeeks.org
|
||||
- tutorialspoint.com
|
||||
- javatpoint.com
|
||||
- w3schools.com
|
||||
- scaler.com
|
||||
- simplilearn.com
|
||||
- udemy.com
|
||||
- coursera.org
|
||||
|
||||
# ── Preference: promoted above the default rank ──────────────────────────────
|
||||
# Primary sources: upstream repositories, official docs, Q&A, vendor
|
||||
# engineering blogs. These are what an agent should be reading.
|
||||
prefer_domains:
|
||||
# upstream repositories and code hosting
|
||||
- github.com
|
||||
- gitlab.com
|
||||
- codeberg.org
|
||||
- sourceforge.net
|
||||
- kernel.org
|
||||
- git.kernel.org
|
||||
# Q&A
|
||||
- stackoverflow.com
|
||||
- stackexchange.com
|
||||
- superuser.com
|
||||
- serverfault.com
|
||||
- askubuntu.com
|
||||
- discourse.org
|
||||
# vendor / project documentation and forums
|
||||
- proxmox.com
|
||||
- forum.proxmox.com
|
||||
- pve.proxmox.com
|
||||
- docs.python.org
|
||||
- developer.mozilla.org
|
||||
- kernelnewbies.org
|
||||
- man7.org
|
||||
- gnu.org
|
||||
- debian.org
|
||||
- ubuntu.com
|
||||
- redhat.com
|
||||
- kernel.dk # io_uring / Jens Axboe
|
||||
- github.io # project pages (docs, papers) — promoted, not authoritative by itself
|
||||
# vendor engineering blogs
|
||||
- anthropic.com
|
||||
- openai.com
|
||||
- googleblog.com
|
||||
- developers.googleblog.com
|
||||
- engineering.fb.com
|
||||
- netflixtechblog.com
|
||||
- aws.amazon.com
|
||||
- cloud.google.com
|
||||
- microsoft.com
|
||||
- learn.microsoft.com
|
||||
- apple.com
|
||||
- nvidia.com
|
||||
- intel.com
|
||||
- amd.com
|
||||
- redislabs.com
|
||||
- cloudflare.com
|
||||
- langchain.com
|
||||
- jetbrains.com
|
||||
- cursor.com
|
||||
# community discussion with high signal
|
||||
- news.ycombinator.com
|
||||
- lobste.rs
|
||||
- reddit.com
|
||||
|
||||
# ── Ranking weights ──────────────────────────────────────────────────────────
|
||||
# Final score = engine_score - demote_penalty + prefer_bonus, then a stable
|
||||
# tiebreak on original position so ordering is reproducible run to run.
|
||||
ranking:
|
||||
demote_penalty: 1000
|
||||
prefer_bonus: 100
|
||||
# Results that several engines independently returned are more likely real.
|
||||
multi_engine_bonus: 25
|
||||
# Shallow paths (e.g. /blog/x) are slightly less likely to be primary docs.
|
||||
host_root_allowed_when_preferred: true
|
||||
|
||||
# ── Extraction budget (criterion 4) ──────────────────────────────────────────
|
||||
# Return CONTENT, not just links, so an agent gets usable material in ONE call.
|
||||
extraction:
|
||||
top_n: 5 # how many results get page text extracted
|
||||
total_chars: 12000 # global budget across all extracted items
|
||||
per_item_chars: 4000 # cap for any single item, so one page cannot eat the budget
|
||||
timeout_seconds: 45 # per scrape
|
||||
# If extraction fails, the result is still returned with an empty excerpt —
|
||||
# a link is better than nothing, but the failure is recorded in the output.
|
||||
on_failure: keep_with_empty_excerpt
|
||||
+15
-11
@@ -643,7 +643,7 @@ contracts:
|
||||
timeout: 120
|
||||
requires:
|
||||
- Zulip API key for abiba-bot@chat.sysloggh.net
|
||||
- SSH access to amdpve (192.168.68.15) for Tanko (CT 112) and the Agent Zero Docker host (.14)
|
||||
- SSH access to minipve (192.168.68.12) for Tanko (CT 112) and the Agent Zero Docker host (.14)
|
||||
verification:
|
||||
postconditions:
|
||||
- check: bot registration active
|
||||
@@ -1965,17 +1965,21 @@ contracts:
|
||||
verify_commands:
|
||||
- infisical run --env=prod -- python3 scripts/daily-infra-report.py --test-email
|
||||
- python3 -m pytest tests/test_daily_infra_report.py -q
|
||||
email_dependency:
|
||||
transport: smtp.gmail.com:587
|
||||
identity: jtabiri@gmail.com
|
||||
secret: EMAIL_PASSWORD (must be a Google app password)
|
||||
status: DEGRADED as of 2026-09-25 - 534 5.7.9 Application-specific password required
|
||||
note: A delivery failure is a credential dependency, not a code defect. Tracked
|
||||
as daily-digest-mail-transport-20260921.
|
||||
delivery:
|
||||
transport: zulip-dm-attachment
|
||||
recipient_user_id: 9
|
||||
sender: abiba-bot@chat.sysloggh.net
|
||||
key_source: abiba-bot Zulip key already on the execution host, read from the
|
||||
600-mode env file /root/.pi/agent/extensions/zulip/.env
|
||||
key_policy: do NOT add a vault entry - that is a captain decision under the auth-keys charter
|
||||
body: short Markdown pointer; the HTML attachment IS the report
|
||||
artifact: /var/log/daily-infra-report/infra-report-<UTCstamp>.html
|
||||
note: Replaced SMTP/mail on 2026-09-26 by captain decision. Removes the Google
|
||||
dependency entirely; closes daily-digest-mail-transport-20260921.
|
||||
exit_semantics:
|
||||
'1': missing PVE_TOKEN, unreachable Proxmox probe, or failed email send - raises an alert
|
||||
'0': healthy, or a deliberate DEGRADED leg where the email credential is absent
|
||||
and the report is still produced
|
||||
'1': missing PVE_TOKEN, unreachable Proxmox probe, missing/rejected Zulip
|
||||
credential, or a failed upload/post - raises an alert
|
||||
'0': healthy delivery only - there is no degraded delivery leg any more
|
||||
depends_on: []
|
||||
last_run: null
|
||||
last_status: null
|
||||
|
||||
@@ -16,11 +16,12 @@ description: >
|
||||
|
||||
Exit-code semantics (as they actually behave, verified 2026-09-25):
|
||||
* missing PVE_TOKEN, or an unreachable Proxmox probe -> exit 1 + alert
|
||||
* missing EMAIL credential -> deliberate DEGRADED leg, exit 0, report still
|
||||
produced
|
||||
* email send failure -> exit 1 (a delivery fault, not a code defect)
|
||||
* missing or rejected Zulip credential -> exit 1 (delivery is the only
|
||||
output path, so it is a real failure, not a degraded leg)
|
||||
* delivery failure -> exit 1, and the report body is printed AND persisted
|
||||
so the content is never swallowed
|
||||
|
||||
version: 1.0.0
|
||||
version: 2.0.0
|
||||
---
|
||||
|
||||
## Purpose
|
||||
@@ -93,14 +94,13 @@ $ infisical run --env=prod -- python3 scripts/daily-infra-report.py --json
|
||||
EXIT=0
|
||||
```
|
||||
|
||||
and in mail mode:
|
||||
and in delivery mode:
|
||||
|
||||
```
|
||||
Sending email...
|
||||
✅ All legs fully credentialed
|
||||
📋 Summary:
|
||||
Proxmox: 5/5 nodes online
|
||||
VMs/CTs: 22/22 running
|
||||
report ready: 16208 chars of HTML (delivered as a file attachment)
|
||||
Sending to the captain's Zulip DM...
|
||||
✅ Delivered to Zulip DM (user 9), message id 86221, attachment 16208 bytes
|
||||
at /user_uploads/2/45/m1cQesBFV78BGeNY2lN8xkN5/infra-report-20260926-153406.html
|
||||
```
|
||||
|
||||
Healthy means: every probe reports `ok`, `nodes_online == node_count`, and the
|
||||
@@ -115,9 +115,8 @@ Verified on 2026-09-25 by running each case deliberately.
|
||||
| all probes reachable, email sent | 0 | — | healthy |
|
||||
| **missing `PVE_TOKEN`** | **1** | yes | `PROBE FAILURES: proxmox: node list unreachable (PVE_TOKEN missing or API down)`, and `cluster resources unreachable` |
|
||||
| **Proxmox probe unreachable** | **1** | yes | same path as above; `pve_probe_status: unreachable` |
|
||||
| **missing `EMAIL_PASSWORD`** | **0** | no | deliberate **DEGRADED** leg (`credential-missing: EMAIL_PASSWORD`); the report is still produced |
|
||||
| **email send fails** | **1** | yes | e.g. Gmail `534 5.7.9 Application-specific password required` |
|
||||
| degraded legs present (non-email) | 0 | no | logged under `⚠️ Degraded legs` |
|
||||
| **missing/rejected Zulip credential** | **1** | yes | delivery is the only output path; report printed and persisted |
|
||||
| **upload or message post fails** | **1** | yes | report printed and persisted; message names which step failed |
|
||||
|
||||
The distinction is deliberate and must not be flattened:
|
||||
|
||||
@@ -130,22 +129,31 @@ The distinction is deliberate and must not be flattened:
|
||||
`PROBE_FAILURES` and `DEGRADED_LEGS` are separate lists for exactly this
|
||||
reason. Do not merge them.
|
||||
|
||||
## Email-delivery dependency
|
||||
## Delivery: Zulip DM carrying the report as an HTML ATTACHMENT
|
||||
|
||||
Delivery is a **credential dependency, not a code path**. The producer
|
||||
authenticates to `smtp.gmail.com:587` as `jtabiri@gmail.com` with
|
||||
`EMAIL_PASSWORD` from the vault and sends to `jerome@sysloggh.com`.
|
||||
Captain's decision 2026-09-26, clarified the same day: the digest is delivered to
|
||||
his **Zulip DM (user id 9)** from `abiba-bot@chat.sysloggh.net`, as an **HTML
|
||||
FILE** — an attachment, not HTML rendered in the message body and not a Markdown
|
||||
translation of it.
|
||||
|
||||
* Since that Google account has two-step verification, `EMAIL_PASSWORD` must be
|
||||
a Google **app password**, not the account password.
|
||||
* As of 2026-09-25 delivery is **failing** with
|
||||
`534 5.7.9 Application-specific password required`; the fix is for the
|
||||
captain to generate a fresh app password and place it in Infisical
|
||||
(`infrastructure/production`) as `EMAIL_PASSWORD`.
|
||||
* **A delivery failure is not a code defect.** Investigation of a failed send
|
||||
should start at the credential, not the script. Chasing it as a code bug
|
||||
wastes the effort; verify the credential path first with `--test-email`.
|
||||
* Tracked separately as `daily-digest-mail-transport-20260921`.
|
||||
* the styled dashboard is built exactly as before and written to
|
||||
`/var/log/daily-infra-report/infra-report-<UTCstamp>.html`;
|
||||
* it is uploaded through `POST /api/v1/user_uploads`;
|
||||
* the **message body stays short Markdown** — subject line, top-line status
|
||||
(nodes online, guests running, any degraded legs), and a link to the
|
||||
attachment. The attachment IS the report; the body does not reproduce it.
|
||||
|
||||
This removes the Google dependency entirely: **no SMTP, no `EMAIL_PASSWORD`, no
|
||||
app password, nothing to rotate.** `daily-digest-mail-transport-20260921` is
|
||||
closed under this option.
|
||||
|
||||
The **10,000-character message cap does not apply** — it bounds message TEXT
|
||||
only, and the report travels as a file. Do not shrink the report to fit it.
|
||||
|
||||
The credential is abiba-bot's Zulip key already on the execution host at
|
||||
`/root/.pi/agent/extensions/zulip/.env` (`ABIBA_ZULIP_API_KEY`, mode 600,
|
||||
root-readable). **Do not place a new credential in the vault** — under the
|
||||
auth-keys charter that is a captain decision.
|
||||
|
||||
## What counts as a failure
|
||||
|
||||
@@ -153,10 +161,18 @@ A run FAILS (exit 1) when the report cannot be trusted or delivered:
|
||||
|
||||
* any probe is unreachable, so a section would silently be empty;
|
||||
* `PVE_TOKEN` is missing;
|
||||
* the email send fails.
|
||||
* the Zulip credential is missing or rejected, or the upload/post fails.
|
||||
|
||||
A run is DEGRADED (exit 0, report still produced) when a non-load-bearing
|
||||
credential is absent, currently only `EMAIL_PASSWORD`.
|
||||
There is **no degraded delivery leg any more**. Delivery is the only output
|
||||
path, so a missing credential is a failure rather than a survivable degradation —
|
||||
the previous "missing `EMAIL_PASSWORD` still exits 0" rule is retired with the
|
||||
mail transport.
|
||||
|
||||
**A delivery failure must never swallow the report.** On failure the script
|
||||
prints the report body to stdout *and* leaves the HTML artifact on disk, so the
|
||||
content is always recoverable from the run log. That closes the queued defect
|
||||
where a failed send printed only the transport error and the report never
|
||||
surfaced.
|
||||
|
||||
## Failure behaviour
|
||||
|
||||
@@ -175,7 +191,7 @@ infisical run --env=prod -- python3 scripts/daily-infra-report.py --json \
|
||||
| grep -E 'pve_probe_status|node_count|nodes_online'
|
||||
|
||||
# delivery path
|
||||
infisical run --env=prod -- python3 scripts/daily-infra-report.py --test-email
|
||||
infisical run --env=prod -- python3 scripts/daily-infra-report.py --test-zulip
|
||||
```
|
||||
|
||||
Regression tests: `tests/test_daily_infra_report.py` (7 tests). Four of them
|
||||
@@ -183,5 +199,5 @@ fail against the pre-fix script, which is what makes them bite.
|
||||
|
||||
## Maintains
|
||||
|
||||
- daily-infra-dashboard: { status: "degraded", reason: "email credential", last_check: timestamp }
|
||||
- daily-infra-dashboard: { status: "ok|undelivered", transport: zulip-dm-attachment, last_check: timestamp }
|
||||
- pve-probe: { status: "ok|unreachable", last_check: timestamp }
|
||||
|
||||
@@ -414,7 +414,7 @@ one-off GPU builds. No automated post-migration cleanup was in place.
|
||||
| 108 | media | storepve | lxc | ✅ reachable |
|
||||
| 110 | gitea | minipve | lxc | ✅ reachable |
|
||||
| 111 | tdunna | **storepve** | lxc | ⛔ **REPORT-ONLY** (192.168.68.129, Theo's box — no GC at any level) |
|
||||
| 112 | tanko | amdpve | lxc | ✅ reachable |
|
||||
| 112 | tanko | minipve | lxc | ✅ reachable |
|
||||
| 113 | baggy | amdpve | lxc | ✅ reachable |
|
||||
| 115 | scottdenya | amdpve | lxc | ✅ reachable |
|
||||
| 116 | syslog-api | minipve | lxc | ✅ reachable |
|
||||
|
||||
@@ -1,5 +1,14 @@
|
||||
# Probe-drift round 2 — per-leg before/after evidence
|
||||
|
||||
> **Historical record** — 2026-09-28: The lines below that describe tanko as
|
||||
> "DSH (DeepSeek Harness)" only reflect what the check reported when it was
|
||||
> running. Tanko's runtime was later found to be **hybrid (DSH + Hermes)** —
|
||||
> the check had a `/root/` hardcoding bug that made it probe the wrong home
|
||||
> directory and report `wrapper-missing:tanko` for an agent with a working
|
||||
> wrapper. This document records the observed output, not the underlying
|
||||
> truth; see `fix/agent-health-root-hardcoding-20260928` for the correction.
|
||||
|
||||
|
||||
**Date:** 2026-09-10
|
||||
**Worktree (absolute execution path):** `/root/.treehouse/prose-contracts-9ce5f3/3/prose-contracts`
|
||||
**Branch:** `fm/probe-drift-round2-20260909`
|
||||
|
||||
@@ -70,10 +70,10 @@ Agent (systemd) → LITELLM_API_KEY → LiteLLM (:116/v1) → GPU (llama-server)
|
||||
|
||||
## Config Pattern — Mandatory Fields
|
||||
|
||||
### For Hermes Agents (Mumuni, Koonimo)
|
||||
### For Hermes Agents (Mumuni, Koonimo, Tanko-hybrid)
|
||||
|
||||
Every Hermes agent's `/root/.hermes/config.yaml` (or `/home/jerome/.hermes/config.yaml`) MUST have:
|
||||
(Tanko is excluded — migrated to DSH/DeepSeek Harness on 2026-08-27, no longer uses Hermes config.)
|
||||
(Tanko is hybrid — runs both DSH and Hermes since 2026-08-27, so its Hermes config is also checked.)
|
||||
|
||||
### 1. Main Model
|
||||
```yaml
|
||||
@@ -297,7 +297,7 @@ Run the consolidated health check:
|
||||
```bash
|
||||
python3 /root/scripts/agent-health-check.py
|
||||
```
|
||||
This validates all 4 LiteLLM keys, detects GPU port conflicts (ghost processes),
|
||||
This validates each agent's live LiteLLM key against the gateway, including tanko, which runs HYBRID (DSH + Hermes) since 2026-08-27; detects GPU port conflicts (ghost processes),
|
||||
verifies gateway liveness, confirms Zulip streaming (`edit_message` present),
|
||||
and counts recent errors. Non-disruptive — never restarts anything.
|
||||
|
||||
|
||||
@@ -5,6 +5,11 @@ description: >
|
||||
Standard Hermes configuration template for Syslog Solution LLC agents.
|
||||
Enforces shared infrastructure setup (Firecrawl, SearXNG, local models,
|
||||
RA-H OS MCP) while keeping agent-specific API keys and model choices.
|
||||
UPDATED 2026-09-27: Clarified the Auxiliary Tasks policy — light aux (vision,
|
||||
web_extract/browsing) -> gpu-vision (RTX 5070); context-heavy aux (compression) ->
|
||||
syslog-auto (2026-07-23 decision, Rule 7). Removed the false "one model for all
|
||||
auxiliary" / "never syslog-auto" claim; stated gpu-dense + strix-moe are the reasoning
|
||||
hosts and aux should not be pinned to them. Now matches audit-hermes-config.py line-for-line.
|
||||
UPDATED 2026-08-07: Added litellm MCP server entry; updated Rule 15 (MCP Validation)
|
||||
to enforce REAL key headers (not env-vars) from the 2026-08-07 keyless-MCP incident.
|
||||
Added Rule 12 (Context-Issue Diagnostic) + Rule 13 (.env fallback enforcement) from the
|
||||
@@ -167,13 +172,16 @@ compression:
|
||||
abort_on_summary_failure: false
|
||||
|
||||
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
||||
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
||||
# model: gpu-vision # stable alias (NOT a raw model name)
|
||||
# Auxiliary tasks split into TWO model classes — do NOT assume one model for all:
|
||||
# Light auxiliary (vision, web_extract/browsing) -> model: gpu-vision # RTX 5070
|
||||
# Keeps the reasoning hosts (gpu-dense / strix-moe) free for agent prompts.
|
||||
# Context-heavy auxiliary (compression) -> model: syslog-auto # weighted pool
|
||||
# Deliberate per the 2026-07-23 OPERATIONAL DECISION in Rule 7: summarization
|
||||
# runs against long histories and must be able to use the pool.
|
||||
# Do NOT pin auxiliary work to the reasoning hosts (gpu-dense / strix-moe).
|
||||
# All auxiliary services share identical ROUTING (base_url + api_key_env), not model:
|
||||
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||
# api_key_env: LITELLM_API_KEY
|
||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||
# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||
# NEVER use retired model names (qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
|
||||
# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
auxiliary:
|
||||
|
||||
@@ -28,7 +28,7 @@ connectivity recovery including end-to-end DM validation.
|
||||
|
||||
| Param | Type | Required | Default | Description |
|
||||
|-------|------|----------|---------|-------------|
|
||||
| `target` | string | yes | — | Agent name: `mumuni`, `koby`, or `shumba` (Tanko excluded — on DSH since 2026-08-27, no Hermes plugin) |
|
||||
| `target` | string | yes | — | Agent name: `mumuni`, `koby`, or `shumba` (Tanko excluded — hybrid (DSH + Hermes) since 2026-08-27, no Hermes plugin) |
|
||||
| `branch` | string | no | `master` | Git branch to pull (overridable for pinning) |
|
||||
|
||||
## Maintains
|
||||
@@ -55,7 +55,7 @@ connectivity recovery including end-to-end DM validation.
|
||||
|
||||
| Host | CT | Proxmox | IP (direct) | Hermes Home | User |
|
||||
|------|-----|---------|-------------|-------------|------|
|
||||
| Tanko | CT112 | amdpve | 192.168.68.122 | /home/jerome/.hermes | jerome | *(DSH since 2026-08-27 — historical, plugin retired on this host)* |
|
||||
| Tanko | CT112 | minipve | 192.168.68.122 | /home/jerome/.hermes | jerome | *(hybrid (DSH + Hermes) since 2026-08-27 — historical, plugin retired on this host)* |
|
||||
| Koby | CT111 | storepve | 192.168.68.129 | /root/.hermes | root |
|
||||
| Shumba | — | — | 192.168.68.119 | /home/lucky/.hermes | lucky |
|
||||
|
||||
@@ -72,7 +72,7 @@ connectivity recovery including end-to-end DM validation.
|
||||
### Step 1: Resolve Target
|
||||
|
||||
Map `target` to host, CT ID, hermes_home, and user from the live-state table.
|
||||
For CT112 route through `ssh root@amdpve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
For CT112 route through `ssh root@minipve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
|
||||
### Step 2: Pull Latest Plugin Source
|
||||
|
||||
@@ -121,7 +121,7 @@ cp plugins/platforms/zulip/adapter.py \
|
||||
{{hermes_home}}/hermes-agent/plugins/platforms/zulip/
|
||||
|
||||
# Fix ownership (was Tanko-only, runs as jerome user)
|
||||
# RETIRED 2026-08-27: tanko no longer uses the Hermes Zulip plugin (DSH).
|
||||
# RETIRED 2026-08-27: tanko no longer uses the Hermes Zulip plugin (hybrid: DSH + Hermes).
|
||||
[ "{{target}}" = "tanko" ] && chown -R jerome:jerome \
|
||||
{{hermes_home}}/hermes-agent/plugins/platforms/zulip/
|
||||
|
||||
|
||||
@@ -24,7 +24,7 @@ gateway restart, and connection validation.
|
||||
|
||||
| Param | Type | Required | Default | Description |
|
||||
|-------|------|----------|---------|-------------|
|
||||
| `target` | string | yes | — | Agent name: `mumuni`, `koby`, or `shumba` (Tanko excluded — DSH since 2026-08-27) |
|
||||
| `target` | string | yes | — | Agent name: `mumuni`, `koby`, or `shumba` (Tanko excluded — hybrid (DSH + Hermes) since 2026-08-27) |
|
||||
|
||||
## Maintains
|
||||
|
||||
@@ -67,7 +67,7 @@ gateway restart, and connection validation.
|
||||
### Step 1: Locate Target
|
||||
|
||||
Map `target` to connectivity parameters from the live-state table above.
|
||||
For CT112 route through `ssh root@amdpve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
For CT112 route through `ssh root@minipve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
|
||||
### Step 2: Deploy Zulip Adapter
|
||||
|
||||
@@ -93,7 +93,7 @@ cp zulip-platform-plugins/plugins/platforms/zulip/adapter.py \
|
||||
zulip-platform-plugins/plugins/platforms/zulip/plugin.yaml \
|
||||
<HERMES_HOME>/hermes-agent/plugins/platforms/zulip/
|
||||
|
||||
# Fix ownership (was Tanko-only; RETIRED 2026-08-27 — tanko on DSH, no Hermes plugin)
|
||||
# Fix ownership (was Tanko-only; RETIRED 2026-08-27 — tanko on hybrid (DSH + Hermes), no Hermes plugin)
|
||||
chown -R jerome:jerome <HERMES_HOME>/hermes-agent/plugins/platforms/zulip/ # Tanko only (historical)
|
||||
|
||||
# Clean up
|
||||
|
||||
@@ -105,8 +105,8 @@ description: >
|
||||
|
||||
| Node | IP | CPU | RAM | VMs/CTs | Role |
|
||||
|------|----|-----|-----|---------|------|
|
||||
| minipve | .12 | 16C | 30GB | abiba, authentik, gitea, syslog-api, infisical-vault, jitsi | Auth, git, messaging |
|
||||
| amdpve | .15 | 32C | 62GB | kagentz, tanko, baggy, scottdenya, adguard2 | Agents, compute |
|
||||
| minipve | .12 | 16C | 30GB | abiba, tanko, authentik, gitea, syslog-api, infisical-vault, jitsi | Auth, git, messaging |
|
||||
| amdpve | .15 | 32C | 62GB | kagentz, baggy, scottdenya, adguard2 | Agents, compute |
|
||||
| storepve | .6 | 28C | 31GB | docker-vm, ra-h-os, PBS, media, jdownloader, zulip, tdunna | Docker, storage, chat |
|
||||
| acerpve | .9 | 28C | 31GB | llm-gpu | GPU VMs |
|
||||
| ocupve | .5 | 12C | 14GB | ocu-llm | GPU VMs |
|
||||
@@ -682,7 +682,7 @@ ssh root@192.168.68.110 "systemctl restart llama-server"
|
||||
| 109 | docker-vm | storepve | .7 | Docker host | ❌ |
|
||||
| 110 | gitea | minipve | **.17** | Git | ❌ |
|
||||
| 111 | tdunna | storepve | .129 | Hermes agent — ⛔ REPORT-ONLY (Theo's box, no GC) | ✅ |
|
||||
| 112 | tanko | amdpve | .122 | DSH (DeepSeek Harness) agent | ✅ |
|
||||
| 112 | tanko | minipve | .122 | hybrid (DSH + Hermes) agent | ✅ |
|
||||
| 113 | baggy | amdpve | .114 | Hermes agent | ✅ |
|
||||
| 115 | scottdenya | amdpve | .75 | Denya OneCare | ❌ |
|
||||
| 116 | syslog-api | minipve | .116 | LiteLLM + Grafana | ❌ |
|
||||
@@ -712,7 +712,7 @@ Source of truth: `/root/scripts/pct-run.sh` or `prose-contracts/scripts/pct-run.
|
||||
| 100 | abiba | minipve | `pct-run 100` |
|
||||
| 105 | kagentz | amdpve | `pct-run 105` |
|
||||
| 111 | tdunna | storepve | `pct-run 111` (⛔ report-only — no GC) |
|
||||
| 112 | tanko | amdpve | `pct-run 112` |
|
||||
| 112 | tanko | minipve | `pct-run 112` |
|
||||
| 113 | baggy | amdpve | `pct-run 113` |
|
||||
| 115 | scottdenya | amdpve | `pct-run 115` |
|
||||
| 104 | authentik | minipve | `pct-run 104` |
|
||||
|
||||
@@ -59,7 +59,7 @@ Before ANY update wave:
|
||||
| ocupve (.5) | Proxmox node | `apt update && apt upgrade -y` | 5 min |
|
||||
| CT 100 (.24) | Abiba (pi) | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 116 (.116) | syslog-api (LiteLLM host) | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 112 (tanko, amdpve) | Tanko | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 112 (tanko, minipve) | Tanko | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 105 (kagentz, minipve) | Mumuni | `apt update && apt upgrade -y` | 3 min |
|
||||
| VM 101 (.8) | llm-gpu (RTX 3090) | `apt update && apt upgrade -y` | 3 min |
|
||||
| VM 103 (.110) | ocu-llm (RTX 5070) | `apt update && apt upgrade -y` | 3 min |
|
||||
|
||||
+44
-6
@@ -6,13 +6,19 @@ name: memory-fixer
|
||||
description: >
|
||||
Auto-fix low-hanging fruit in the RA-H OS knowledge graph. No judgment calls — only deterministic Level 1 operations.
|
||||
Escalate anything that needs Kwame's input. Executes confirmed Kwame decisions to completion (state + updated_at).
|
||||
version: 2.1.0
|
||||
version: 2.2.0
|
||||
---
|
||||
---
|
||||
|
||||
# Memory Fixer
|
||||
|
||||
> **Canonical copy:** `/root/.hermes/contracts/memory-fixer-v3.md` (used by the `memory-fixer-daily` cron job). This file is the institutional record of the same contract. When the two diverge, treat the v3 source in `/root/.hermes/contracts/` as executable truth.
|
||||
> **Executable copy:** the `okyeame-memory-fixer` cron job on kagentz (`hermes cron list`) holds its instruction
|
||||
> set **inline in `~/.hermes/cron/jobs.json`** (`hermes cron edit <id> --prompt …`; there is no `--prompt-file`, and
|
||||
> `~/.hermes/cron/memory-fixer-prompt.md` is a synced draft, not the live instruction). This file is the institutional
|
||||
> record of the same contract; when the two diverge, the job prompt is what actually runs — diff it against this file
|
||||
> before claiming a prompt change landed.
|
||||
> ⚠️ Corrected 2026-09-26: the previous pointer (`/root/.hermes/contracts/memory-fixer-v3.md`) does not exist on
|
||||
> kagentz — no `/root` access from this container — and was verified unreachable, not merely stale.
|
||||
|
||||
## Purpose
|
||||
Auto-fix low-hanging fruit in the graph. No judgment calls — only deterministic Level 1 operations. Escalate anything that needs Kwame's input. When Kwame replies to an escalation, **execute the decision to completion** (update state and timestamps), never leaving a node in review-pending forever.
|
||||
@@ -125,15 +131,44 @@ updateNode(id, {
|
||||
|
||||
**Archive candidates are identified by the fix 3 query's `suggested_action = 'archive'` branch** (the `ELSE 'archive'` case: anything not an infrastructure/skill/documentation/strategic/audit type).
|
||||
|
||||
### 5. Duplicate-Node Detection (Level 1 — read-only, every run)
|
||||
|
||||
The graph's duplicate problem is rarely an agent mistyping a title: it is **recurring writers creating a new
|
||||
node per run instead of updating one**. This phase detects that class and reports it. It is read-only and
|
||||
**never merges**.
|
||||
|
||||
```bash
|
||||
python3 /home/hermes/.hermes/scripts/memory_dup_detect.py --json
|
||||
```
|
||||
Read-only, ~15s over the whole graph, exit 0. That script is the source of truth for the clustering logic —
|
||||
do not re-implement it in the prompt or hand-count "duplicates" from titles.
|
||||
|
||||
Consume each `items[]` entry's `verdict` field; do not invent your own:
|
||||
|
||||
| `verdict` | Meaning | Required action |
|
||||
|---|---|---|
|
||||
| `WRITER-DEFECT` (`run_family: true`) | ONE scheduled task writes a new node per run | Report the ids, the `agents` (the writer) and `span_days`. **Never merge** — each node is that run's audit record. If the family grew since the last report, say `UNFIXED` and name the writer. |
|
||||
| `SAFE-MERGE` | Bodies identical | Still requires an explicit `merge #A into #B` decision from Kwame. |
|
||||
| `HUMAN-DECISION` | Same subject, bodies differ | Propose **connect (an edge)**, never merge. |
|
||||
|
||||
- **Title overlap alone is not duplication.** Four distinct client workflows of one family (#357-#361) and two
|
||||
different machines' migrations (#1792/#1793) both score high on title tokens while their bodies sit 0.1-0.3
|
||||
apart. Confirm against body similarity before calling anything a duplicate.
|
||||
- Report clusters as **candidates for Kwame's decision**, never as established duplicates — a wrong auto-merge
|
||||
destroys distinct content irrecoverably.
|
||||
- Per-run history nodes are kept deliberately. Bulk-merging a run family destroys the audit trail the family exists for.
|
||||
|
||||
## Level 2 Escalations (Kwame Decision Required)
|
||||
|
||||
1. **Refresh-suggested stale nodes** flagged with `[REVIEW: refresh]` — refresh or keep? (Archive-suggested nodes are auto-archived under fix 4 and are not escalated.)
|
||||
2. **Duplicate Nodes** (same title or >70% title overlap) — Merge or keep?
|
||||
2. **Duplicate Nodes** — as detected by fix 5, by `verdict`, never by raw title overlap. `WRITER-DEFECT` is a writer fix (update one canonical node), not a merge decision; `SAFE-MERGE` and `HUMAN-DECISION` clusters are escalated for merge-or-connect.
|
||||
3. **Orphan Nodes >90 days old** — Archive or connect?
|
||||
|
||||
## Reporting Format
|
||||
|
||||
The fixer reports to Kwame via this Zulip DM:
|
||||
The fixer does **not** send anything. Under the single-egress model (2026-09-21) every report leaves the node
|
||||
through Mumuni's gate (`comms_drop.py` for the queue, `comms_gate.py` to release and read-back verify), so
|
||||
exit 0 means QUEUED, never delivered. A report body is written to a file and handed to the outbox helper:
|
||||
|
||||
```
|
||||
🦅 Memory Fixer — [HH:MM UTC]
|
||||
@@ -147,8 +182,10 @@ Stale nodes needing review (max 10):
|
||||
2. [Node #YYY] Title — Y days stale, SUGGEST: archive
|
||||
...
|
||||
|
||||
Duplicates needing decision:
|
||||
1. [Node #AAA] vs [Node #BBB] — Same title
|
||||
Duplicate clusters (candidates — Kwame decides; the fixer never merges unilaterally):
|
||||
1. [WRITER-DEFECT] #AAA/#BBB/#CCC — writer <agent>, N nodes, span Nd (UNFIXED if it grew since the last report)
|
||||
2. [HUMAN-DECISION] #DDD/#EEE — same subject, bodies differ, SUGGEST: connect
|
||||
3. "none" when the scan returned no clusters
|
||||
|
||||
Orphans >90 days:
|
||||
1. [Node #EEE] Title — X days stale, orphaned
|
||||
@@ -195,6 +232,7 @@ The result must be 0 rows when all decisions are executed. Report what was done.
|
||||
- **State integrity:** archived nodes have `state: archived` + `[ARCHIVED]` prefix; kept nodes are `state: active` without a `[REVIEW:]` tag.
|
||||
- **Auto-archive applied:** no node should ever be left tagged `[REVIEW: archive]` — that tag is retired. Any `[REVIEW: archive]` found means fix 4 was skipped; archive it and report.
|
||||
- **No review-pending forever:** after executing Kwame's decisions, `[REVIEW:%` node count must be 0.
|
||||
- **Duplicate scan ran:** every report carries the fix 5 block (`none` when there were no clusters). A report with no duplicate section means phase 5 was skipped — a silently skipped detection phase is the failure this phase exists to prevent.
|
||||
- **Timestamps:** every executed decision (and every auto-archive) bumps `updated_at`, so the node exits the stale window on the next run.
|
||||
|
||||
## Logging
|
||||
|
||||
@@ -50,6 +50,11 @@ Changelog:
|
||||
(kagentz CT 105 on minipve, .14, dedicated `hermes` user) and is monitored
|
||||
from her side. This script must not probe mumuni or .24 — the v2 changelog
|
||||
roster line was the last reference still placing her at .24 / CT100.
|
||||
v6 (2026-09-28): .8 GPU health probe now runs as `llmuser` instead of `root`.
|
||||
Root SSH to .8 was lost when the guest was rebuilt, so every .8 leg read as
|
||||
UNREACHABLE for a healthy host. llmuser owns llama-server and can read
|
||||
`systemctl is-active`, `systemctl show -p MainPID`, and the :8080 pid.
|
||||
.110 and .15 keep the default `root` user.
|
||||
"""
|
||||
|
||||
import subprocess, json, sys, os, time, re, io, contextlib
|
||||
@@ -70,7 +75,7 @@ PVE_NODES = {
|
||||
|
||||
# Agent definitions: ct, host, user, pve_node, vault_key_name
|
||||
AGENTS = {
|
||||
"tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"},
|
||||
"tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "minipve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "hybrid"},
|
||||
# abiba = pi agent (.24) — no vault key; its LiteLLM key is read from its
|
||||
# local env file (key_env below), not from the shared vault or .bashrc.
|
||||
# runtime=pi: abiba has run pi-only since the harness purge. There is no
|
||||
@@ -95,7 +100,7 @@ AGENTS = {
|
||||
# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active)
|
||||
# .15 strixhalo (amdpve) -> strix-server.service (active)
|
||||
GPU_HOSTS = {
|
||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"},
|
||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service", "user": "llmuser"},
|
||||
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"},
|
||||
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"},
|
||||
}
|
||||
@@ -147,6 +152,18 @@ def ssh(host, cmd, user="root"):
|
||||
except:
|
||||
return None
|
||||
|
||||
def get_user_home(user):
|
||||
"""Resolve the home directory for a user.
|
||||
|
||||
For 'root', returns '/root'. For any other user, returns '/home/<user>'.
|
||||
This is used to construct paths that reference a user's home directory
|
||||
(e.g., ~/.local/bin/hermes, ~/.hermes/config.yaml) instead of hardcoding /root/.
|
||||
"""
|
||||
if user == "root":
|
||||
return "/root"
|
||||
else:
|
||||
return f"/home/{user}"
|
||||
|
||||
def http_get(url, headers=None, timeout=5):
|
||||
"""Return HTTP status code as string."""
|
||||
try:
|
||||
@@ -300,13 +317,14 @@ def check_gpu_ports():
|
||||
host = gpu["host"]
|
||||
port = gpu["port"]
|
||||
svc = gpu["service"]
|
||||
user = gpu.get("user", "root") # default root, overridden per-host where needed
|
||||
|
||||
# `systemctl is-active` exits non-zero when the unit is inactive or
|
||||
# missing, which the ssh() helper would swallow as an SSH failure and
|
||||
# report as UNREACHABLE. `|| true` keeps the real state word so we can
|
||||
# tell "unit inactive" from "host unreachable".
|
||||
svc_status = ssh(host, f"systemctl is-active {svc} || true")
|
||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1")
|
||||
svc_status = ssh(host, f"systemctl is-active {svc} || true", user=user)
|
||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1", user=user)
|
||||
|
||||
if not svc_status:
|
||||
print(f" ❌ {label}: UNREACHABLE")
|
||||
@@ -317,14 +335,14 @@ def check_gpu_ports():
|
||||
print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})")
|
||||
FAIL.append(f"gpu-no-port:{label}")
|
||||
elif svc_status != "active":
|
||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2")
|
||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2", user=user)
|
||||
if svc_pid and port_owner != svc_pid:
|
||||
print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})")
|
||||
FAIL.append(f"gpu-ghost:{label}:{port_owner}")
|
||||
else:
|
||||
print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}")
|
||||
else:
|
||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health")
|
||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health", user=user)
|
||||
if health and '"status":"ok"' in health:
|
||||
print(f" ✅ {label}: healthy (pid={port_owner})")
|
||||
elif health and '"status":"no slot available"' in health:
|
||||
@@ -376,10 +394,10 @@ def check_agents():
|
||||
ct = agent["ct"]
|
||||
report_only = agent.get("report_only", False)
|
||||
|
||||
# Tanko runs on DSH (DeepSeek Harness) since 2026-08-27 — it no longer runs a
|
||||
# Tanko runs hybrid (DSH + Hermes) since 2026-08-27 — it runs both DSH and Hermes gateway.
|
||||
# Hermes gateway, so skip the Hermes gateway/state/streaming/journal checks.
|
||||
# Non-Hermes runtimes have no gateway to probe. dsh = Tanko since
|
||||
# 2026-08-27; pi = abiba since the harness purge (.24 is pi-only).
|
||||
# Non-Hermes runtimes have no gateway to probe. dsh/pi-only skip the check;
|
||||
# hybrid runs both DSH and Hermes and is checked normally.
|
||||
if agent.get("runtime") in ("dsh", "pi"):
|
||||
is_dsh = agent.get("runtime") == "dsh"
|
||||
label = "DSH (DeepSeek Harness)" if is_dsh else "pi-only runtime"
|
||||
@@ -500,7 +518,7 @@ def check_ct_liveness():
|
||||
def check_config_integrity():
|
||||
"""Verify agent config.yaml parses as valid YAML."""
|
||||
for name, agent in AGENTS.items():
|
||||
# Tanko runs on DSH (DeepSeek Harness) since 2026-08-27 — no Hermes config.yaml.
|
||||
# DSH/pi-only runtimes have no Hermes config.yaml; hybrid has both.
|
||||
if agent.get("runtime") == "dsh":
|
||||
print(f" ⏭️ {name}: DSH — no Hermes config.yaml since 2026-08-27")
|
||||
continue
|
||||
@@ -513,11 +531,11 @@ def check_config_integrity():
|
||||
print(f" ⬜ {name}: cannot SSH — skip config check")
|
||||
continue
|
||||
|
||||
home = get_user_home(user)
|
||||
|
||||
# Check YAML parses
|
||||
yaml_ok = ssh(host,
|
||||
"python3 -c "
|
||||
'"import yaml; yaml.safe_load(open(\'/root/.hermes/config.yaml\')); print(\'OK\')" '
|
||||
"2>&1 || echo 'FAIL'",
|
||||
f"python3 -c \"import yaml; yaml.safe_load(open('{home}/.hermes/config.yaml')); print('OK')\" 2>&1 || echo 'FAIL'",
|
||||
user=user)
|
||||
if not yaml_ok:
|
||||
print(f" ❌ {name}: SSH UNREACHABLE (config check skipped)")
|
||||
@@ -556,7 +574,7 @@ def _infisical_invocation_paths(wrapper_body):
|
||||
def check_wrapper_integrity():
|
||||
"""Verify the hermes CLI wrapper exists and can reach hermes-real."""
|
||||
for name, agent in AGENTS.items():
|
||||
# Tanko runs on DSH (DeepSeek Harness) since 2026-08-27 — no hermes CLI wrapper.
|
||||
# DSH/pi-only runtimes have no hermes CLI wrapper; hybrid has both.
|
||||
if agent.get("runtime") == "dsh":
|
||||
print(f" ⏭️ {name}: DSH — no hermes CLI wrapper since 2026-08-27")
|
||||
continue
|
||||
@@ -570,7 +588,8 @@ def check_wrapper_integrity():
|
||||
continue
|
||||
|
||||
# Check wrapper exists
|
||||
wrapper = ssh(host, "ls -la /root/.local/bin/hermes 2>/dev/null", user=user)
|
||||
home = get_user_home(user)
|
||||
wrapper = ssh(host, f"ls -la {home}/.local/bin/hermes 2>/dev/null", user=user)
|
||||
if not wrapper:
|
||||
# Check alternate wrapper locations
|
||||
wrapper = ssh(host, "which hermes 2>/dev/null; command -v hermes 2>/dev/null", user=user)
|
||||
@@ -593,7 +612,7 @@ def check_wrapper_integrity():
|
||||
# a removed path (litellm-api-keys.prose.md documents
|
||||
# `rm -f /usr/local/bin/infisical`) must neither produce a dangling path
|
||||
# nor trigger the PATH check — it is not an invocation.
|
||||
wrapper_body = ssh(host, "cat /root/.local/bin/hermes 2>/dev/null", user=user) or ""
|
||||
wrapper_body = ssh(host, f"cat {home}/.local/bin/hermes 2>/dev/null", user=user) or ""
|
||||
wrapper_code = "\n".join(line.split("#", 1)[0] for line in wrapper_body.splitlines())
|
||||
invoked_paths = _infisical_invocation_paths(wrapper_body)
|
||||
if "infisical" in wrapper_code:
|
||||
@@ -627,24 +646,35 @@ def check_wrapper_integrity():
|
||||
else:
|
||||
print(f" ℹ️ {name}: wrapper resolves creds without infisical (e.g. ~/.hermes/.env) — OK")
|
||||
|
||||
# Check hermes-real exists
|
||||
# Check that the wrapper's target resolves. The fleet's wrappers do NOT
|
||||
# all use a hermes-real indirection — some exec the venv module directly.
|
||||
# Verify the wrapper actually points to something runnable.
|
||||
hermes_real = ssh(host,
|
||||
"ls -la /root/.local/bin/hermes-real 2>/dev/null || echo MISS",
|
||||
f"ls -la {home}/.local/bin/hermes-real 2>/dev/null || echo MISS",
|
||||
user=user)
|
||||
if not hermes_real or hermes_real.strip() == "MISS":
|
||||
# Check venv path
|
||||
hermes_real = ssh(host,
|
||||
"ls -la /usr/local/lib/hermes-agent/venv/bin/hermes 2>/dev/null || echo MISS",
|
||||
if hermes_real and hermes_real.strip() != "MISS":
|
||||
print(f" ✅ {name}: wrapper shape: hermes-real at {home}/.local/bin/hermes-real")
|
||||
else:
|
||||
# Try the venv under home
|
||||
venv_home = ssh(host,
|
||||
f"test -x {home}/.hermes/hermes-agent/venv/bin/python && echo OK || echo MISS",
|
||||
user=user)
|
||||
if not hermes_real or hermes_real.strip() == "MISS":
|
||||
print(f" ❌ {name}: hermes-real NOT FOUND (wrapper broken)")
|
||||
_fail(f"wrapper-no-hermes-real:{name}", name)
|
||||
if venv_home and venv_home.strip().splitlines()[-1] == "OK":
|
||||
print(f" ✅ {name}: wrapper shape: direct venv exec ({home}/.hermes/hermes-agent/venv/bin/python)")
|
||||
else:
|
||||
print(f" ✅ {name}: hermes-real at alt path")
|
||||
# Try the system-wide venv
|
||||
venv_sys = ssh(host,
|
||||
"test -x /usr/local/lib/hermes-agent/venv/bin/python && echo OK || echo MISS",
|
||||
user=user)
|
||||
if venv_sys and venv_sys.strip().splitlines()[-1] == "OK":
|
||||
print(f" ✅ {name}: wrapper shape: system venv (/usr/local/lib/hermes-agent/venv/bin/python)")
|
||||
else:
|
||||
print(f" ❌ {name}: wrapper target NOT RESOLVABLE (no hermes-real, no venv)")
|
||||
_fail(f"wrapper-no-hermes-real:{name}", name)
|
||||
|
||||
# Check the .env file has the key
|
||||
env_has_key = ssh(host,
|
||||
"grep -c 'LITELLM_API_KEY' /root/.hermes/.env 2>/dev/null || echo 0",
|
||||
f"grep -c 'LITELLM_API_KEY' {home}/.hermes/.env 2>/dev/null || echo 0",
|
||||
user=user)
|
||||
if env_has_key and env_has_key.strip() not in ("", "0"):
|
||||
print(f" ✅ {name}: wrapper + .env key present")
|
||||
|
||||
+169
-38
@@ -243,7 +243,7 @@ def collect():
|
||||
("Pulse", "https://pulse.sysloggh.net"),
|
||||
("Proxmox", "https://192.168.68.12:8006"),
|
||||
("SearXNG", "http://192.168.68.7:8888"),
|
||||
("Firecrawl", "http://192.168.68.7:3002/health"),
|
||||
("Firecrawl", "http://192.168.68.7:3002/"), # Firecrawl serves no /health - the root is its liveness endpoint
|
||||
]
|
||||
report["endpoints"] = []
|
||||
for name, url in endpoints:
|
||||
@@ -389,6 +389,28 @@ def collect():
|
||||
|
||||
# ── HTML Dashboard ──
|
||||
|
||||
def classify_endpoint(code):
|
||||
"""Classify an endpoint probe per the fleet's probe policy.
|
||||
|
||||
Codified 2026-09-14 in the monitoring contracts: ANY HTTP status proves the
|
||||
service answered, so the service is ALIVE - 200/301/302/401/403/404 alike.
|
||||
Only a failed CONNECTION (000 / timeout / refused) is a failed probe. A 404
|
||||
from a wrong path is not a service fault and must not render as one.
|
||||
|
||||
This replaces a string comparison that was wrong in both directions
|
||||
(`ep["code"] >= "400"`): it rendered 301 as red, 404 as yellow, and a real
|
||||
500 as yellow. 5xx is kept as its own "server error" signal rather than
|
||||
being merged with 4xx.
|
||||
"""
|
||||
if not code or code == "000":
|
||||
return "red", "no connection"
|
||||
if code.startswith("5"):
|
||||
return "yellow", "server error"
|
||||
if code.startswith(("2", "3", "4")):
|
||||
return "green", "alive"
|
||||
return "yellow", f"unexpected {code}"
|
||||
|
||||
|
||||
def build_html(r):
|
||||
issues = []
|
||||
|
||||
@@ -624,7 +646,7 @@ Proxmox: {r.get('pve_probe_status', 'ok')} ({r['nodes_online']}/{r['node_count']
|
||||
# ── Network Endpoints ──
|
||||
html += '<div class="card"><h2>🌐 Network Endpoints</h2><table><tr><th>Service</th><th>Status</th></tr>'
|
||||
for ep in r["endpoints"]:
|
||||
color = "green" if ep["code"] in ("200","302","401") else ("yellow" if ep["code"] >= "400" else "red")
|
||||
color = classify_endpoint(ep["code"])[0]
|
||||
html += f'<tr><td>{ep["name"]}</td><td class="{color}">HTTP {ep["code"]}</td></tr>'
|
||||
html += '</table></div>'
|
||||
|
||||
@@ -688,42 +710,152 @@ Proxmox: {r.get('pve_probe_status', 'ok')} ({r['nodes_online']}/{r['node_count']
|
||||
return html
|
||||
|
||||
|
||||
# ── Send Email ──
|
||||
# ── Delivery: Zulip DM carrying the report as an HTML ATTACHMENT ──
|
||||
#
|
||||
# Captain's decision, clarified 2026-09-26: the report is sent as an HTML FILE,
|
||||
# i.e. an attachment - NOT HTML rendered in the message body, and NOT a Markdown
|
||||
# translation of it. So the styled dashboard is built exactly as before, uploaded
|
||||
# through Zulip's file-upload API, and the message body stays short: subject,
|
||||
# top-line status, and a pointer to the attachment.
|
||||
#
|
||||
# This removes the Google dependency entirely (no SMTP, no EMAIL_PASSWORD).
|
||||
# The 10,000-character message cap does not apply: it bounds message TEXT only,
|
||||
# and the report travels as a file.
|
||||
|
||||
def send_email(html_content, subject_prefix=""):
|
||||
FROM = "abiba@sysloggh.com"
|
||||
TO = "jerome@sysloggh.com"
|
||||
SUBJECT = f"{subject_prefix}{'🏗️ Infrastructure Report — ' + DATE_STR}"
|
||||
|
||||
msg = MIMEMultipart("alternative")
|
||||
msg["From"] = FROM
|
||||
msg["To"] = TO
|
||||
msg["Subject"] = SUBJECT
|
||||
msg.attach(MIMEText("Infrastructure report in HTML format — enable images to view.", "plain"))
|
||||
msg.attach(MIMEText(html_content, "html"))
|
||||
|
||||
ZULIP_SITE = "https://chat.sysloggh.net"
|
||||
ZULIP_BOT_EMAIL = "abiba-bot@chat.sysloggh.net"
|
||||
CAPTAIN_USER_ID = 9
|
||||
ZULIP_KEY_FILE = "/root/.pi/agent/extensions/zulip/.env"
|
||||
REPORT_ARTIFACT_DIR = "/var/log/daily-infra-report"
|
||||
|
||||
|
||||
def zulip_key():
|
||||
"""abiba-bot's Zulip key, from the env or the on-host 600 file."""
|
||||
key = os.environ.get("ABIBA_ZULIP_API_KEY")
|
||||
if key:
|
||||
return key.strip()
|
||||
try:
|
||||
EMAIL_PASSWORD = os.environ.get("EMAIL_PASSWORD") or os.environ.get("SMTP_PASSWORD") or os.environ.get("MAIL_PASSWORD")
|
||||
if not EMAIL_PASSWORD:
|
||||
print(" ⚠️ Degraded leg: credential-missing: EMAIL_PASSWORD (or SMTP_PASSWORD/MAIL_PASSWORD)", file=sys.stderr)
|
||||
DEGRADED_LEGS.append("credential-missing: EMAIL_PASSWORD")
|
||||
return True, "✅ Email leg degraded (no credential) — report still produced"
|
||||
GMAIL_EMAIL = "jtabiri@gmail.com"
|
||||
|
||||
server = smtplib.SMTP("smtp.gmail.com", 587)
|
||||
server.starttls()
|
||||
server.login(GMAIL_EMAIL, EMAIL_PASSWORD)
|
||||
server.sendmail(FROM, [TO], msg.as_string())
|
||||
server.quit()
|
||||
return True, "✅ Email sent to jerome@sysloggh.com"
|
||||
except Exception as e:
|
||||
return False, f"❌ Email failed: {e}"
|
||||
with open(ZULIP_KEY_FILE) as fh:
|
||||
for line in fh:
|
||||
if line.startswith("ABIBA_ZULIP_API_KEY="):
|
||||
return line.split("=", 1)[1].strip()
|
||||
except OSError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def build_summary(r, filename, test=False):
|
||||
"""Short Markdown body: subject, top-line status, pointer to the attachment.
|
||||
|
||||
Deliberately NOT a reproduction of the report - the attachment is the report.
|
||||
"""
|
||||
nodes = f"{r.get('nodes_online', 0)}/{r.get('node_count', 0)} nodes online"
|
||||
guests = f"{r.get('running_vms', 0)}/{r.get('total_vms', 0)} guests running"
|
||||
lines = [
|
||||
("\U0001F9EA **TEST — **" if test else "") + "\U0001F3D7\uFE0F **Infrastructure Report — " + DATE_STR + "**",
|
||||
f"**{nodes}** \u00b7 **{guests}** \u00b7 generated {TIME_STR}",
|
||||
]
|
||||
problems = []
|
||||
if r.get("pve_probe_status") != "ok":
|
||||
problems.append(f"\u274c Proxmox probe: {r.get('pve_probe_status')}")
|
||||
if r.get("resources_probe_status") != "ok":
|
||||
problems.append(f"\u274c Resources probe: {r.get('resources_probe_status')}")
|
||||
lit = r.get("litellm", {}) or {}
|
||||
checks = lit.get("checks", []) or []
|
||||
if checks:
|
||||
passed = sum(1 for c in checks if c.get("status") == "pass")
|
||||
if passed != len(checks):
|
||||
problems.append(f"\u274c LiteLLM: {passed}/{len(checks)} checks pass")
|
||||
if not (r.get("zulip_ext", {}) or {}).get("connected"):
|
||||
problems.append("\u274c Zulip extension: not connected")
|
||||
for leg in DEGRADED_LEGS:
|
||||
problems.append(f"\u26a0\uFE0F degraded: {leg}")
|
||||
|
||||
lines.append("\n".join(problems) if problems else "\u2705 All monitored services healthy")
|
||||
lines.append(f"\U0001F4CE **Full report attached:** `{filename}`")
|
||||
return "\n\n".join(lines)
|
||||
|
||||
|
||||
def _curl(args, timeout=60):
|
||||
r = subprocess.run(["curl", "-s", "-m", str(timeout)] + args,
|
||||
capture_output=True, text=True)
|
||||
try:
|
||||
return json.loads(r.stdout or "{}"), r.stdout
|
||||
except json.JSONDecodeError:
|
||||
return {}, r.stdout
|
||||
|
||||
|
||||
def _curl_json(args, timeout=90):
|
||||
r = subprocess.run(["curl", "-s", "-m", str(timeout)] + args,
|
||||
capture_output=True, text=True)
|
||||
try:
|
||||
return json.loads(r.stdout or "{}"), r.stdout
|
||||
except json.JSONDecodeError:
|
||||
return {}, r.stdout
|
||||
|
||||
|
||||
def send_zulip(html_content, report, test=False):
|
||||
"""Upload the styled HTML and post a short pointer to the captain's DM.
|
||||
|
||||
Returns (ok, message). On ANY failure the report body is also printed to
|
||||
stdout and persisted to disk, so a delivery failure can never swallow the
|
||||
content - the defect this folds in.
|
||||
"""
|
||||
os.makedirs(REPORT_ARTIFACT_DIR, exist_ok=True)
|
||||
stamp = NOW.strftime("%Y%m%d-%H%M%S")
|
||||
filename = f"infra-report-{stamp}.html"
|
||||
html_path = os.path.join(REPORT_ARTIFACT_DIR, filename)
|
||||
try:
|
||||
with open(html_path, "w") as fh:
|
||||
fh.write(html_content)
|
||||
except OSError as e:
|
||||
print(f" \u26a0\uFE0F could not persist report artifact: {e}", file=sys.stderr)
|
||||
|
||||
key = zulip_key()
|
||||
if not key:
|
||||
print(html_content) # never swallow the content
|
||||
return False, ("\u274c Delivery FAILED: no Zulip credential "
|
||||
"(ABIBA_ZULIP_API_KEY unset and "
|
||||
f"{ZULIP_KEY_FILE} unreadable). Report persisted to {html_path}")
|
||||
|
||||
auth = ["-u", f"{ZULIP_BOT_EMAIL}:{key}"]
|
||||
|
||||
# 1. Upload the report as a file.
|
||||
up, up_raw = _curl_json(auth + [
|
||||
"-X", "POST", f"{ZULIP_SITE}/api/v1/user_uploads",
|
||||
"-F", f"file=@{html_path};type=text/html",
|
||||
])
|
||||
if up.get("result") != "success" or not up.get("uri"):
|
||||
print(html_content)
|
||||
return False, (f"\u274c Delivery FAILED at upload: {up.get('msg') or up_raw[:160]} "
|
||||
f"(report persisted to {html_path})")
|
||||
|
||||
uri = up["uri"]
|
||||
size = os.path.getsize(html_path)
|
||||
|
||||
# 2. Post a short message pointing at it.
|
||||
body = build_summary(report, filename, test=test)
|
||||
link = f"[{filename}]({uri})"
|
||||
body = body.replace(f"`{filename}`", link)
|
||||
payload, raw = _curl_json(auth + [
|
||||
"-X", "POST", f"{ZULIP_SITE}/api/v1/messages",
|
||||
"-d", "type=private",
|
||||
"-d", f"to=[{CAPTAIN_USER_ID}]",
|
||||
"--data-urlencode", f"content={body}",
|
||||
])
|
||||
if payload.get("result") == "success":
|
||||
return True, (f"\u2705 Delivered to Zulip DM (user {CAPTAIN_USER_ID}), "
|
||||
f"message id {payload.get('id')}, attachment {size} bytes at {uri}")
|
||||
|
||||
print(html_content)
|
||||
return False, (f"\u274c Delivery FAILED at message post: {payload.get('msg') or raw[:160]} "
|
||||
f"(uploaded {uri}; report persisted to {html_path})")
|
||||
|
||||
|
||||
# ── Main ──
|
||||
|
||||
if __name__ == "__main__":
|
||||
is_test = "--test-email" in sys.argv
|
||||
is_test = ("--test-email" in sys.argv) or ("--test-zulip" in sys.argv)
|
||||
|
||||
print(f"{'🧪 TEST MODE' if is_test else '📊'} Collecting infrastructure data...")
|
||||
report = collect()
|
||||
@@ -738,15 +870,14 @@ if __name__ == "__main__":
|
||||
|
||||
print(" Building dashboard...")
|
||||
html = build_html(report)
|
||||
|
||||
print(f" report ready: {len(html)} chars of HTML (delivered as a file attachment)")
|
||||
|
||||
if is_test:
|
||||
prefix = "🧪 TEST — "
|
||||
print(" Sending test email...")
|
||||
print(" Sending TEST message to the captain's Zulip DM...")
|
||||
else:
|
||||
prefix = ""
|
||||
print(" Sending email...")
|
||||
|
||||
ok, msg = send_email(html, subject_prefix=prefix)
|
||||
print(" Sending to the captain's Zulip DM...")
|
||||
|
||||
ok, msg = send_zulip(html, report, test=is_test)
|
||||
print(f" {msg}")
|
||||
|
||||
# Show summary
|
||||
|
||||
@@ -101,8 +101,6 @@ GUESTS: list[Guest] = [
|
||||
# amdpve (192.168.68.15)
|
||||
Guest(ct_id="105", hostname="kagentz", ip="192.168.68.105", node="amdpve",
|
||||
access_method="ssh-host", probe_target="kagentz (CT 105, amdpve)"),
|
||||
Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="amdpve",
|
||||
access_method="pct-run", probe_target="tanko (CT 112, amdpve)"),
|
||||
Guest(ct_id="113", hostname="baggy", ip="192.168.68.113", node="amdpve",
|
||||
access_method="pct-run", probe_target="baggy (CT 113, amdpve)"),
|
||||
Guest(ct_id="115", hostname="scottdenya", ip="192.168.68.115", node="amdpve",
|
||||
@@ -110,6 +108,8 @@ GUESTS: list[Guest] = [
|
||||
Guest(ct_id="120", hostname="adguard2", ip="192.168.68.120", node="amdpve",
|
||||
access_method="pct-run", probe_target="adguard2 (CT 120, amdpve)"),
|
||||
# minipve (192.168.68.12)
|
||||
Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="minipve",
|
||||
access_method="pct-run", probe_target="tanko (CT 112, minipve)"),
|
||||
Guest(ct_id="100", hostname="abiba", ip="192.168.68.100", node="minipve",
|
||||
access_method="pct-run", probe_target="abiba (CT 100, minipve)"),
|
||||
Guest(ct_id="102", hostname="adguard", ip="192.168.68.102", node="minipve",
|
||||
|
||||
+1
-1
@@ -12,7 +12,6 @@ set -euo pipefail
|
||||
declare -A CT_NODES=(
|
||||
# amdpve (192.168.68.15)
|
||||
[105]=amdpve # kagentz (was hwepve — corrected 2026-09-12; live per pvesh)
|
||||
[112]=amdpve # tanko
|
||||
[113]=amdpve # baggy
|
||||
[115]=amdpve # scottdenya
|
||||
[120]=amdpve # adguard2 (added 2026-09-12)
|
||||
@@ -21,6 +20,7 @@ declare -A CT_NODES=(
|
||||
[102]=minipve # adguard (was acerpve)
|
||||
[104]=minipve # authentik
|
||||
[110]=minipve # gitea
|
||||
[112]=minipve # tanko (was amdpve — migrated 2026-09-27; live per pvesh)
|
||||
[116]=minipve # syslog-api
|
||||
[119]=minipve # infisical-vault
|
||||
# storepve (192.168.68.6)
|
||||
|
||||
@@ -47,8 +47,8 @@ You are a code reviewer for OpenProse infrastructure contracts in the Syslog Sol
|
||||
The infrastructure-control.prose.md contract is the canonical reference for the cluster topology:
|
||||
|
||||
**Proxmox Cluster "Tabiri" (5 nodes):**
|
||||
- amdpve (192.168.68.15): kagentz, tanko, baggy, scottdenya, adguard2
|
||||
- minipve (192.168.68.12): abiba, adguard, authentik, gitea, syslog-api, infisical-vault
|
||||
- amdpve (192.168.68.15): kagentz, baggy, scottdenya, adguard2
|
||||
- minipve (192.168.68.12): abiba, tanko, adguard, authentik, gitea, syslog-api, infisical-vault
|
||||
- storepve (192.168.68.6): docker-vm, ra-h-os, PBS, media, jdownloader, zulip, tdunna
|
||||
- acerpve (192.168.68.9): llm-gpu
|
||||
- ocupve (192.168.68.5): ocu-llm
|
||||
|
||||
Executable
+395
@@ -0,0 +1,395 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Agent-consumption layer in front of SearXNG + Firecrawl.
|
||||
|
||||
Multi-engine aggregation returns results with no dedupe, no filtering and no
|
||||
reranking. Measured 2026-09-26 that put bestbuy.com and merriam-webster.com into
|
||||
"best practices agent context management", and put four SEO blogs ABOVE the real
|
||||
Proxmox forum threads on a precise technical query. Identical queries also ranked
|
||||
differently between runs, so the fix has to be deterministic rather than
|
||||
dependent on engine mood.
|
||||
|
||||
This module turns the raw result list into something an agent can actually use:
|
||||
|
||||
1. DEDUPE the same page arriving from several engines
|
||||
2. DROP clear non-answers (homepages, shopping, dictionaries, logins)
|
||||
3. DEMOTE config-listed low-authority hosts; PROMOTE primary sources
|
||||
4. STABLE SORT so ordering is reproducible run to run
|
||||
5. EXTRACT page text for the top N under an explicit character budget,
|
||||
so one call returns usable material instead of a snippet
|
||||
6. EMIT stable JSON with engine provenance
|
||||
|
||||
Policy lives in config/search-ranking.yaml, not in this file.
|
||||
|
||||
Usage:
|
||||
search-agent-consume.py "query text" # JSON to stdout
|
||||
search-agent-consume.py --no-extract "query" # ranking only, no Firecrawl
|
||||
search-agent-consume.py --explain "query" # include drop/demote reasons
|
||||
|
||||
Exit: 0 ok, 1 no results survived filtering, 2 the layer could not run.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/")
|
||||
FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/")
|
||||
CONFIG_PATH = os.environ.get(
|
||||
"SEARCH_RANKING_CONFIG",
|
||||
str(Path(__file__).resolve().parent.parent / "config" / "search-ranking.yaml"),
|
||||
)
|
||||
HTTP_TIMEOUT = float(os.environ.get("SEARCH_CONSUME_TIMEOUT", "25"))
|
||||
|
||||
|
||||
def _load_config() -> dict:
|
||||
"""Load the ranking policy.
|
||||
|
||||
PyYAML is used when present; otherwise a tiny built-in parser handles the
|
||||
flat lists in this specific file, so the layer never hard-fails on a host
|
||||
without PyYAML.
|
||||
"""
|
||||
text = Path(CONFIG_PATH).read_text()
|
||||
try:
|
||||
import yaml # type: ignore
|
||||
|
||||
return yaml.safe_load(text)
|
||||
except ImportError:
|
||||
return _parse_flat_yaml(text)
|
||||
|
||||
|
||||
def _parse_flat_yaml(text: str) -> dict:
|
||||
"""Minimal fallback parser: top-level keys, nested one level, flat lists."""
|
||||
import re
|
||||
|
||||
out: dict = {}
|
||||
stack: list[tuple[int, dict]] = [(-1, out)]
|
||||
section: dict | None = None
|
||||
for raw in text.splitlines():
|
||||
line = raw.split("#", 1)[0].rstrip()
|
||||
if not line.strip():
|
||||
continue
|
||||
indent = len(line) - len(line.lstrip())
|
||||
body = line.strip()
|
||||
if body.startswith("- "):
|
||||
if section is not None:
|
||||
section.setdefault("_list", []).append(
|
||||
body[2:].strip().strip("'\"")
|
||||
)
|
||||
continue
|
||||
if ":" in body:
|
||||
key, _, val = body.partition(":")
|
||||
key, val = key.strip(), val.strip()
|
||||
if val:
|
||||
# write to the INNERMOST open section, not the document root
|
||||
stack[-1][1][key] = _scalar(val)
|
||||
section = None
|
||||
else:
|
||||
while stack and indent <= stack[-1][0]:
|
||||
stack.pop()
|
||||
parent = stack[-1][1]
|
||||
new: dict = {}
|
||||
parent[key] = new
|
||||
stack.append((indent, new))
|
||||
section = new
|
||||
# flatten "_list" holders back into their parent as plain lists
|
||||
def fix(node):
|
||||
if isinstance(node, dict):
|
||||
if set(node.keys()) == {"_list"}:
|
||||
return node["_list"]
|
||||
return {k: fix(v) for k, v in node.items()}
|
||||
return node
|
||||
|
||||
return fix(out)
|
||||
|
||||
|
||||
def _scalar(v: str):
|
||||
if v.lower() in ("true", "false"):
|
||||
return v.lower() == "true"
|
||||
try:
|
||||
return int(v)
|
||||
except ValueError:
|
||||
pass
|
||||
try:
|
||||
return float(v)
|
||||
except ValueError:
|
||||
pass
|
||||
return v.strip("'\"")
|
||||
|
||||
|
||||
# ── filtering ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _host(url: str) -> str:
|
||||
return (urllib.parse.urlparse(url).netloc or "").lower().split(":")[0]
|
||||
|
||||
|
||||
def _registrable(host: str) -> str:
|
||||
"""Best-effort registrable domain so sub.forum.proxmox.com matches proxmox.com."""
|
||||
parts = host.split(".")
|
||||
if len(parts) <= 2:
|
||||
return host
|
||||
# handle common two-label public suffixes
|
||||
two = ".".join(parts[-2:])
|
||||
if parts[-2] in ("co", "com", "org", "net", "ac", "gov") and len(parts) >= 3:
|
||||
return ".".join(parts[-3:])
|
||||
return two
|
||||
|
||||
|
||||
def _host_in(host: str, domains) -> bool:
|
||||
if not domains:
|
||||
return False
|
||||
reg = _registrable(host)
|
||||
for d in domains:
|
||||
d = str(d).lower()
|
||||
if host == d or host.endswith("." + d) or reg == d:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _normalise_url(url: str) -> str:
|
||||
"""Strip tracking params and fragments so the same page dedupes."""
|
||||
p = urllib.parse.urlparse(url)
|
||||
q = [
|
||||
(k, v)
|
||||
for k, v in urllib.parse.parse_qsl(p.query, keep_blank_values=True)
|
||||
if not k.lower().startswith(("utm_", "fbclid", "gclid", "mc_", "ref"))
|
||||
]
|
||||
path = p.path.rstrip("/") or "/"
|
||||
return urllib.parse.urlunparse(
|
||||
(p.scheme.lower(), p.netloc.lower(), path, "", urllib.parse.urlencode(q), "")
|
||||
)
|
||||
|
||||
|
||||
def non_answer_reason(result: dict, cfg: dict) -> str | None:
|
||||
"""Return why this result is a non-answer, or None if it may be returned."""
|
||||
na = cfg.get("non_answer", {}) or {}
|
||||
url = result.get("url", "")
|
||||
p = urllib.parse.urlparse(url)
|
||||
host = _host(url)
|
||||
path = p.path or ""
|
||||
|
||||
if _host_in(host, na.get("hosts")):
|
||||
return "shopping_or_dictionary_host"
|
||||
|
||||
if na.get("host_root", True) and path in ("", "/"):
|
||||
# A preferred host's front door may legitimately be the answer
|
||||
# (a repo, a docs site). Everything else is navigational.
|
||||
if not _host_in(host, cfg.get("prefer_domains")):
|
||||
return "navigational_host_root"
|
||||
|
||||
low = url.lower()
|
||||
for pat in na.get("path_patterns", []) or []:
|
||||
if pat.lower() in low:
|
||||
return f"path_pattern:{pat}"
|
||||
|
||||
qkeys = {k.lower() for k in (na.get("query_keys") or [])}
|
||||
if qkeys & {k.lower() for k, _ in urllib.parse.parse_qsl(p.query)}:
|
||||
return "search_or_shopping_query"
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def source_type(url: str, cfg: dict) -> str:
|
||||
host = _host(url)
|
||||
if _host_in(host, ["github.com", "gitlab.com", "codeberg.org", "sourceforge.net"]):
|
||||
return "code"
|
||||
if _host_in(host, ["stackoverflow.com", "stackexchange.com", "superuser.com",
|
||||
"serverfault.com", "askubuntu.com"]):
|
||||
return "qa"
|
||||
if _host_in(host, ["forum.proxmox.com", "forum.", "discourse"]) or "forum." in host:
|
||||
return "forum"
|
||||
if _host_in(host, ["news.ycombinator.com", "lobste.rs", "reddit.com"]):
|
||||
return "discussion"
|
||||
if _host_in(host, cfg.get("prefer_domains")):
|
||||
return "official"
|
||||
if _host_in(host, cfg.get("demote_domains")):
|
||||
return "content-farm"
|
||||
return "web"
|
||||
|
||||
|
||||
def rank(results: list[dict], cfg: dict) -> tuple[list[dict], list[dict]]:
|
||||
"""Dedupe, drop non-answers, demote/ promote, stable sort.
|
||||
|
||||
Returns (kept, dropped) where dropped carries the reason, because a filter
|
||||
nobody can audit is a filter nobody should trust.
|
||||
"""
|
||||
rank_cfg = cfg.get("ranking", {}) or {}
|
||||
demote_pen = float(rank_cfg.get("demote_penalty", 1000))
|
||||
prefer_bonus = float(rank_cfg.get("prefer_bonus", 100))
|
||||
multi_bonus = float(rank_cfg.get("multi_engine_bonus", 25))
|
||||
|
||||
seen: dict[str, dict] = {}
|
||||
dropped: list[dict] = []
|
||||
|
||||
for pos, r in enumerate(results):
|
||||
url = r.get("url")
|
||||
if not url:
|
||||
continue
|
||||
key = _normalise_url(url)
|
||||
engine = r.get("engine", "?")
|
||||
|
||||
# 1. dedupe: same normalised URL from several engines
|
||||
if key in seen:
|
||||
seen[key].setdefault("engines", []).append(engine)
|
||||
seen[key]["duplicate_of"] = True
|
||||
continue
|
||||
|
||||
reason = non_answer_reason(r, cfg)
|
||||
if reason:
|
||||
dropped.append({"url": url, "reason": reason, "position": pos + 1})
|
||||
continue
|
||||
|
||||
seen[key] = {
|
||||
"title": (r.get("title") or "").strip(),
|
||||
"url": url,
|
||||
"engines": [engine],
|
||||
"position": pos,
|
||||
"score": 0.0,
|
||||
}
|
||||
|
||||
kept = []
|
||||
for item in seen.values():
|
||||
host = _host(item["url"])
|
||||
score = -float(item["position"]) # original order is the base signal
|
||||
if _host_in(host, cfg.get("demote_domains")):
|
||||
score -= demote_pen
|
||||
if _host_in(host, cfg.get("prefer_domains")):
|
||||
score += prefer_bonus
|
||||
if len(item["engines"]) > 1:
|
||||
score += multi_bonus * (len(item["engines"]) - 1)
|
||||
item["score"] = round(score, 2)
|
||||
item["host"] = host
|
||||
item["source_type"] = source_type(item["url"], cfg)
|
||||
kept.append(item)
|
||||
|
||||
# stable: score desc, then original position asc => reproducible run to run
|
||||
kept.sort(key=lambda i: (-i["score"], i["position"]))
|
||||
return kept, dropped
|
||||
|
||||
|
||||
# ── extraction ───────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _post_json(url: str, payload: dict, timeout: float) -> dict:
|
||||
req = urllib.request.Request(
|
||||
url,
|
||||
data=json.dumps(payload).encode(),
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
return json.loads(resp.read().decode("utf-8", "replace"))
|
||||
|
||||
|
||||
def extract(items: list[dict], cfg: dict) -> dict:
|
||||
"""Fetch page text for the top N under a global character budget."""
|
||||
ex = cfg.get("extraction", {}) or {}
|
||||
top_n = int(ex.get("top_n", 5))
|
||||
total_budget = int(ex.get("total_chars", 12000))
|
||||
per_item = int(ex.get("per_item_chars", 4000))
|
||||
timeout = float(ex.get("timeout_seconds", 45))
|
||||
|
||||
used = 0
|
||||
failures = 0
|
||||
t0 = time.time()
|
||||
for item in items[:top_n]:
|
||||
remaining = total_budget - used
|
||||
if remaining <= 200:
|
||||
item["excerpt"] = ""
|
||||
item["extraction"] = "skipped_budget_exhausted"
|
||||
continue
|
||||
cap = min(per_item, remaining)
|
||||
try:
|
||||
data = _post_json(
|
||||
f"{FIRECRAWL_URL}/v1/scrape",
|
||||
{"url": item["url"], "formats": ["markdown"]},
|
||||
timeout,
|
||||
)
|
||||
md = ((data.get("data") or {}).get("markdown") or "").strip()
|
||||
if not md:
|
||||
item["excerpt"] = ""
|
||||
item["extraction"] = "empty"
|
||||
failures += 1
|
||||
continue
|
||||
item["excerpt"] = md[:cap]
|
||||
item["extraction"] = "ok" if len(md) <= cap else "truncated"
|
||||
used += len(item["excerpt"])
|
||||
except Exception as exc: # noqa: BLE001
|
||||
item["excerpt"] = ""
|
||||
item["extraction"] = f"failed:{type(exc).__name__}"
|
||||
failures += 1
|
||||
return {
|
||||
"extracted": min(top_n, len(items)),
|
||||
"chars_used": used,
|
||||
"budget": total_budget,
|
||||
"failures": failures,
|
||||
"seconds": round(time.time() - t0, 2),
|
||||
}
|
||||
|
||||
|
||||
# ── entry point ──────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def consume(query: str, do_extract: bool = True, explain: bool = False) -> dict:
|
||||
cfg = _load_config()
|
||||
url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode(
|
||||
{"q": query, "format": "json"}
|
||||
)
|
||||
with urllib.request.urlopen(url, timeout=HTTP_TIMEOUT) as resp:
|
||||
raw = json.loads(resp.read().decode("utf-8", "replace"))
|
||||
|
||||
results = raw.get("results", [])
|
||||
kept, dropped = rank(results, cfg)
|
||||
extraction = extract(kept, cfg) if do_extract else None
|
||||
|
||||
out = {
|
||||
"query": query,
|
||||
"raw_result_count": len(results),
|
||||
"returned_count": len(kept),
|
||||
"dropped_count": len(dropped),
|
||||
"engines": sorted({r.get("engine", "?") for r in results}),
|
||||
"results": [
|
||||
{
|
||||
"rank": i + 1,
|
||||
"title": it["title"],
|
||||
"url": it["url"],
|
||||
"host": it["host"],
|
||||
"source_type": it["source_type"],
|
||||
"engines": sorted(set(it["engines"])),
|
||||
"score": it["score"],
|
||||
"excerpt": it.get("excerpt", ""),
|
||||
"extraction": it.get("extraction", "not_attempted"),
|
||||
}
|
||||
for i, it in enumerate(kept)
|
||||
],
|
||||
"extraction": extraction,
|
||||
}
|
||||
if explain:
|
||||
out["dropped"] = dropped
|
||||
return out
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||
do_extract = "--no-extract" not in sys.argv
|
||||
explain = "--explain" in sys.argv
|
||||
if not args:
|
||||
print(__doc__)
|
||||
return 2
|
||||
query = " ".join(args)
|
||||
try:
|
||||
out = consume(query, do_extract=do_extract, explain=explain)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
print(f"LAYER FAILED: {type(exc).__name__}: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
print(json.dumps(out, indent=2))
|
||||
return 0 if out["returned_count"] else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -114,6 +114,79 @@ def unresponsive_names(pairs: list) -> dict[str, str]:
|
||||
return out
|
||||
|
||||
|
||||
# ── QUALITY GUARD (search-agent-consumption) ─────────────────────────────────
|
||||
# The agent-consumption layer applies a deterministic demote/drop policy. Without
|
||||
# an assertion here it could silently rot back to raw engine ordering - the same
|
||||
# way the endpoint colours silently rotted before 2026-09-26.
|
||||
QUALITY_QUERIES = [
|
||||
"best practices agent context management",
|
||||
"proxmox thin pool metadata exhaustion recovery",
|
||||
]
|
||||
# A demoted (content-farm) host must never occupy the top 3 for these queries.
|
||||
QUALITY_TOP_N = 3
|
||||
# Non-answers that must never be returned for these queries at all.
|
||||
QUALITY_BANNED_HOSTS = ["bestbuy.com", "merriam-webster.com"]
|
||||
|
||||
|
||||
def _consumption_layer_path():
|
||||
here = os.path.dirname(os.path.abspath(__file__))
|
||||
return os.path.join(here, "search-agent-consume.py")
|
||||
|
||||
|
||||
def check_ranking_quality() -> list[str]:
|
||||
"""Return a list of quality failures; empty means healthy."""
|
||||
import subprocess as _sp
|
||||
|
||||
layer = _consumption_layer_path()
|
||||
if not os.path.exists(layer):
|
||||
return [f"agent-consumption layer missing: {layer}"]
|
||||
|
||||
failures: list[str] = []
|
||||
for query in QUALITY_QUERIES:
|
||||
r = _sp.run([sys.executable, layer, "--no-extract", "--explain", query],
|
||||
capture_output=True, text=True, timeout=120)
|
||||
if r.returncode != 0:
|
||||
failures.append(f"{query!r}: layer exited {r.returncode} ({r.stderr[:120]})")
|
||||
continue
|
||||
try:
|
||||
data = json.loads(r.stdout)
|
||||
except json.JSONDecodeError:
|
||||
failures.append(f"{query!r}: layer returned unparseable JSON")
|
||||
continue
|
||||
|
||||
results = data.get("results", [])
|
||||
if len(results) < QUALITY_TOP_N:
|
||||
failures.append(f"{query!r}: only {len(results)} results returned")
|
||||
continue
|
||||
|
||||
# load the demote list from the SAME config the layer uses
|
||||
cfg_path = os.path.join(os.path.dirname(layer), "..", "config", "search-ranking.yaml")
|
||||
demoted: set[str] = set()
|
||||
try:
|
||||
sys.path.insert(0, os.path.dirname(layer))
|
||||
import importlib.util as _iu
|
||||
spec = _iu.spec_from_file_location("_sac_cfg", layer)
|
||||
mod = _iu.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
demoted = set(mod._load_config().get("demote_domains", []) or [])
|
||||
except Exception: # noqa: BLE001
|
||||
failures.append(f"{query!r}: could not load demote_domains from config")
|
||||
|
||||
for item in results[:QUALITY_TOP_N]:
|
||||
host = (item.get("host") or "")
|
||||
for d in demoted:
|
||||
if host == d or host.endswith("." + d):
|
||||
failures.append(
|
||||
f"{query!r}: demoted host {host} in top {QUALITY_TOP_N}"
|
||||
)
|
||||
for item in results:
|
||||
host = (item.get("host") or "")
|
||||
for b in QUALITY_BANNED_HOSTS:
|
||||
if host == b or host.endswith("." + b):
|
||||
failures.append(f"{query!r}: non-answer host {host} returned")
|
||||
return failures
|
||||
|
||||
|
||||
def main() -> int:
|
||||
failures: list[str] = []
|
||||
print(f"Search stack check -- {SEARXNG_URL}")
|
||||
@@ -216,6 +289,16 @@ def main() -> int:
|
||||
print(f" FAIL: {msg}")
|
||||
failures.append(msg)
|
||||
|
||||
print("-" * 72)
|
||||
print("RANKING QUALITY (agent-consumption layer)")
|
||||
quality = check_ranking_quality()
|
||||
if quality:
|
||||
for q in quality:
|
||||
print(f" FAIL: {q}")
|
||||
failures.extend(quality)
|
||||
else:
|
||||
print(" ok: no demoted host in the top 3; no banned non-answer returned")
|
||||
|
||||
print("=" * 72)
|
||||
if failures:
|
||||
print("VERDICT: FAIL")
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/bin/bash
|
||||
# swap-gpu-dense-model.sh — Swap RTX 3090 from qwen3.6-27B-code to SmartCode-Fable-5
|
||||
# Run when download completes: ssh root@192.168.68.8 'bash -s' < this script
|
||||
# Run when download completes: ssh llmuser@192.168.68.8 'sudo bash -s' < this script
|
||||
#
|
||||
# Usage: bash swap-gpu-dense-model.sh
|
||||
# Requires: new model at /home/llmuser/models/SmartCode-Fable-5-27B-UD-Q4_K_XL.gguf
|
||||
|
||||
@@ -135,16 +135,16 @@ case "$PI_VERDICT" in
|
||||
esac
|
||||
# -- abiba-leg-end
|
||||
|
||||
# ── Platform B: Tanko (DSH dsh-web on amdpve CT 112) ──
|
||||
# ── Platform B: Tanko (DSH dsh-web on minipve CT 112) ──
|
||||
# Direct SSH to 192.168.68.122 is not a dependency of this monitor — per-worker
|
||||
# key availability varies — so probes run from the amdpve vantage via `pct exec`.
|
||||
# Tanko's Zulip gateway runs as the dsh-web systemd unit inside CT 112 on amdpve
|
||||
# (192.168.68.15). The gateway binds 127.0.0.1:3080 loopback-only by design — a
|
||||
# key availability varies — so probes run from the minipve vantage via `pct exec`.
|
||||
# Tanko's Zulip gateway runs as the dsh-web systemd unit inside CT 112 on minipve
|
||||
# (192.168.68.12). The gateway binds 127.0.0.1:3080 loopback-only by design — a
|
||||
# remote :3080 probe is refused and is NOT a fault.
|
||||
TANKO_SVC=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \
|
||||
TANKO_SVC=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.12 \
|
||||
"pct exec 112 -- systemctl is-active dsh-web" 2>/dev/null || true)
|
||||
[ -n "$TANKO_SVC" ] || TANKO_SVC="unknown"
|
||||
TANKO_HTTP=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \
|
||||
TANKO_HTTP=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.12 \
|
||||
"pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/" 2>/dev/null || true)
|
||||
[ -n "$TANKO_HTTP" ] || TANKO_HTTP="000"
|
||||
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
---
|
||||
kind: function
|
||||
name: search-agent-consumption
|
||||
description: >
|
||||
Agent-consumption layer in front of SearXNG + Firecrawl. Raw multi-engine
|
||||
aggregation returns results with no dedupe, no filtering and no reranking;
|
||||
measured 2026-09-26 that put bestbuy.com and merriam-webster.com into "best
|
||||
practices agent context management", and put four SEO blogs above the real
|
||||
Proxmox forum threads on a precise technical query. Identical queries also
|
||||
ranked DIFFERENTLY between runs, which is why the layer is deterministic
|
||||
rather than dependent on engine behaviour.
|
||||
|
||||
Pipeline: dedupe -> drop non-answers -> demote content farms / promote primary
|
||||
sources -> stable sort -> extract page text for the top N under an explicit
|
||||
character budget -> stable JSON. Policy lives in config, not code.
|
||||
|
||||
Call it when an agent needs search RESULTS rather than links: it returns usable
|
||||
page text in one call instead of a snippet plus a second fetch.
|
||||
|
||||
version: 1.0.0
|
||||
---
|
||||
|
||||
## Where the policy lives
|
||||
|
||||
`config/search-ranking.yaml` — reviewable, no code change needed to adjust:
|
||||
|
||||
| key | effect |
|
||||
| --- | --- |
|
||||
| `non_answer.hosts` / `path_patterns` / `query_keys` / `host_root` | dropped outright |
|
||||
| `demote_domains` | ranked below everything, never dropped |
|
||||
| `prefer_domains` | promoted above default rank |
|
||||
| `ranking.*` | `demote_penalty`, `prefer_bonus`, `multi_engine_bonus` |
|
||||
| `extraction.*` | `top_n`, `total_chars`, `per_item_chars`, `timeout_seconds` |
|
||||
|
||||
**Demotion, not deletion, for content farms**: a genuinely useful hit is not lost,
|
||||
it simply cannot outrank a primary source. Non-answers are dropped because they
|
||||
cannot answer a question at all.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
python3 scripts/search-agent-consume.py "query text" # JSON
|
||||
python3 scripts/search-agent-consume.py --no-extract "query" # ranking only
|
||||
python3 scripts/search-agent-consume.py --explain "query" # + drop reasons
|
||||
```
|
||||
|
||||
Exit `0` ok, `1` nothing survived filtering, `2` the layer could not run.
|
||||
|
||||
## Output shape
|
||||
|
||||
Stable JSON:
|
||||
|
||||
```json
|
||||
{
|
||||
"query": "...",
|
||||
"raw_result_count": 46,
|
||||
"returned_count": 44,
|
||||
"dropped_count": 2,
|
||||
"engines": ["bing", "brave", "duckduckgo", "yandex"],
|
||||
"results": [
|
||||
{"rank": 1, "title": "...", "url": "...", "host": "...",
|
||||
"source_type": "official|code|qa|forum|discussion|web|content-farm",
|
||||
"engines": ["bing"], "score": 100.0,
|
||||
"excerpt": "...", "extraction": "ok|truncated|skipped_budget_exhausted|empty|failed:<Type>"}
|
||||
],
|
||||
"extraction": {"extracted": 5, "chars_used": 12000, "budget": 12000,
|
||||
"failures": 0, "seconds": 5.28}
|
||||
}
|
||||
```
|
||||
|
||||
`--explain` adds `dropped: [{url, reason, position}]` so the filter is auditable
|
||||
rather than magic.
|
||||
|
||||
## Measured before/after (2026-09-26)
|
||||
|
||||
Fixed query set. Relevance judged per query, not by impression.
|
||||
|
||||
**`best practices agent context management`**
|
||||
|
||||
| | before (raw SearXNG) | after (layer) |
|
||||
| --- | --- | --- |
|
||||
| 1-2 | anthropic, stackai | anthropic, langchain |
|
||||
| 3-4 | aitechmonk, agentic-design | jetbrains, blog.jetbrains |
|
||||
| 5-6 | mindstudio, sparkco | docs.langchain, reddit |
|
||||
| 7-8 | langchain, medium | cursor, reddit |
|
||||
| verdict | 4 relevant of 10; 4 content farms; medium.com twice | top 8 all primary/discussion; no content farm in the top 8 |
|
||||
|
||||
**`proxmox thin pool metadata exhaustion recovery`**
|
||||
|
||||
| | before | after |
|
||||
| --- | --- | --- |
|
||||
| 1-4 | vormox, linuxoperatingsystem, riparazioneserver, bigiron (all SEO/thin) | forum.proxmox.com, forum.proxmox.com, gist.github, github |
|
||||
| 5-9 | forum.proxmox.com x2, voxfor, github, gist | forum.proxmox.com, serverfault, forum.proxmox.com, reddit |
|
||||
|
||||
The primary sources moved from positions 5-9 to 1-4.
|
||||
|
||||
**Rule proof** (`--explain`, and a direct check of the classifier):
|
||||
|
||||
```
|
||||
DigitalOcean docs -> KEEP (a '/products/' path rule was REMOVED after the
|
||||
before/after run caught it dropping this page)
|
||||
Best Buy -> DROP shopping_or_dictionary_host
|
||||
Merriam-Webster -> DROP shopping_or_dictionary_host
|
||||
bare homepage -> DROP navigational_host_root
|
||||
proxmox.com home -> KEEP (preferred host root: a repo/docs front door is
|
||||
legitimately the answer)
|
||||
github repo -> KEEP
|
||||
```
|
||||
|
||||
**Extraction cost (criterion 4):**
|
||||
|
||||
```
|
||||
extracted 5 items, 12000 chars used of 12000 budget, 0 failures, 5.28s
|
||||
whole run end-to-end: 6.4s wall
|
||||
```
|
||||
|
||||
## Regression guard
|
||||
|
||||
`search-stack-visibility` asserts the layer still ranks correctly: for the fixed
|
||||
query set, no `demote_domains` host may appear in the top 3, and the two known
|
||||
non-answers must not be returned. Without it this layer could silently rot back
|
||||
to raw ordering, which is exactly what happened to the endpoint colours.
|
||||
|
||||
## Reachability, and one honest gap
|
||||
|
||||
- **Hermes agents** reach it directly: it reads the same `SEARXNG_URL` and
|
||||
`FIRECRAWL_URL` they already use.
|
||||
- **pi agents (MCP search server)**: the MCP server's request/response shape is
|
||||
**not ours to change**, so this layer is **NOT** wired into it. That is a real
|
||||
gap, stated rather than claimed as coverage. Closing it would require a change
|
||||
on the MCP side, which is outside this repo.
|
||||
|
||||
## Constraints
|
||||
|
||||
Does not touch the live SearXNG or Firecrawl service paths. Third-party
|
||||
`google cse` is not a hard requirement of this layer — if it 429s, ranking still
|
||||
works from the remaining engines. No credential is added or required.
|
||||
@@ -0,0 +1,236 @@
|
||||
"""Regression test for the fallback_providers list-shape crash in audit-hermes-config.py.
|
||||
|
||||
WHY THIS FILE EXISTS: audit-hermes-config.py assumed `fallback_providers` was always a dict
|
||||
(single provider). Two live agents (koby, koonimo) carry it as a LIST of dicts (one entry per
|
||||
fallback), so the script crashed with:
|
||||
|
||||
File "audit-hermes-config.py", line 211, in audit
|
||||
fb.get("provider") == "deepseek",
|
||||
AttributeError: 'list' object has no attribute 'get'
|
||||
|
||||
Both are REAL agent configs, so this is not a malformed-input case — the script simply could not
|
||||
audit two of the four agents it exists to audit. Until fixed, the key-hygiene check had no
|
||||
coverage for half the fleet while appearing to run.
|
||||
|
||||
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert:
|
||||
1. A config whose `fallback_providers` is a LIST of valid dicts does NOT crash (exit code is 0 or 1,
|
||||
never a traceback/AttributeError).
|
||||
2. A config whose `fallback_providers` contains a MALFORMED entry (a list element that is not a
|
||||
mapping) reports a VIOLATION naming the offending entry, NOT an uncaught exception.
|
||||
3. The dict shape still works (existing tests must stay green).
|
||||
|
||||
No network, vault, or SSH access is required.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||
AUDIT = ROOT / "audit-hermes-config.py"
|
||||
|
||||
# A valid config where fallback_providers is a LIST of dicts (the real koby/koonimo shape).
|
||||
# One entry, well-formed: provider=deepseek, model=deepseek-v4-flash, api_key_env=DEEPSEEK_API_KEY.
|
||||
# This must produce a real verdict (PASS or FAIL) without crashing.
|
||||
LIST_SHAPE_VALID = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# A valid config where fallback_providers is a LIST with TWO entries (multiple fallbacks).
|
||||
# Both entries well-formed. Must not crash and should produce a real verdict.
|
||||
LIST_SHAPE_MULTI = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# A config where fallback_providers is a LIST containing a MALFORMED entry:
|
||||
# one element is a plain string, not a mapping. The checker must report a VIOLATION
|
||||
# naming the offending entry (fallback_providers[1]) and NOT crash.
|
||||
LIST_SHAPE_MALFORMED = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
- "not-a-mapping"
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# The original DICT shape (single provider) must still work — existing behaviour preserved.
|
||||
DICT_SHAPE_VALID = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
|
||||
def _run_config(tmp_path, name, text):
|
||||
cfg = tmp_path / name
|
||||
cfg.write_text(text)
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(AUDIT), str(cfg)],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
return proc.returncode, proc.stdout, proc.stderr
|
||||
|
||||
|
||||
def test_list_shape_single_entry_does_not_crash(tmp_path):
|
||||
"""A LIST with one valid dict must not raise AttributeError; exit 0 (PASS)."""
|
||||
code, out, err = _run_config(tmp_path, "list-single.yaml", LIST_SHAPE_VALID)
|
||||
# Must NOT be a crash (traceback). A clean run exits 0 (PASS) or 1 (FAIL), never 2+ (exception).
|
||||
assert code in (0, 1), f"Expected clean exit 0 or 1, got {code}\nSTDOUT:\n{out}\nSTDERR:\n{err}"
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
# The valid single-entry list should PASS (all rules satisfied).
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}"
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_list_shape_multiple_entries_does_not_crash(tmp_path):
|
||||
"""A LIST with two valid dicts must not raise AttributeError; exit 0 (PASS)."""
|
||||
code, out, err = _run_config(tmp_path, "list-multi.yaml", LIST_SHAPE_MULTI)
|
||||
assert code in (0, 1), f"Expected clean exit 0 or 1, got {code}\nSTDOUT:\n{out}\nSTDERR:\n{err}"
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}"
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_list_shape_malformed_entry_reports_violation_not_crash(tmp_path):
|
||||
"""A LIST containing a non-mapping element must be a reported VIOLATION, not a crash."""
|
||||
code, out, err = _run_config(tmp_path, "list-malformed.yaml", LIST_SHAPE_MALFORMED)
|
||||
# Must NOT be a crash.
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
# Should be a FAIL (exit 1) because the malformed entry is a violation.
|
||||
assert code == 1, f"Expected FAIL (exit 1) but got {code}\n{out}"
|
||||
assert "RESULT: FAIL" in out
|
||||
# The violation must name the offending entry (fallback_providers[1]).
|
||||
assert "fallback_providers[1]" in out, f"Violation did not name the offending entry:\n{out}"
|
||||
|
||||
|
||||
def test_dict_shape_still_passes(tmp_path):
|
||||
"""The original DICT shape (single provider) must still PASS — existing behaviour preserved."""
|
||||
code, out, err = _run_config(tmp_path, "dict-valid.yaml", DICT_SHAPE_VALID)
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}\nSTDERR:\n{err}"
|
||||
assert "RESULT: PASS" in out
|
||||
@@ -49,7 +49,7 @@ HEALTH_CONTRACT = ROOT / "zulip-health.prose.md"
|
||||
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
|
||||
|
||||
MUMUNI_IP = "192.168.68.24" # Mumuni's old (decommissioned) deployment
|
||||
TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec
|
||||
TANKO_VANTAGE = "192.168.68.12" # minipve — Tanko CT 112 via pct exec
|
||||
AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker
|
||||
|
||||
|
||||
@@ -82,7 +82,7 @@ done
|
||||
printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts"
|
||||
cmd="${*: -1}"
|
||||
case "$host" in
|
||||
192.168.68.15)
|
||||
192.168.68.12)
|
||||
case "$cmd" in
|
||||
*"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;;
|
||||
*curl*) printf '%s' "$TANKO_HTTP" ;;
|
||||
|
||||
@@ -104,6 +104,59 @@ def test_koby_ct111_is_on_storepve(ahc):
|
||||
assert ahc.AGENTS["koby"]["pve"] == "storepve"
|
||||
|
||||
|
||||
def test_tanko_ct112_is_probed_on_minipve(ahc, monkeypatch, capsys):
|
||||
# CT 112 (tanko) was live-migrated to minipve (.12) on 2026-09-27; the
|
||||
# amdpve mapping made `pct status 112` fail and read as ct-unreachable.
|
||||
# Execute the probe and assert the host the script actually contacts.
|
||||
probes = []
|
||||
monkeypatch.setattr(
|
||||
ahc, "ssh",
|
||||
lambda host, cmd, user="root": probes.append((host, cmd)) or "status: running",
|
||||
)
|
||||
ahc.FAIL.clear()
|
||||
ahc.REPORT_ONLY.clear()
|
||||
try:
|
||||
ahc.check_ct_liveness()
|
||||
tanko_hosts = [h for h, cmd in probes if cmd == "pct status 112 2>/dev/null"]
|
||||
assert tanko_hosts == ["192.168.68.12"]
|
||||
finally:
|
||||
ahc.FAIL.clear()
|
||||
ahc.REPORT_ONLY.clear()
|
||||
|
||||
|
||||
def test_gpu_rtx3090_probe_uses_llmuser_not_root(ahc, monkeypatch, capsys):
|
||||
# 2026-09-28: root SSH to .8 was lost when the guest was rebuilt; llmuser
|
||||
# owns llama-server and can read systemctl status and the :8080 pid. A root
|
||||
# probe reads as UNREACHABLE for a healthy host (the reported bug). Execute
|
||||
# check_gpu_ports() against an SSH boundary that only accepts llmuser@.8 and
|
||||
# assert the .8 leg does not produce the false UNREACHABLE failure.
|
||||
seen = []
|
||||
|
||||
def fake_ssh(host, cmd, user="root"):
|
||||
seen.append((host, user))
|
||||
if host == "192.168.68.8" and user != "llmuser":
|
||||
return None # root SSH denied -> baseline false UNREACHABLE
|
||||
if cmd.startswith("systemctl is-active"):
|
||||
return "active"
|
||||
if cmd.startswith("ss -tlnp"):
|
||||
return "48351"
|
||||
if cmd.startswith("curl"):
|
||||
return '{"status":"ok"}'
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(ahc, "ssh", fake_ssh)
|
||||
ahc.FAIL.clear()
|
||||
try:
|
||||
ahc.check_gpu_ports()
|
||||
out = capsys.readouterr().out
|
||||
assert "gpu-unreachable:192.168.68.8" not in ahc.FAIL
|
||||
assert "\u2705 gpu-rtx3090 (.8): healthy" in out
|
||||
assert ("192.168.68.8", "llmuser") in seen
|
||||
assert not any(host == "192.168.68.8" and user == "root" for host, user in seen)
|
||||
finally:
|
||||
ahc.FAIL.clear()
|
||||
|
||||
|
||||
def test_report_only_legs_never_count_as_failures(ahc):
|
||||
for agent, report_only in (("koby", True), ("koonimo", False), ("tanko", False)):
|
||||
ahc.FAIL.clear()
|
||||
|
||||
@@ -34,7 +34,7 @@ ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh"
|
||||
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
|
||||
|
||||
TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec
|
||||
TANKO_VANTAGE = "192.168.68.12" # minipve — Tanko CT 112 via pct exec
|
||||
AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker
|
||||
|
||||
|
||||
@@ -52,7 +52,7 @@ done
|
||||
printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts"
|
||||
cmd="${*: -1}"
|
||||
case "$host" in
|
||||
192.168.68.15)
|
||||
192.168.68.12)
|
||||
case "$cmd" in
|
||||
*"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;;
|
||||
*curl*) printf '%s' "$TANKO_HTTP" ;;
|
||||
|
||||
+24
-24
@@ -28,7 +28,7 @@ session start.
|
||||
## Requires
|
||||
|
||||
- **Zulip API key** for `abiba-bot@chat.sysloggh.net` in `$ZULIP_API_KEY`
|
||||
- **SSH access** to amdpve (192.168.68.15) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14)
|
||||
- **SSH access** to minipve (192.168.68.12) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14)
|
||||
- **PM2** on localhost for pi process management
|
||||
- **Network access** to `chat.sysloggh.net`, `kagentz.sysloggh.net` (C3 public path), `localhost:9200`
|
||||
- **Write access** to `/root/zulip-health-monitor.log` and `/tmp/zulip-monitor-debounce`
|
||||
@@ -228,20 +228,20 @@ grep -a "Finalized\|Failed to finalize" /root/.pm2/logs/abiba-zulip-out.log | ta
|
||||
| Crash loop >10/h | Alert user |
|
||||
|
||||
|
||||
### Step 3: Platform B — Tanko (DSH on amdpve CT 112)
|
||||
### Step 3: Platform B — Tanko (DSH on minipve CT 112)
|
||||
|
||||
Mumuni is out of scope for this host (see the note above): she runs on her own
|
||||
container and is monitored on her side.
|
||||
|
||||
Tanko runs on DSH (DeepSeek Harness) — it no longer runs a Hermes gateway, so
|
||||
there is no `~/.hermes/gateway_state.json` on CT 112. Tanko's Zulip gateway runs
|
||||
as the `dsh-web` systemd unit inside **CT 112**, which resides on the **amdpve**
|
||||
PVE host (**192.168.68.15**). Direct SSH to 192.168.68.122 is not a dependency
|
||||
as the `dsh-web` systemd unit inside **CT 112**, which resides on the **minipve**
|
||||
PVE host (**192.168.68.12**). Direct SSH to 192.168.68.122 is not a dependency
|
||||
of this contract — per-worker key availability varies — so CT 112 probes run
|
||||
from the amdpve vantage via `pct exec`:
|
||||
from the minipve vantage via `pct exec`:
|
||||
|
||||
```bash
|
||||
ssh root@192.168.68.15 "pct exec 112 -- <command>"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- <command>"
|
||||
```
|
||||
|
||||
> **By design (verified 2026-09-08):** the `dsh-web` gateway binds
|
||||
@@ -253,7 +253,7 @@ ssh root@192.168.68.15 "pct exec 112 -- <command>"
|
||||
**B1: Gateway Service State (Tanko)**
|
||||
|
||||
```bash
|
||||
ssh root@192.168.68.15 "pct exec 112 -- systemctl is-active dsh-web"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- systemctl is-active dsh-web"
|
||||
```
|
||||
|
||||
Expected: `active`. Anything else → gateway service down → apply the Tanko heal
|
||||
@@ -262,7 +262,7 @@ Expected: `active`. Anything else → gateway service down → apply the Tanko h
|
||||
**B2: Gateway HTTP Liveness (Tanko — loopback-only :3080)**
|
||||
|
||||
```bash
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/"
|
||||
```
|
||||
|
||||
Alive = **ANY** HTTP status response from the endpoint — the expected set is
|
||||
@@ -273,13 +273,13 @@ process answering `503` is running and self-heal must NOT restart-loop it.
|
||||
Down = connection refused (`000`) or timeout only. Statuses outside the
|
||||
expected set are logged/reported as a warning — reported, never healed on.
|
||||
|
||||
**B3: Public-URL Fallback Probe (Tanko — for nodes without pct/ssh access to amdpve)**
|
||||
**B3: Public-URL Fallback Probe (Tanko — for nodes without pct/ssh access to minipve)**
|
||||
|
||||
```bash
|
||||
curl -s --connect-timeout 10 --max-time 15 -o /dev/null -w '%{http_code}' https://tankodhs.sysloggh.net/
|
||||
```
|
||||
|
||||
Fallback only — used when the monitoring node has no pct/SSH path to amdpve.
|
||||
Fallback only — used when the monitoring node has no pct/SSH path to minipve.
|
||||
Alive = **ANY** HTTP status response from the endpoint — healthy signals are
|
||||
`302` (authentik proxy-auth redirect) and `401` (auth-gated), and any other
|
||||
status, including `404`/`5xx`, also counts alive: the endpoint is up and
|
||||
@@ -404,29 +404,29 @@ ExecStartPost=/bin/systemctl --no-block start dsh-web-token.service
|
||||
4. Every later request through `/` presents that cookie; the token is not needed
|
||||
again until the cookie expires or a new browser is used.
|
||||
|
||||
**Verification** (amdpve vantage):
|
||||
**Verification** (minipve vantage):
|
||||
```bash
|
||||
# 1. Login endpoint is Authentik-gated: unauthenticated -> 302 (not 200/303).
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}\n' \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}\n' \
|
||||
-H 'Host: tankodhs.sysloggh.net' http://127.0.0.1/dsh-web-login"
|
||||
# Expected: 302
|
||||
|
||||
# 2. Legacy :8081 endpoint is gone (connection refused -> 000).
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s --max-time 3 -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s --max-time 3 -o /dev/null \
|
||||
-w '%{http_code}\n' http://192.168.68.122:8081/"
|
||||
# Expected: 000
|
||||
|
||||
# 3. Backend cookie mint + reuse (exactly what /dsh-web-login proxies to).
|
||||
TOKEN=$(ssh root@192.168.68.15 "pct exec 112 -- cat /etc/dsh-web/launch-token")
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -c /tmp/dsh.jar -o /dev/null \
|
||||
TOKEN=$(ssh root@192.168.68.12 "pct exec 112 -- cat /etc/dsh-web/launch-token")
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -c /tmp/dsh.jar -o /dev/null \
|
||||
-H 'Host: tankodhs.sysloggh.net' 'http://127.0.0.1:3080/?token=$TOKEN'"
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
-w '%{http_code}\n' -H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/"
|
||||
# Expected: 200 — the minted dsh-auth-... cookie (authority
|
||||
# tankodhs.sysloggh.net) is replayed on the next request and accepted.
|
||||
|
||||
# 4. Token refresh is non-disruptive and idempotent.
|
||||
ssh root@192.168.68.15 "pct exec 112 -- /opt/deepseek-harness/capture-dsh-token.sh"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- /opt/deepseek-harness/capture-dsh-token.sh"
|
||||
# Expected: "token unchanged; nginx not reloaded" when nothing changed
|
||||
```
|
||||
|
||||
@@ -437,32 +437,32 @@ fresh cookie. Both verified live 2026-09-11.
|
||||
|
||||
```bash
|
||||
# 5. Cookie survives a dsh-web restart, and the new token mints a new cookie.
|
||||
ssh root@192.168.68.15 "pct exec 112 -- systemctl restart dsh-web"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- systemctl restart dsh-web"
|
||||
# dsh-web is Type=simple: restart returns before :3080 is listening. Bounded-poll
|
||||
# until the socket answers (any status but 000) before asserting the cookie.
|
||||
for i in $(seq 1 60); do
|
||||
UP=$(ssh root@192.168.68.15 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
UP=$(ssh root@192.168.68.12 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
-H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/")
|
||||
[ "$UP" != "000" ] && break
|
||||
sleep 2
|
||||
done
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
-w '%{http_code}\n' -H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/"
|
||||
# Expected: 200 — the pre-restart cookie is still accepted.
|
||||
# The restart's ExecStartPost (or the 2-minute timer) refreshes the include. A
|
||||
# manual run may no-op on the flock, so poll until the include carries a token
|
||||
# the running process accepts (bounded wait) before the mint+reuse check.
|
||||
for i in $(seq 1 60); do
|
||||
TOKEN=$(ssh root@192.168.68.15 "pct exec 112 -- sed -n 's/.*token=//p' /etc/dsh-web/nginx-login.conf | tr -d ';\n'")
|
||||
CODE=$(ssh root@192.168.68.15 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
TOKEN=$(ssh root@192.168.68.12 "pct exec 112 -- sed -n 's/.*token=//p' /etc/dsh-web/nginx-login.conf | tr -d ';\n'")
|
||||
CODE=$(ssh root@192.168.68.12 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
-H 'Host: tankodhs.sysloggh.net' 'http://127.0.0.1:3080/?token=$TOKEN'")
|
||||
[ "$CODE" = "303" ] && break
|
||||
sleep 2
|
||||
done
|
||||
# Expected: 303 — the include now holds the token the running process accepts.
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -c /tmp/dsh-new.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -c /tmp/dsh-new.jar -o /dev/null \
|
||||
-H 'Host: tankodhs.sysloggh.net' 'http://127.0.0.1:3080/?token=$TOKEN'"
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null \
|
||||
-w '%{http_code}\n' -H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/"
|
||||
# Expected: 200 — the refreshed token minted a fresh cookie.
|
||||
```
|
||||
|
||||
Reference in New Issue
Block a user