Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
245a4ffbea | ||
|
|
7c0adefdeb | ||
|
|
88b8cb96a5 | ||
|
|
2623e05752 | ||
|
|
42a1d0bd91 | ||
|
|
e052a069cb | ||
|
|
2ace79fcab | ||
|
|
87b4d67067 | ||
|
|
7d1db62a8e | ||
|
|
9943be5e68 | ||
|
|
fa26b7a579 | ||
|
|
cd479caeec | ||
|
|
1b1de8b0fc |
@@ -0,0 +1 @@
|
||||
__pycache__/
|
||||
@@ -0,0 +1,230 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Hermes Config Audit — validates a live config.yaml against the prose contract rules.
|
||||
|
||||
Usage:
|
||||
python3 audit-hermes-config.py <config.yaml>
|
||||
python3 audit-hermes-config.py /root/.hermes/config.yaml
|
||||
|
||||
Exit codes:
|
||||
0 = all checks pass
|
||||
1 = one or more contract violations found
|
||||
|
||||
This script encodes every rule from hermes-config-template.prose.md so config
|
||||
changes can be verified before and after application. It is the single automated
|
||||
enforcement layer for the prose contract.
|
||||
|
||||
Contract: /root/prose-contracts/hermes-config-template.prose.md
|
||||
"""
|
||||
|
||||
import sys
|
||||
import yaml
|
||||
|
||||
VIOLATIONS = []
|
||||
WARNINGS = []
|
||||
PASSES = []
|
||||
|
||||
|
||||
def check(condition, rule, message):
|
||||
if condition:
|
||||
PASSES.append(f"[{rule}] {message}")
|
||||
else:
|
||||
VIOLATIONS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def warn(rule, message):
|
||||
WARNINGS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def audit(path):
|
||||
with open(path) as f:
|
||||
cfg = yaml.safe_load(f)
|
||||
|
||||
model = cfg.get("model", {})
|
||||
fb = cfg.get("fallback_providers", {})
|
||||
comp = cfg.get("compression", {})
|
||||
aux = cfg.get("auxiliary", {})
|
||||
deleg = cfg.get("delegation", {})
|
||||
cps = cfg.get("custom_providers", [])
|
||||
cp = cps[0] if cps else {}
|
||||
|
||||
# --- Rule 3: API Keys via Environment ---
|
||||
check(
|
||||
model.get("api_key") in ("", None),
|
||||
"Rule 3",
|
||||
f"model.api_key must be empty (got {model.get('api_key')!r}) — keys via env var, not hardcoded",
|
||||
)
|
||||
check(
|
||||
model.get("api_key_env") == "LITELLM_API_KEY",
|
||||
"Rule 3",
|
||||
f"model.api_key_env must be LITELLM_API_KEY (got {model.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- Rule 5: Main Config Base URL ---
|
||||
expected_base = "http://192.168.68.116/v1"
|
||||
check(
|
||||
model.get("base_url") == expected_base,
|
||||
"Rule 5",
|
||||
f"model.base_url must be {expected_base} (got {model.get('base_url')!r}) — /v1 not /litellm/v1",
|
||||
)
|
||||
|
||||
# --- Rule 6: max_tokens Is Required ---
|
||||
check(
|
||||
isinstance(model.get("max_tokens"), int) and model.get("max_tokens") <= 8192,
|
||||
"Rule 6",
|
||||
f"model.max_tokens must be set and <= 8192 (got {model.get('max_tokens')!r}) — thermal safety",
|
||||
)
|
||||
|
||||
# --- Rule 7: Auxiliary Model Consistency ---
|
||||
check(
|
||||
comp.get("model") == "syslog-auto",
|
||||
"Rule 7",
|
||||
f"compression.model must be syslog-auto (got {comp.get('model')!r}) — auto-routing to prevent Strix Halo overload",
|
||||
)
|
||||
aux_comp = aux.get("compression", {})
|
||||
check(
|
||||
aux_comp.get("model") == "syslog-auto",
|
||||
"Rule 7",
|
||||
f"auxiliary.compression.model must be syslog-auto (got {aux_comp.get('model')!r}) — must match compression.model",
|
||||
)
|
||||
|
||||
# --- Rule 8: GPU Workload Distribution ---
|
||||
check(
|
||||
aux.get("vision", {}).get("model") == "gpu-light",
|
||||
"Rule 8",
|
||||
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
check(
|
||||
aux.get("web_extract", {}).get("model") == "gpu-light",
|
||||
"Rule 8",
|
||||
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
|
||||
# --- Rule 9: Compression Threshold ---
|
||||
check(
|
||||
comp.get("threshold") == 0.65,
|
||||
"Rule 9",
|
||||
f"compression.threshold must be 0.65 for 128K models (got {comp.get('threshold')!r})",
|
||||
)
|
||||
check(
|
||||
comp.get("max_context_window") == 131072,
|
||||
"Rule 9",
|
||||
f"compression.max_context_window must be 131072 (got {comp.get('max_context_window')!r}) — matches 128K GPU capacity",
|
||||
)
|
||||
|
||||
# --- Rule 10: Default Model Must Be syslog-auto ---
|
||||
check(
|
||||
model.get("default") == "syslog-auto",
|
||||
"Rule 10",
|
||||
f"model.default must be syslog-auto (got {model.get('default')!r}) — auto-routing default",
|
||||
)
|
||||
|
||||
# --- Rule 14: Provider Name Must Match custom_providers Name ---
|
||||
check(
|
||||
model.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"model.provider must be 'harness' (got {model.get('provider')!r}) — NOT 'custom'. "
|
||||
f"provider: custom causes generic resolution path that ignores key_env → 'no-key-required' → 401",
|
||||
)
|
||||
check(
|
||||
comp.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"compression.provider must be 'harness' (got {comp.get('provider')!r})",
|
||||
)
|
||||
for aux_name in ("vision", "web_extract", "compression"):
|
||||
aux_provider = aux.get(aux_name, {}).get("provider")
|
||||
check(
|
||||
aux_provider == "harness",
|
||||
"Rule 14",
|
||||
f"auxiliary.{aux_name}.provider must be 'harness' (got {aux_provider!r})",
|
||||
)
|
||||
check(
|
||||
deleg.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
fb.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- custom_providers sanity ---
|
||||
check(
|
||||
cp.get("name") == "harness",
|
||||
"custom_providers",
|
||||
f"custom_providers[0].name must be 'harness' (got {cp.get('name')!r})",
|
||||
)
|
||||
check(
|
||||
cp.get("key_env") == "LITELLM_API_KEY" or cp.get("api_key_env") == "LITELLM_API_KEY",
|
||||
"custom_providers",
|
||||
f"custom_providers[0] must have key_env or api_key_env = LITELLM_API_KEY "
|
||||
f"(got key_env={cp.get('key_env')!r}, api_key_env={cp.get('api_key_env')!r})",
|
||||
)
|
||||
check(
|
||||
cp.get("base_url", "").endswith("/v1"),
|
||||
"custom_providers",
|
||||
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
||||
)
|
||||
|
||||
# --- No raw model names (Rule 7/8 spirit) ---
|
||||
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
|
||||
for section_path, section_dict in [
|
||||
("model", model), ("compression", comp),
|
||||
("auxiliary.vision", aux.get("vision", {})),
|
||||
("auxiliary.web_extract", aux.get("web_extract", {})),
|
||||
("auxiliary.compression", aux.get("compression", {})),
|
||||
("delegation", deleg),
|
||||
]:
|
||||
m = section_dict.get("model", "")
|
||||
if m in raw_names:
|
||||
warn(
|
||||
"Rule 7/8",
|
||||
f"{section_path}.model = {m!r} — raw model name, use stable alias instead "
|
||||
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
|
||||
)
|
||||
|
||||
# --- Report ---
|
||||
print(f"{'=' * 60}")
|
||||
print(f"Hermes Config Audit: {path}")
|
||||
print(f"{'=' * 60}")
|
||||
print(f"\n✅ PASSED ({len(PASSES)}):")
|
||||
for p in PASSES:
|
||||
print(f" ✅ {p}")
|
||||
|
||||
if WARNINGS:
|
||||
print(f"\n⚠️ WARNINGS ({len(WARNINGS)}):")
|
||||
for w in WARNINGS:
|
||||
print(f" ⚠️ {w}")
|
||||
|
||||
if VIOLATIONS:
|
||||
print(f"\n❌ VIOLATIONS ({len(VIOLATIONS)}):")
|
||||
for v in VIOLATIONS:
|
||||
print(f" ❌ {v}")
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"RESULT: FAIL — {len(VIOLATIONS)} violation(s) must be fixed")
|
||||
print(f"{'=' * 60}")
|
||||
return 1
|
||||
else:
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"RESULT: PASS — all contract rules satisfied")
|
||||
print(f"{'=' * 60}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python3 audit-hermes-config.py <config.yaml>")
|
||||
sys.exit(2)
|
||||
sys.exit(audit(sys.argv[1]))
|
||||
@@ -35,8 +35,7 @@ Docker hosts get special attention:
|
||||
| All other CTs | — | LOW — no Docker | apt clean, log rotate |
|
||||
| GPU bare metal (.8, .110) | — | LOW — no Docker on GPU hosts | log rotate |
|
||||
|
||||
> **Decommissioned:** CT 118 (jitsi) — intentionally stopped, not scanned.
|
||||
> **Migrated:** CT 101 (llm-gpu) → bare metal .8, CT 103 (ocu-llm) → bare metal .110.
|
||||
> **Note:** CT 118 is now jdownloader (active on storepve). CT 101 (llm-gpu) → bare metal .8, CT 103 (ocu-llm) → bare metal .110.
|
||||
|
||||
## Threat Levels
|
||||
|
||||
@@ -275,10 +274,10 @@ one-off GPU builds. No automated post-migration cleanup was in place.
|
||||
### CT Access (via pct-run)
|
||||
| CT | Name | Node | Status |
|
||||
|----|------|------|--------|
|
||||
| 100 | abiba | amdpve | local |
|
||||
| 102 | adguard | acerpve | ✅ reachable |
|
||||
| 100 | abiba | hwepve | local |
|
||||
| 102 | adguard | minipve | ✅ reachable |
|
||||
| 104 | authentik | minipve | ✅ reachable |
|
||||
| 105 | kagentz | amdpve | ✅ reachable |
|
||||
| 105 | kagentz | hwepve | ✅ reachable |
|
||||
| 106 | ra-h-os | storepve | ✅ reachable |
|
||||
| 107 | pbs | storepve | ✅ reachable |
|
||||
| 108 | media | storepve | ✅ reachable |
|
||||
@@ -286,7 +285,7 @@ one-off GPU builds. No automated post-migration cleanup was in place.
|
||||
| 111 | tdunna | amdpve | ✅ reachable |
|
||||
| 112 | tanko | amdpve | ✅ reachable |
|
||||
| 113 | baggy | amdpve | ✅ reachable |
|
||||
| 114 | mumuni | minipve | ✅ reachable |
|
||||
| 114 | mumuni | hwepve | ✅ reachable |
|
||||
| 115 | scottdenya | amdpve | ✅ reachable |
|
||||
| 116 | syslog-api | minipve | ✅ reachable |
|
||||
| 117 | zulip | storepve | ✅ reachable |
|
||||
@@ -303,4 +302,4 @@ one-off GPU builds. No automated post-migration cleanup was in place.
|
||||
|------|-----|------|--------|
|
||||
| docker-vm | 192.168.68.7 | 16 Docker containers, 4 stacks | ✅ reachable |
|
||||
|
||||
> **Decommissioned:** CT 118 (jitsi) — intentionally stopped.\n> **Migrated:** CT 101 → .8, CT 103 → .110 (bare metal GPU).\n> **KVM VM:** CT 109 (docker-vm) is a KVM VM, not LXC — access via SSH .7.
|
||||
> **Note:** CT 118 is now jdownloader (active on storepve). CT 119 (infisical-vault) added on minipve.\n> **Migrated:** CT 101 → .8, CT 103 → .110 (bare metal GPU).\n> **KVM VM:** CT 109 (docker-vm) is a KVM VM, not LXC — access via SSH .7.
|
||||
@@ -185,7 +185,7 @@ what, and why should I care?
|
||||
|
||||
```
|
||||
❌ "Monitors infrastructure health"
|
||||
✅ "Scans all 5 Proxmox nodes and 19 CTs for disk pressure, checks Docker
|
||||
✅ "Scans all 6 Proxmox nodes and 19 CTs for disk pressure, checks Docker
|
||||
container health on .7/.116/.17, alerts via Telegram DM on RED/CRITICAL"
|
||||
```
|
||||
|
||||
|
||||
@@ -25,12 +25,12 @@ done
|
||||
| Agent | CT | Node | IP | LiteLLM Alias | Key Source | Platform |
|
||||
|-------|-----|------|-----|---------------|------------|----------|
|
||||
| Tanko | 112 | amdpve | .122 | `tanko` | Infisical vault | Hermes |
|
||||
| Mumuni | 100 (abiba) | hwepve | .24 | `mumuni` | Infisical vault | Hermes (Pi gateway) |
|
||||
| Koby | 129 | amdpve | srv1079750 | `koby` | Infisical vault | **Hermes** |
|
||||
| Koonimo | 114 | amdpve | ? | `koonimo` | Infisical vault | Hermes |
|
||||
| Mumuni | 114 | hwepve | .123 | `mumuni` | Infisical vault | Hermes |
|
||||
| Koby | 111 | amdpve | srv1079750 | `koby` | Infisical vault | **Hermes** |
|
||||
| Koonimo | 113 | amdpve | .114 | `koonimo` | Infisical vault | Hermes |
|
||||
| Shumba | — | 192.168.68.119 | N/A | N/A (DeepSeek) | Hermes (RETIRED — CT119 now Infisical vault) |
|
||||
|
||||
> **Note**: CT hostnames (tdunna→CT129, baggy→CT114) differ from agent identities (koby, koonimo).
|
||||
> **Note**: CT hostnames (tdunna→CT111, baggy→CT113) differ from agent identities (koby, koonimo).
|
||||
|
||||
Access: `pct-run <CT_ID> <command>` — no IPs needed. GPU hosts (.8, .110, .15) use SSH.
|
||||
Keys are stored in Infisical vault (project=agents, env=production) and injected at
|
||||
@@ -169,9 +169,9 @@ pct-run <CT> grep -A8 "vision:" /root/.hermes/config.yaml | grep api_key
|
||||
# Must show both api_key: sk-... and api_key_env: LITELLM_API_KEY
|
||||
```
|
||||
|
||||
### For Koby (CT 129 / tdunna)
|
||||
### For Koby (CT 111 / tdunna)
|
||||
|
||||
Koby runs Hermes on CT 129 (tdunna). Config files at `/root/.hermes/config.yaml`.
|
||||
Koby runs Hermes on CT 111 (tdunna). Config files at `/root/.hermes/config.yaml`.
|
||||
Same Hermes pattern as Tanko/Mumuni/Koonimo — see config sections above.
|
||||
|
||||
**LiteLLM key**: alias `koby` in LiteLLM DB, injected via `infisical run --` wrapper.
|
||||
@@ -253,7 +253,7 @@ EnvironmentFile=/etc/environment # sources LITELLM_API_KEY
|
||||
|
||||
Run the consolidated health check:
|
||||
```bash
|
||||
python3 /root/scripts/agent-health-check.py # v2: now checks all 5 agents including Koby/Koonimo SSH, CT liveness, config YAML, wrapper integrity, vault non-emptiness
|
||||
python3 /root/scripts/agent-health-check.py
|
||||
```
|
||||
This validates all 4 LiteLLM keys, detects GPU port conflicts (ghost processes),
|
||||
verifies gateway liveness, confirms Zulip streaming (`edit_message` present),
|
||||
|
||||
@@ -5,14 +5,11 @@ description: >
|
||||
Standard Hermes configuration template for Syslog Solution LLC agents.
|
||||
Enforces shared infrastructure setup (Firecrawl, SearXNG, local models,
|
||||
RA-H OS MCP) while keeping agent-specific API keys and model choices.
|
||||
UPDATED 2026-07-18: Compression model switched to `syslog-auto` (was `strix-moe`)
|
||||
to relieve Strix Halo pressure. syslog-auto distributes compression across the
|
||||
weighted pool (55% RTX 3090, 30% Strix Halo, 15% RTX 5070).
|
||||
UPDATED 2026-07-16: Compression model was the stable alias `strix-moe` (NOT `ornith-1.0-35b`,
|
||||
UPDATED 2026-07-16: Compression model is the stable alias `strix-moe` (NOT `ornith-1.0-35b`,
|
||||
which LiteLLM does not serve). All 3 GPUs verified at 128K (reduced from 256K 2026-07-17 for stability).
|
||||
Added Rule 12 (Context-Issue Diagnostic) + Rule 13 (.env fallback enforcement) from the
|
||||
2026-07-16 Mumuni root-cause investigation (WAL #1300).
|
||||
UPDATED 2026-07-12: GPU workload redistributed. Compression → Strix Halo (later switched to syslog-auto 2026-07-18). RTX 3090 context verified at 128K. Infisical .env fallback required (Rule 3/13).
|
||||
UPDATED 2026-07-12: GPU workload redistributed. Compression → Strix Halo. RTX 3090 context verified at 128K. Infisical .env fallback required (Rule 3/13).
|
||||
---
|
||||
|
||||
## Maintains
|
||||
@@ -35,7 +32,7 @@ Sub-agent profiles inherit auth from the main config — no separate keys needed
|
||||
| Agent | Key Alias | Host | SSH | Sub-Agents |
|
||||
|-------|-----------|------|-----|-----------|
|
||||
| Tanko | `tanko-*` | 192.168.68.122 | jerome@.122 | — |
|
||||
| Mumuni | `mumuni` | 192.168.68.24 (CT100 abiba) | root@.24 | 6 profiles ✱ |
|
||||
| Mumuni | `mumuni` | 192.168.68.123 | root@.123 | 6 profiles ✱ |
|
||||
| Abiba | `abiba-pi` | 192.168.68.24 | local | — |
|
||||
| Koby | `koby` | CT 111 (tdunna) | Zulip | — |
|
||||
| Koonimo | `koonimo` | CT 114 (baggy) | SSH root | — |
|
||||
@@ -133,9 +130,7 @@ mcp_servers:
|
||||
# ─── Compression ───
|
||||
compression:
|
||||
enabled: true
|
||||
model: syslog-auto # ⚠️ Switched from strix-moe 2026-07-18 to relieve Strix Halo.
|
||||
# syslog-auto distributes across weighted pool (55% RTX 3090,
|
||||
# 30% Strix Halo, 15% RTX 5070). All GPUs at 128K.
|
||||
model: syslog-auto # ⚠️ Must match auxiliary.compression.model. Stable alias (gpu-fleet § Stable Role-Based Aliases). NOT ornith-1.0-35b (LiteLLM does not serve that name).
|
||||
provider: harness
|
||||
max_context_window: 131072 # MUST match actual GPU capacity. All 3 GPUs are 128K (Jul 17).
|
||||
threshold: 0.65 # Fires at ~170K for 262K window, ~85K for 128K
|
||||
@@ -150,9 +145,8 @@ compression:
|
||||
# model: gpu-light # stable alias (NOT raw "gemma-4-12b")
|
||||
# base_url: http://192.168.68.116/v1
|
||||
# api_key_env: LITELLM_API_KEY
|
||||
# Compression uses syslog-auto (switched from strix-moe 2026-07-18) to distribute
|
||||
# load across the weighted pool and relieve Strix Halo pressure.
|
||||
# Vision and web_extract use gpu-light = RTX 5070 (12B).
|
||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||
# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4)
|
||||
# in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
@@ -172,7 +166,7 @@ auxiliary:
|
||||
timeout: 30
|
||||
compression:
|
||||
provider: harness
|
||||
model: syslog-auto # Switched from strix-moe 2026-07-18. Relieves Strix Halo pressure.
|
||||
model: syslog-auto # MUST match compression.model above. Stable alias for Strix Halo (weighted pool).
|
||||
base_url: http://192.168.68.116/v1 # Rule 5: /v1 NOT /litellm/v1
|
||||
api_key_env: LITELLM_API_KEY
|
||||
timeout: 300 # gpu-fleet: 300s for large-history summarization (was 60)
|
||||
@@ -253,30 +247,34 @@ The following MUST be identical across ALL profiles:
|
||||
- Apply to BOTH main config AND all sub-agent profiles
|
||||
- For agents needing longer outputs: raise to 8192, but never omit
|
||||
|
||||
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-18)
|
||||
- Vision and web_extract use `gpu-light` (stable alias, RTX 5070 — 12GB, vision-optimized)
|
||||
- Compression now uses `syslog-auto` (switched from `strix-moe` 2026-07-18) to distribute
|
||||
compression load across the weighted pool (55% RTX 3090, 30% Strix Halo, 15% RTX 5070).
|
||||
This relieves Strix Halo pressure while keeping compression functional on all GPUs.
|
||||
- **`syslog-auto` is the valid compression model** — LiteLLM serves it as the weighted pool.
|
||||
Old configs with `strix-moe` for compression should be updated to `syslog-auto`.
|
||||
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
|
||||
- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized)
|
||||
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized)
|
||||
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b`
|
||||
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
||||
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
|
||||
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
|
||||
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
|
||||
fall back to other GPUs if Strix gets hot. Both `compression.model` and `auxiliary.compression.model`
|
||||
MUST be `syslog-auto`.
|
||||
- All auxiliary services MUST use identical routing:
|
||||
- `base_url: http://192.168.68.116/v1` (Rule 5: `/v1`, NOT `/litellm/v1`)
|
||||
- `api_key_env: LITELLM_API_KEY`
|
||||
- **Compression via syslog-auto**: Routes through the weighted pool. Strix Halo still handles
|
||||
~30% of compression calls (at 60 RPM via pool vs 40 RPM direct), but the bulk (55%)
|
||||
goes to RTX 3090 which has ample spare capacity.
|
||||
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably
|
||||
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
|
||||
(64GB UMA, 128K context) — the designated compression GPU. This frees the
|
||||
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
|
||||
- The `compression:` block's `model` MUST match `auxiliary: compression: model`
|
||||
- The `compression: max_context_window: 131072` MUST match actual GPU capacity (128K)
|
||||
|
||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-18)
|
||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations — also handles ~55% of compression via syslog-auto pool
|
||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract — handles ~15% of compression via syslog-auto pool
|
||||
- **Strix Halo (64GB, 128K ctx, qwen3.6-35B-udq4)**: Agent reasoning, compression (~30% via syslog-auto pool), fallback for other GPUs
|
||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
|
||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
|
||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
||||
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
|
||||
- Agent profiles MUST route auxiliary tasks to the correct GPU:
|
||||
- `auxiliary.vision.model: gpu-light` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gpu-light` (RTX 5070)
|
||||
- `auxiliary.compression.model: syslog-auto` (distributed pool, switched from strix-moe 2026-07-18)
|
||||
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
|
||||
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
|
||||
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
|
||||
- Do NOT use `threshold: 0.25` — this fires at 65K, causing premature context loss
|
||||
@@ -378,6 +376,26 @@ curl -s -o /dev/null -w '%{http_code}' -H "Authorization: Bearer $K" http://192.
|
||||
- `/etc/environment` is NO LONGER the canonical key source (stale values there caused 401s).
|
||||
- Do NOT leave a hardcoded stale key in `/etc/environment` — it shadows the drop-in/wrapper.
|
||||
|
||||
### Rule 14: Provider Name Must Match custom_providers Name (ADDED 2026-07-19, WAL #1471)
|
||||
|
||||
- `model.provider` MUST be `harness` (the `custom_providers[0].name`), NOT the literal string `custom`
|
||||
- When `provider: custom`, Hermes' `_get_named_custom_provider("custom")` returns None (no provider is
|
||||
named "custom" — it is named "harness"), causing a fall-through to the generic resolution path
|
||||
(`source: env/config`) at `runtime_provider.py:1156`
|
||||
- The generic path builds `api_key_candidates` from `model.api_key` (empty), host-gated
|
||||
OLLAMA/OPENAI/OPENROUTER keys, and `_host_derived_api_key` (returns "" for IP addresses)
|
||||
- **The generic path does NOT resolve `model.api_key_env` or `custom_providers.key_env`** —
|
||||
`LITELLM_API_KEY` is never read, producing `api_key = "no-key-required"` → HTTP 401
|
||||
- The named custom provider path (`source: custom_provider:harness`) DOES read `key_env` —
|
||||
but only triggers when `provider` matches the `custom_providers[0].name`
|
||||
- All sections MUST use `provider: harness`: `model`, `compression`, `auxiliary.vision`,
|
||||
`auxiliary.web_extract`, `auxiliary.compression`, `delegation`
|
||||
- Only `fallback_providers` uses a different provider (`deepseek`) for true fallback diversity
|
||||
- **Diagnostic**: If you see `source: env/config` in a request dump or log, the provider name
|
||||
is wrong. It should be `source: custom_provider:harness`.
|
||||
- **Audit script**: Run `python3 /root/prose-contracts/audit-hermes-config.py <config.yaml>`
|
||||
before and after any config change to catch this and all other rule violations.
|
||||
|
||||
## Execution
|
||||
|
||||
1. **Check current config** — Read the target agent's config.yaml
|
||||
|
||||
@@ -13,10 +13,10 @@ description: >
|
||||
against the live system. Policy fields are authoritative. See the
|
||||
`verify-before-mutate` skill.
|
||||
|
||||
**Last verified:** 2026-07-27 — gpu-dense swapped to SmartCode-Fable-5
|
||||
(Qwen3.6-27B distilled, ~50% fewer thinking tokens, improved coding reasoning).
|
||||
Model pricing reduced 100x across all models ($0.15/$0.60 per 1M tokens).
|
||||
All LiteLLM models set to 128K max_model_tokens. See data/learnings.md.
|
||||
**Last verified:** 2026-07-24 — corrected Gitea IP (.17 not .110),
|
||||
AdGuard IP (.10 not .102), AdGuard placement (minipve not acerpve),
|
||||
Abiba placement (hwepve not amdpve), added hwepve as 6th node,
|
||||
added dns.sysloggh.net route.
|
||||
---
|
||||
|
||||
# Infrastructure Control Pattern
|
||||
@@ -72,7 +72,7 @@ description: >
|
||||
| docker-vm (.7) | SSH root | SSH key | ✅ |
|
||||
| CT 116 (syslog-api) | SSH root | SSH key | ✅ |
|
||||
| Tanko CT (.122) | SSH jerome | id_ed25519 | ✅ |
|
||||
| Mumuni (inside CT100 abiba .24) | SSH root | id_ed25519 | ✅ |
|
||||
| Mumuni CT (.123) | SSH root | id_ed25519 | ✅ |
|
||||
| Baggy CT (113) | SSH jerome | ❌ no key access |
|
||||
| Netbird (.17) | SSH root | SSH key | ✅ |
|
||||
| Gitea | API token | Infisical vault (`GITEA_BOT_TOKEN`) | ✅ |
|
||||
@@ -90,11 +90,11 @@ description: >
|
||||
|
||||
### Reachability Matrix
|
||||
|
||||
| From / To | PVE API | docker-vm (.7) | CT 116 | Tanko (.122) | Mumuni (.24) | Baggy (.114) |
|
||||
| From / To | PVE API | docker-vm (.7) | CT 116 | Tanko (.122) | Mumuni (.123) | Baggy (.114) |
|
||||
|-----------|---------|----------------|--------|-------------|---------------|----------------|
|
||||
| **Abiba** (.24) | ✅ :443 | ✅ SSH | ✅ SSH | ✅ SSH jerome | ✅ SSH root | ❌ SSH |
|
||||
| **Tanko** (.122) | ❌ | ❌ | ❌ via NetBird | ✅ | ❌ | ❌ |
|
||||
| **Mumuni** (.24) | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ |
|
||||
| **Mumuni** (.123) | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ |
|
||||
|
||||
**Conclusion:** Only Abiba has cross-infrastructure access. All monitoring contracts run from Abiba.
|
||||
|
||||
@@ -109,12 +109,12 @@ description: >
|
||||
| storepve | .6 | 28C | 31GB | docker-vm, ra-h-os, PBS, media, jdownloader, zulip | Docker, storage, chat |
|
||||
| acerpve | .9 | 28C | 31GB | llm-gpu | GPU VMs |
|
||||
| ocupve | .5 | 12C | 14GB | ocu-llm | GPU VMs |
|
||||
| hwepve | .4 | ? | ? | abiba, (mumuni CT 114 stopped) | Agents (new node) |
|
||||
| hwepve | .4 | 12C | 15GB | abiba, kagentz, (mumuni CT 114 stopped) | Agents (new node) |
|
||||
|
||||
> **Note:** CTs on storepve include jdownloader (CT 118). AdGuard (CT 102) is on
|
||||
> minipve at .10, not acerpve. Abiba (CT 100) is on hwepve and now runs Mumuni
|
||||
> Zulip gateway internally. CT 114 (old mumuni container) destroyed 2026-07-26.
|
||||
> Mumuni now runs inside Abiba CT100 (.24). Old CT114 destroyed 2026-07-26.
|
||||
> minipve at .10, not acerpve. Abiba (CT 100) is on hwepve, not amdpve. Mumuni
|
||||
> (CT 114) is on hwepve (currently stopped), not minipve. Mumuni also has a
|
||||
> second instance on minipve at .123 — distinguish by CT ID, not hostname.
|
||||
|
||||
### Checks (every 5 min)
|
||||
|
||||
@@ -365,7 +365,7 @@ fine. Services that resolve directly to a LAN IP are NetBird-independent.
|
||||
|---------|--------|-------------|-------------|-------------|--------|
|
||||
| Proxmox API | minipve.sysloggh.net:8006 | 192.168.68.12 | LAN IP | No | ✅ |
|
||||
| LiteLLM | litellm.sysloggh.net | 192.168.68.116 | LAN IP | No | ✅ |
|
||||
| Authentik | auth.sysloggh.net:443 | 192.168.68.11 | CNAME → netbird | **Yes** | ⚠️ |
|
||||
| Authentik | auth.sysloggh.net:443 | 192.168.68.11:9000 | CNAME → netbird | **Yes** | ⚠️ |
|
||||
| Gitea | git.sysloggh.net:443 | 192.168.68.17:3000 | CNAME → netbird | **Yes** | ⚠️ |
|
||||
| Zulip | chat.sysloggh.net:443 | 192.168.68.19 | CNAME → netbird | **Yes** | ⚠️ VERIFY-BEFORE-USE |
|
||||
| Pulse | pulse.sysloggh.net:443 | 192.168.68.7 | CNAME → netbird | **Yes** | ⚠️ |
|
||||
@@ -586,8 +586,6 @@ ssh root@192.168.68.110 "systemctl restart llama-server"
|
||||
|
||||
## Appendix B: CT Inventory
|
||||
|
||||
| CT | Name | Node | IP | Role | Agent |
|
||||
|----|------|------|----|------|-------|
|
||||
| CT | Name | Node | IP | Role | Agent |
|
||||
|----|------|------|----|------|-------|
|
||||
| 100 | abiba | **hwepve** | .24 | Pi agent | ✅ pi |
|
||||
@@ -604,6 +602,7 @@ ssh root@192.168.68.110 "systemctl restart llama-server"
|
||||
| 111 | tdunna | amdpve | .129 | Hermes agent | ✅ |
|
||||
| 112 | tanko | amdpve | .122 | Hermes agent | ✅ |
|
||||
| 113 | baggy | amdpve | .114 | Hermes agent | ✅ |
|
||||
| 114 | mumuni | **hwepve** | .123 | Hermes agent (stopped) | ✅ |
|
||||
| 115 | scottdenya | amdpve | .75 | Denya OneCare | ❌ |
|
||||
| 116 | syslog-api | minipve | .116 | LiteLLM + Grafana | ❌ |
|
||||
| 117 | zulip | storepve | .19 | Chat | ❌ |
|
||||
@@ -629,14 +628,14 @@ Source of truth: `/root/scripts/pct-run.sh` or `prose-contracts/scripts/pct-run.
|
||||
| CT | Name | Node | pct-run |
|
||||
|-----|------|------|---------|
|
||||
| 100 | abiba | hwepve | `pct-run 100` |
|
||||
| 105 | kagentz | **hwepve** | `pct-run 105` |
|
||||
| 105 | kagentz | hwepve | `pct-run 105` |
|
||||
| 111 | tdunna | amdpve | `pct-run 111` |
|
||||
| 112 | tanko | amdpve | `pct-run 112` |
|
||||
| 113 | baggy | amdpve | `pct-run 113` |
|
||||
| 115 | scottdenya | amdpve | `pct-run 115` |
|
||||
| 104 | authentik | minipve | `pct-run 104` |
|
||||
| 110 | gitea | minipve | `pct-run 110` |
|
||||
|
||||
| 114 | mumuni | hwepve | `pct-run 114` |
|
||||
| 116 | syslog-api | minipve | `pct-run 116` |
|
||||
| 106 | ra-h-os | storepve | `pct-run 106` |
|
||||
| 107 | proxmox-backup | storepve | `pct-run 107` |
|
||||
@@ -649,35 +648,27 @@ GPU bare-metal hosts (.8 acerpve, .110 ocupve, .15 amdpve) are NOT CTs — use S
|
||||
ssh root@192.168.68.8 # RTX 3090
|
||||
ssh root@192.168.68.110 # RTX 5070
|
||||
ssh root@192.168.68.15 # Strix Halo
|
||||
ssh root@192.168.68.4 # hwepve (abiba, kagentz) — Mumuni runs inside CT 100
|
||||
ssh root@192.168.68.4 # hwepve (abiba, kagentz, mumuni)
|
||||
```
|
||||
|
||||
## Section 7: Agent Health Check v2 (2026-07-26)
|
||||
## Section 7: Agent Health Check (consolidated — 2026-07-05)
|
||||
|
||||
Single non-disruptive check running every 10 minutes via cron
|
||||
(`/root/scripts/agent-health-check.py`). v2 fixes critical gaps:
|
||||
- Koby (.129) and Koonimo (.114) now have SSH hosts — no longer skipped
|
||||
- Agent-specific vault keys (`{NAME}_LITELLM_API_KEY`) not shared master key
|
||||
- CT liveness check via `pct status` on PVE nodes
|
||||
- Config YAML integrity check
|
||||
- Wrapper/CLI integrity check
|
||||
- Vault secret non-emptiness check
|
||||
Replaces 7 scattered Zulip health scripts with a single non-disruptive check
|
||||
running every 10 minutes via cron (`/root/scripts/agent-health-check.py`).
|
||||
|
||||
The script is **read-only** — it never restarts, kills, or modifies anything.
|
||||
Disruptive cron-based gateways restarts (like Mumuni's zulip-watchdog.sh,
|
||||
which was kill+nohup outside systemd) are banned by policy.
|
||||
|
||||
### Checks Performed
|
||||
|
||||
| Check | Frequency | What It Detects |
|
||||
|-------|-----------|-----------------|
|
||||
| LiteLLM key validation | 10 min | All 5 agent-specific keys authenticate (not shared master key) |
|
||||
| LiteLLM key validation | 10 min | All 4 agent keys authenticate and return models |
|
||||
| GPU port conflict | 10 min | Ghost processes squatting port 8080 (ss vs systemd MainPID) |
|
||||
| Gateway liveness | 10 min | Gateway process running, state file readable (all 5 agents) |
|
||||
| Gateway liveness | 10 min | Gateway process running, state file readable |
|
||||
| Zulip streaming | 10 min | `edit_message` present in adapter (streaming supported) |
|
||||
| Recent errors | 10 min | Error count in journald for last 10 min |
|
||||
| CT liveness | 10 min | `pct status` on PVE nodes — catches stopped CTs |
|
||||
| Config YAML integrity | 10 min | Python `yaml.safe_load()` — catches syntax errors |
|
||||
| Wrapper/CLI integrity | 10 min | hermes wrapper exists, infisical path correct, hermes-real reachable |
|
||||
| Vault secrets | 10 min | Agent-specific vault keys are non-empty and start with `sk-` |
|
||||
|
||||
### Disabled Scripts
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
kind: responsibility
|
||||
name: infrastructure-update
|
||||
description: >
|
||||
Autonomous system-wide update contract covering all 5 Proxmox nodes,
|
||||
Autonomous system-wide update contract covering all 6 Proxmox nodes,
|
||||
15+ containers/VMs, and 4 Docker ecosystems. Updates apt packages,
|
||||
Docker images, and container stacks in safe waves with health checks
|
||||
and automatic rollback on failure.
|
||||
@@ -56,10 +56,11 @@ Before ANY update wave:
|
||||
| amdpve (.15) | Proxmox node | `apt update && apt upgrade -y` | 5 min |
|
||||
| acerpve (.9) | Proxmox node | `apt update && apt upgrade -y` | 5 min |
|
||||
| ocupve (.5) | Proxmox node | `apt update && apt upgrade -y` | 5 min |
|
||||
| hwepve (.4) | Proxmox node | `apt update && apt upgrade -y` | 5 min |
|
||||
| CT 100 (.24) | Abiba (pi) | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 116 (.116) | syslog-api (LiteLLM host) | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 112 (tanko, amdpve) | Tanko | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 114 (mumuni, minipve) | Mumuni | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 114 (mumuni, hwepve) | Mumuni | `apt update && apt upgrade -y` | 3 min |
|
||||
| VM 101 (.8) | llm-gpu (RTX 3090) | `apt update && apt upgrade -y` | 3 min |
|
||||
| VM 103 (.110) | ocu-llm (RTX 5070) | `apt update && apt upgrade -y` | 3 min |
|
||||
|
||||
@@ -218,7 +219,7 @@ When LiteLLM is upgraded to a version supporting per-key MCP grants:
|
||||
|
||||
## Success Criteria
|
||||
|
||||
- [ ] All 5 PVE nodes updated, no reboot-loop
|
||||
- [ ] All 6 PVE nodes updated, no reboot-loop
|
||||
- [ ] All VMs/CTs running post-update
|
||||
- [ ] All Docker containers healthy (VM 109 + CT 116 + CT 117)
|
||||
- [ ] LiteLLM inference passing (syslog-auto test)
|
||||
@@ -234,7 +235,7 @@ After completion, send Zulip DM:
|
||||
```
|
||||
📋 Infrastructure Update — YYYY-MM-DD
|
||||
|
||||
Updated: 5 PVE nodes, 12 CTs/VMs, 30+ containers
|
||||
Updated: 6 PVE nodes, 12 CTs/VMs, 30+ containers
|
||||
Security fixes: N CVEs patched
|
||||
Downtime: <service> <duration>
|
||||
Failures: none / <details>
|
||||
|
||||
@@ -302,3 +302,20 @@ With failures:
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## gitea-logger Implementation
|
||||
|
||||
When this step executes, write the report JSON to a temp file and push to Gitea:
|
||||
|
||||
```bash
|
||||
REPO="https://abiba-bot:${GITEA_PAT}@git.sysloggh.net/SyslogSolution/health-logs"
|
||||
DIR="litellm"
|
||||
FILE="${run_id}.json"
|
||||
echo "${report_json}" > /tmp/${FILE}
|
||||
(cd /tmp && git clone --depth 1 "${REPO}" &&
|
||||
cp ${FILE} health-logs/${DIR}/${FILE} &&
|
||||
cd health-logs && git add ${DIR}/${FILE} &&
|
||||
git commit -m "litellm-health: ${run_id}" && git push)
|
||||
rm -rf /tmp/health-logs /tmp/${FILE}
|
||||
```
|
||||
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
---
|
||||
kind: pattern
|
||||
name: memory-fixer
|
||||
description: >
|
||||
Auto-fix low-hanging fruit in the graph. No judgment calls — only deterministic Level 1 operations.
|
||||
Escalate anything that needs Kwame's input.
|
||||
version: 1.1.0
|
||||
---
|
||||
|
||||
# Memory Fixer
|
||||
|
||||
## Purpose
|
||||
Auto-fix low-hanging fruit in the graph. No judgment calls — only deterministic Level 1 operations. Escalate anything that needs Kwame's input.
|
||||
|
||||
## Level 0 Auto-Deletes (Allowed Without Approval)
|
||||
Ephemeral heartbeat and log nodes that violate "Logs NEVER go in the graph":
|
||||
|
||||
- `[LITELLM-HEALTH]`, `[GPU-SELF-HEAL]`, `[PM2-SELF-HEAL]`
|
||||
- `[PROXMOX-MONITOR]`, `[GPU-MONITOR]`, `[INFRA-MONITOR]`, `[AGENT-HEALTH]`, `[DISK-GC]`
|
||||
- `[WAL]` entries older than 30 days
|
||||
|
||||
**Condition:** node must be an orphan (no edges). Deleting a connected node risks breaking other nodes.
|
||||
|
||||
**Method:** direct SQLite on `.65` (MCP has no delete tool):
|
||||
```bash
|
||||
ssh root@192.168.68.65 "sqlite3 /root/.local/share/RA-H/db/rah.sqlite \"
|
||||
DELETE FROM nodes WHERE id IN (
|
||||
SELECT id FROM nodes WHERE id NOT IN (SELECT from_node_id FROM edges)
|
||||
AND id NOT IN (SELECT to_node_id FROM edges)
|
||||
AND title LIKE '[LITELLM-HEALTH]%' -- add more prefixes as needed
|
||||
);\""
|
||||
```
|
||||
|
||||
## Level 1 Auto-Fixes (No Judgment Required)
|
||||
|
||||
### 1. Missing `type` Field
|
||||
For nodes with content but no `metadata.type`:
|
||||
- Title contains "Proxmox" or "infrastructure" → `type: infrastructure`
|
||||
- Title contains "skill" or "how to" or "guide" → `type: skill`
|
||||
- Title contains "doc" or "template" or "brand" → `type: documentation`
|
||||
- Title starts with "WAL:" or "TASK:" → `type: note`
|
||||
- Title starts with "[LEARN]" → `type: documentation`
|
||||
- Otherwise → `type: note` (default)
|
||||
|
||||
### 2. Missing `tenant` / `namespace`
|
||||
For any node with NULL tenant or namespace:
|
||||
```sql
|
||||
UPDATE nodes
|
||||
SET metadata = json_set(
|
||||
COALESCE(metadata, '{}'),
|
||||
'$.tenant', 'syslogsolution',
|
||||
'$.namespace', 'syslogsolution'
|
||||
)
|
||||
WHERE json_extract(metadata, '$.tenant') IS NULL
|
||||
OR json_extract(metadata, '$.namespace') IS NULL;
|
||||
```
|
||||
|
||||
### 3. Staleness State Transitions
|
||||
Using the type-based windows from the memory-monitor contract:
|
||||
- Nodes stale > their window → transition to `state: review_pending`
|
||||
- Nodes in `review_pending` for >7 days → escalate to Kwame (Level 2)
|
||||
|
||||
## Level 2 Escalations (Kwame Decision Required)
|
||||
1. **Nodes in `review_pending` >7 days** — Archive, refresh, or keep?
|
||||
2. **Orphan Nodes >90 days old** — Delete or Connect?
|
||||
3. **Potential Duplicate Nodes** — Same title or >70% overlap. Merge or Keep?
|
||||
4. **Conflicting Metadata** — Content suggests one tenant but metadata says another.
|
||||
|
||||
## Logging
|
||||
Every Level 1 fix logged to `~/.hermes/logs/memory-fixer/YYYY-MM-DD.md`
|
||||
Every Level 2 escalation logged and delivered to Kwame.
|
||||
@@ -31,7 +31,7 @@ Workers execute tasks on whatever infrastructure they're given — SSH to .6,
|
||||
## Why This Matters
|
||||
|
||||
Without enforced delegation, the manager consumes the full iteration budget
|
||||
(60 calls) on single-turn tasks — SSH to 5 nodes, check each VM, read logs —
|
||||
(60 calls) on single-turn tasks — SSH to 6 nodes, check each VM, read logs —
|
||||
leaving no capacity for actual coordination. The result: context overflow
|
||||
(59K tokens in system prompt), iteration exhaustion, and degraded response
|
||||
quality. This contract exists because I blew through my budget checking
|
||||
@@ -82,7 +82,7 @@ it asks the manager (via relay) — it doesn't go find it on its own.
|
||||
|
||||
**This is a hard rule, not a recommendation.** Violating it produces the exact
|
||||
type of discrepancy the kanban pipeline exists to prevent: a review worker finds
|
||||
"5 nodes present" in the raw data but "5/5 online" in the report — even though
|
||||
"6 nodes present" in the raw data but "6/6 online" in the report — even though
|
||||
one of those nodes was unreachable. The report lied because it used data the
|
||||
raw data never provided.
|
||||
|
||||
@@ -137,7 +137,7 @@ delegate_task(
|
||||
```
|
||||
delegate_task(
|
||||
tasks=[
|
||||
{"goal": "Check all 5 Proxmox nodes for VM status", "context": "SSH to each node via 192.168.68.x, run 'qm list'"},
|
||||
{"goal": "Check all 6 Proxmox nodes for VM status", "context": "SSH to each node via 192.168.68.x, run 'qm list'"},
|
||||
{"goal": "Check Docker container health on .7/.116/.17", "context": "SSH to each host, check container status"},
|
||||
]
|
||||
)
|
||||
@@ -191,7 +191,7 @@ Only verified results reach Kwame. Format per channel:
|
||||
{
|
||||
"lane_id": "devops-check",
|
||||
"worker": "syslog-devops",
|
||||
"goal": "Check all 5 Proxmox nodes",
|
||||
"goal": "Check all 6 Proxmox nodes",
|
||||
"status": "dispatched|completed|failed",
|
||||
"output_file": "/tmp/node-report.md"
|
||||
}
|
||||
|
||||
@@ -88,7 +88,7 @@ agent: abiba
|
||||
| acerpve | 192.168.68.9 | PVE (hosts llm-gpu qemu/101) |
|
||||
| minipve | 192.168.68.12 | PVE |
|
||||
| amdpve | 192.168.68.15 | PVE + Strix Halo LLM (qwen3.6-35B-udq4, strix-moe) |
|
||||
| hwepve | 192.168.68.4 | PVE (Huawei Matebook 16, 12C/15GB) — hosts Mumuni (lxc/114) migrated from minipve 2026-07-20 |
|
||||
| hwepve | 192.168.68.4 | PVE (Huawei Matebook 16, 12C/15GB) — hosts abiba (lxc/100), kagentz (lxc/105), mumuni (lxc/114). CTs 100/105 migrated from amdpve, CT 114 from minipve 2026-07-20 |
|
||||
|
||||
## Operations
|
||||
|
||||
|
||||
Regular → Executable
+9
-6
@@ -11,31 +11,33 @@ set -euo pipefail
|
||||
# ── CT ID → PVE Node mapping (maintained HERE, not in prose contracts) ──
|
||||
declare -A CT_NODES=(
|
||||
# amdpve (192.168.68.15)
|
||||
[100]=amdpve # abiba
|
||||
[105]=amdpve # kagentz
|
||||
[111]=amdpve # tdunna
|
||||
[112]=amdpve # tanko
|
||||
[113]=amdpve # baggy
|
||||
[115]=amdpve # scottdenya
|
||||
# minipve (192.168.68.12)
|
||||
[102]=minipve # adguard (was acerpve)
|
||||
[104]=minipve # authentik
|
||||
[110]=minipve # gitea
|
||||
[114]=minipve # mumuni
|
||||
[116]=minipve # syslog-api
|
||||
[119]=minipve # infisical-vault
|
||||
# storepve (192.168.68.6)
|
||||
[106]=storepve # ra-h-os
|
||||
[107]=storepve # proxmox-backup
|
||||
[108]=storepve # media
|
||||
[117]=storepve # zulip
|
||||
# acerpve (192.168.68.9)
|
||||
[102]=acerpve # adguard
|
||||
[118]=storepve # jdownloader
|
||||
# acerpve (192.168.68.9) — no CTs (bare metal GPU .8)
|
||||
# hwepve (192.168.68.4)
|
||||
[100]=hwepve # abiba (was amdpve)
|
||||
[105]=hwepve # kagentz (was amdpve)
|
||||
[114]=hwepve # mumuni (was minipve)
|
||||
# ocupve (192.168.68.5) — no CTs (bare metal GPU .110)
|
||||
#
|
||||
# REMOVED CTs (migrated to bare metal, decommissioned, or VMs):
|
||||
# 101 llm-gpu → bare metal 192.168.68.8 (RTX 3090)
|
||||
# 103 ocu-llm → bare metal 192.168.68.110 (RTX 5070)
|
||||
# 109 docker-vm → KVM VM 192.168.68.7 (use direct SSH)
|
||||
# 118 jitsi → stopped, not in service
|
||||
)
|
||||
|
||||
# Each node must be root-accessible via SSH hostname
|
||||
@@ -48,6 +50,7 @@ declare -A NODE_IPS=(
|
||||
[storepve]=192.168.68.6
|
||||
[acerpve]=192.168.68.9
|
||||
[ocupve]=192.168.68.5
|
||||
[hwepve]=192.168.68.4
|
||||
)
|
||||
|
||||
resolve_node() {
|
||||
|
||||
@@ -46,17 +46,19 @@ You are a code reviewer for OpenProse infrastructure contracts in the Syslog Sol
|
||||
|
||||
The infrastructure-control.prose.md contract is the canonical reference for the cluster topology:
|
||||
|
||||
**Proxmox Cluster "Tabiri" (5 nodes):**
|
||||
- amdpve (192.168.68.15): abiba, kagentz, tanko, tdunna, baggy, scottdenya
|
||||
- minipve (192.168.68.12): authentik, gitea, mumuni, syslog-api, jitsi
|
||||
- storepve (192.168.68.6): docker-vm, ra-h-os, PBS, media, zulip
|
||||
- acerpve (192.168.68.9): llm-gpu, adguard
|
||||
**Proxmox Cluster "Tabiri" (6 nodes):**
|
||||
- amdpve (192.168.68.15): tanko, tdunna, baggy, scottdenya
|
||||
- minipve (192.168.68.12): adguard, authentik, gitea, syslog-api, infisical-vault
|
||||
- storepve (192.168.68.6): docker-vm, ra-h-os, PBS, media, jdownloader, zulip
|
||||
- acerpve (192.168.68.9): llm-gpu
|
||||
- ocupve (192.168.68.5): ocu-llm
|
||||
- hwepve (192.168.68.4): abiba, kagentz, mumuni
|
||||
|
||||
**CT IDs (verified 2026-07-04 against PVE API):**
|
||||
**CT IDs (verified 2026-07-24 against PVE API):**
|
||||
100:abiba 102:adguard 104:authentik 105:kagentz 106:ra-h-os
|
||||
107:pbs 108:media 110:gitea 111:tdunna 112:tanko
|
||||
113:baggy 114:mumuni 115:scottdenya 116:syslog-api 117:zulip
|
||||
118:jdownloader 119:infisical-vault
|
||||
|
||||
**NO CT 122, CT 123, or .19 exist in the cluster.**
|
||||
|
||||
|
||||
Reference in New Issue
Block a user