Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6d975360d8 | ||
|
|
3e246835d5 | ||
|
|
06d2bcbc9e | ||
|
|
c4a8c45835 | ||
|
|
23f3f378c5 |
@@ -1 +0,0 @@
|
||||
__pycache__/
|
||||
@@ -1,230 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Hermes Config Audit — validates a live config.yaml against the prose contract rules.
|
||||
|
||||
Usage:
|
||||
python3 audit-hermes-config.py <config.yaml>
|
||||
python3 audit-hermes-config.py /root/.hermes/config.yaml
|
||||
|
||||
Exit codes:
|
||||
0 = all checks pass
|
||||
1 = one or more contract violations found
|
||||
|
||||
This script encodes every rule from hermes-config-template.prose.md so config
|
||||
changes can be verified before and after application. It is the single automated
|
||||
enforcement layer for the prose contract.
|
||||
|
||||
Contract: /root/prose-contracts/hermes-config-template.prose.md
|
||||
"""
|
||||
|
||||
import sys
|
||||
import yaml
|
||||
|
||||
VIOLATIONS = []
|
||||
WARNINGS = []
|
||||
PASSES = []
|
||||
|
||||
|
||||
def check(condition, rule, message):
|
||||
if condition:
|
||||
PASSES.append(f"[{rule}] {message}")
|
||||
else:
|
||||
VIOLATIONS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def warn(rule, message):
|
||||
WARNINGS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def audit(path):
|
||||
with open(path) as f:
|
||||
cfg = yaml.safe_load(f)
|
||||
|
||||
model = cfg.get("model", {})
|
||||
fb = cfg.get("fallback_providers", {})
|
||||
comp = cfg.get("compression", {})
|
||||
aux = cfg.get("auxiliary", {})
|
||||
deleg = cfg.get("delegation", {})
|
||||
cps = cfg.get("custom_providers", [])
|
||||
cp = cps[0] if cps else {}
|
||||
|
||||
# --- Rule 3: API Keys via Environment ---
|
||||
check(
|
||||
model.get("api_key") in ("", None),
|
||||
"Rule 3",
|
||||
f"model.api_key must be empty (got {model.get('api_key')!r}) — keys via env var, not hardcoded",
|
||||
)
|
||||
check(
|
||||
model.get("api_key_env") == "LITELLM_API_KEY",
|
||||
"Rule 3",
|
||||
f"model.api_key_env must be LITELLM_API_KEY (got {model.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- Rule 5: Main Config Base URL ---
|
||||
expected_base = "http://192.168.68.116/v1"
|
||||
check(
|
||||
model.get("base_url") == expected_base,
|
||||
"Rule 5",
|
||||
f"model.base_url must be {expected_base} (got {model.get('base_url')!r}) — /v1 not /litellm/v1",
|
||||
)
|
||||
|
||||
# --- Rule 6: max_tokens Is Required ---
|
||||
check(
|
||||
isinstance(model.get("max_tokens"), int) and model.get("max_tokens") <= 8192,
|
||||
"Rule 6",
|
||||
f"model.max_tokens must be set and <= 8192 (got {model.get('max_tokens')!r}) — thermal safety",
|
||||
)
|
||||
|
||||
# --- Rule 7: Auxiliary Model Consistency ---
|
||||
check(
|
||||
comp.get("model") == "syslog-auto",
|
||||
"Rule 7",
|
||||
f"compression.model must be syslog-auto (got {comp.get('model')!r}) — auto-routing to prevent Strix Halo overload",
|
||||
)
|
||||
aux_comp = aux.get("compression", {})
|
||||
check(
|
||||
aux_comp.get("model") == "syslog-auto",
|
||||
"Rule 7",
|
||||
f"auxiliary.compression.model must be syslog-auto (got {aux_comp.get('model')!r}) — must match compression.model",
|
||||
)
|
||||
|
||||
# --- Rule 8: GPU Workload Distribution ---
|
||||
check(
|
||||
aux.get("vision", {}).get("model") == "gpu-light",
|
||||
"Rule 8",
|
||||
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
check(
|
||||
aux.get("web_extract", {}).get("model") == "gpu-light",
|
||||
"Rule 8",
|
||||
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
|
||||
# --- Rule 9: Compression Threshold ---
|
||||
check(
|
||||
comp.get("threshold") == 0.65,
|
||||
"Rule 9",
|
||||
f"compression.threshold must be 0.65 for 128K models (got {comp.get('threshold')!r})",
|
||||
)
|
||||
check(
|
||||
comp.get("max_context_window") == 131072,
|
||||
"Rule 9",
|
||||
f"compression.max_context_window must be 131072 (got {comp.get('max_context_window')!r}) — matches 128K GPU capacity",
|
||||
)
|
||||
|
||||
# --- Rule 10: Default Model Must Be syslog-auto ---
|
||||
check(
|
||||
model.get("default") == "syslog-auto",
|
||||
"Rule 10",
|
||||
f"model.default must be syslog-auto (got {model.get('default')!r}) — auto-routing default",
|
||||
)
|
||||
|
||||
# --- Rule 14: Provider Name Must Match custom_providers Name ---
|
||||
check(
|
||||
model.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"model.provider must be 'harness' (got {model.get('provider')!r}) — NOT 'custom'. "
|
||||
f"provider: custom causes generic resolution path that ignores key_env → 'no-key-required' → 401",
|
||||
)
|
||||
check(
|
||||
comp.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"compression.provider must be 'harness' (got {comp.get('provider')!r})",
|
||||
)
|
||||
for aux_name in ("vision", "web_extract", "compression"):
|
||||
aux_provider = aux.get(aux_name, {}).get("provider")
|
||||
check(
|
||||
aux_provider == "harness",
|
||||
"Rule 14",
|
||||
f"auxiliary.{aux_name}.provider must be 'harness' (got {aux_provider!r})",
|
||||
)
|
||||
check(
|
||||
deleg.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
fb.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- custom_providers sanity ---
|
||||
check(
|
||||
cp.get("name") == "harness",
|
||||
"custom_providers",
|
||||
f"custom_providers[0].name must be 'harness' (got {cp.get('name')!r})",
|
||||
)
|
||||
check(
|
||||
cp.get("key_env") == "LITELLM_API_KEY" or cp.get("api_key_env") == "LITELLM_API_KEY",
|
||||
"custom_providers",
|
||||
f"custom_providers[0] must have key_env or api_key_env = LITELLM_API_KEY "
|
||||
f"(got key_env={cp.get('key_env')!r}, api_key_env={cp.get('api_key_env')!r})",
|
||||
)
|
||||
check(
|
||||
cp.get("base_url", "").endswith("/v1"),
|
||||
"custom_providers",
|
||||
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
||||
)
|
||||
|
||||
# --- No raw model names (Rule 7/8 spirit) ---
|
||||
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
|
||||
for section_path, section_dict in [
|
||||
("model", model), ("compression", comp),
|
||||
("auxiliary.vision", aux.get("vision", {})),
|
||||
("auxiliary.web_extract", aux.get("web_extract", {})),
|
||||
("auxiliary.compression", aux.get("compression", {})),
|
||||
("delegation", deleg),
|
||||
]:
|
||||
m = section_dict.get("model", "")
|
||||
if m in raw_names:
|
||||
warn(
|
||||
"Rule 7/8",
|
||||
f"{section_path}.model = {m!r} — raw model name, use stable alias instead "
|
||||
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
|
||||
)
|
||||
|
||||
# --- Report ---
|
||||
print(f"{'=' * 60}")
|
||||
print(f"Hermes Config Audit: {path}")
|
||||
print(f"{'=' * 60}")
|
||||
print(f"\n✅ PASSED ({len(PASSES)}):")
|
||||
for p in PASSES:
|
||||
print(f" ✅ {p}")
|
||||
|
||||
if WARNINGS:
|
||||
print(f"\n⚠️ WARNINGS ({len(WARNINGS)}):")
|
||||
for w in WARNINGS:
|
||||
print(f" ⚠️ {w}")
|
||||
|
||||
if VIOLATIONS:
|
||||
print(f"\n❌ VIOLATIONS ({len(VIOLATIONS)}):")
|
||||
for v in VIOLATIONS:
|
||||
print(f" ❌ {v}")
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"RESULT: FAIL — {len(VIOLATIONS)} violation(s) must be fixed")
|
||||
print(f"{'=' * 60}")
|
||||
return 1
|
||||
else:
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"RESULT: PASS — all contract rules satisfied")
|
||||
print(f"{'=' * 60}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python3 audit-hermes-config.py <config.yaml>")
|
||||
sys.exit(2)
|
||||
sys.exit(audit(sys.argv[1]))
|
||||
+7
-7
@@ -125,7 +125,7 @@ Note: All syslog-auto entries route directly to GPUs with `api_key: not-needed`.
|
||||
|
||||
| Alias | RPM Cap | Routes To | Purpose |
|
||||
|-------|---------|-----------|---------|
|
||||
| `strix-moe` | 40 | Strix Halo | Compression tasks (MoE models) |
|
||||
| `strix-moe` | 40 | Strix Halo | Agent reasoning, compression (~30% via syslog-auto pool) (MoE models) |
|
||||
| `gpu-dense` | 500 | RTX 3090 | Heavy reasoning |
|
||||
| `gpu-light` | 500 | RTX 5070 | Vision, web extract, light tasks |
|
||||
|
||||
@@ -136,7 +136,7 @@ Note: All syslog-auto entries route directly to GPUs with `api_key: not-needed`.
|
||||
- syslog-auto → qwen → gemma → qwen3.6-35B-udq4
|
||||
|
||||
### Why Strix Halo RPM Is Capped
|
||||
- Direct (strix-moe): 40 RPM (tight) — Strix Halo is shared with compression tasks
|
||||
- Direct (strix-moe): 40 RPM (tight) — Strix Halo handles agent reasoning + compression via syslog-auto pool
|
||||
- Via syslog-auto: 60 RPM (moderate) — prevents flooding when multiple agents use syslog-auto simultaneously
|
||||
- Combined max: ~100 RPM across both paths — Strix Halo can sustain this at 80°C
|
||||
|
||||
@@ -284,7 +284,7 @@ History stored at `/root/data/toks-history.json` with 7-day rolling window.
|
||||
### Stable Aliases — CRITICAL
|
||||
|
||||
All agent configs MUST use stable role-based aliases, never model-specific names:
|
||||
- `compression.model: strix-moe` (NOT `qwen3.6-35B-udq4`)
|
||||
- `compression.model: syslog-auto` — switched from strix-moe 2026-07-18 to distribute across weighted pool (55% RTX 3090, 30% Strix Halo, 15% RTX 5070); relieves Strix Halo thermal pressure
|
||||
- `auxiliary.vision.model: gpu-light` (NOT `gemma-4-12b`)
|
||||
- `delegation.model: gpu-dense` (NOT `qwen3.6-27B-code`)
|
||||
- `auxiliary.web_extract.model: gpu-light`
|
||||
@@ -295,7 +295,7 @@ When the underlying model is swapped, only the LiteLLM config changes — agent
|
||||
- RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K**
|
||||
- **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek)
|
||||
- Compression threshold 0.65: fires at ~85K (~43K headroom before 128K ceiling)
|
||||
- Mumuni compression model alias: `strix-moe` with 300s timeout
|
||||
- Mumuni compression model alias: `syslog-auto` (switched from strix-moe 2026-07-18) with 300s timeout
|
||||
|
||||
### Mumuni Agent Profile
|
||||
|
||||
@@ -305,8 +305,8 @@ Mumuni (CT114, 192.168.68.123) is the primary business assistant. This profile i
|
||||
|---------|-------|-------|
|
||||
| `model.default` | `syslog-auto` | Weighted pool (55% qwen, 30% strix, 15% gemma) |
|
||||
| `model.provider` | `custom:litellm` | LiteLLM on CT116 |
|
||||
| `compression.model` | `strix-moe` | Stable alias — survives model swaps |
|
||||
| `aux.compression.model` | `strix-moe` | Compression auxiliary model |
|
||||
| `compression.model` | `syslog-auto` | Switched from strix-moe 2026-07-18 — distributes across weighted pool |
|
||||
| `aux.compression.model` | `syslog-auto` | Compression auxiliary — switched from strix-moe 2026-07-18 |
|
||||
| `aux.vision.model` | `gpu-light` | Vision tasks (RTX 5070) |
|
||||
| `aux.web_extract.model` | `gpu-light` | Web extraction |
|
||||
| `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) |
|
||||
@@ -318,7 +318,7 @@ Mumuni (CT114, 192.168.68.123) is the primary business assistant. This profile i
|
||||
| `personalities` | `creative` | Creative assistant personality |
|
||||
| Platforms | cli, discord, homeassistant, signal, telegram, zulip | All Hermes platforms |
|
||||
| Main model timeout | 300s | LiteLLM global timeout |
|
||||
| Compression model timeout | 300s | strix-moe timeout increased from 120s |
|
||||
| Compression model timeout | 300s | syslog-auto timeout (switched from strix-moe 2026-07-18) |
|
||||
|
||||
### Agent Update Status (2026-07-15)
|
||||
|
||||
|
||||
@@ -46,7 +46,7 @@ depends_on:
|
||||
|-------|-----|------|-------|------|-----|-------|------|
|
||||
| `gpu-dense` | RTX 3090 24GB | ct8 (.8:8080) | ThinkingCap Qwen3.6-27B Q4_K_M + MTP + vision | 21.6/24.6GB (88%) | 128K | 74.9 | Heavy reasoning, code gen |
|
||||
| `gpu-light` | RTX 5070 12GB | ct110 (.110:8080) | HauhauCS Gemma4-12B QAT Q4_K_M + MTP draft | 10.1/12.2GB (83%) | 128K | 169.6 | Vision, web extract, light tasks |
|
||||
| `strix-moe` | Strix Halo 64GB | ct15 (.15:8080) | Genesis Hermes V3 APEX (LuffyTheFox, 24GB) | ~10/64GB (16%) | 128K | 62.9 | Compression, summarization, long docs |
|
||||
| `strix-moe` | Strix Halo 64GB | ct15 (.15:8080) | Genesis Hermes V3 APEX (LuffyTheFox, 24GB) | ~10/64GB (16%) | 128K | 62.9 | Agent reasoning, compression (~30% via syslog-auto pool), summarization, long docs |
|
||||
|
||||
Key notes:
|
||||
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) is deprecated and NOT in the inference path.
|
||||
@@ -157,7 +157,7 @@ Key notes:
|
||||
- **Target distribution**:
|
||||
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM).
|
||||
- RTX 5070 (gpu-light, 12GB, 169.6 tok/s) → Vision/image, web search, lightweight tasks (2.3x faster than 3090 per token). Weight: 0.15 (LiteLLM).
|
||||
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM).
|
||||
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Agent reasoning, compression (~30% via syslog-auto pool), summarization, long docs (MoE model). Weight: 0.30 (LiteLLM).
|
||||
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
|
||||
- **Fix**:
|
||||
- Alert if any GPU is handling workload outside its designated role
|
||||
|
||||
@@ -5,7 +5,7 @@ version: 1.0.0
|
||||
description: >
|
||||
Canonical known-good baseline for all Syslog Hermes agents. Captures the exact
|
||||
configuration state, keys, workarounds, and audit procedure. When an agent's
|
||||
configuration goes sideways, restore from this baseline. Last verified 2026-07-16. All GPUs 256K context (RTX 3090 .8, RTX 5070 .110, Strix Halo .15). Parallel 1 fleet-wide (Strix Halo handles compression solo).
|
||||
configuration goes sideways, restore from this baseline. Last verified 2026-07-16. All GPUs 128K context (RTX 3090 .8, RTX 5070 .110, Strix Halo .15). Parallel 1 fleet-wide (compression via syslog-auto pool — switched from strix-moe 2026-07-18).
|
||||
author: Abiba (pi agent)
|
||||
---
|
||||
|
||||
|
||||
@@ -5,11 +5,14 @@ description: >
|
||||
Standard Hermes configuration template for Syslog Solution LLC agents.
|
||||
Enforces shared infrastructure setup (Firecrawl, SearXNG, local models,
|
||||
RA-H OS MCP) while keeping agent-specific API keys and model choices.
|
||||
UPDATED 2026-07-16: Compression model is the stable alias `strix-moe` (NOT `ornith-1.0-35b`,
|
||||
UPDATED 2026-07-18: Compression model switched to `syslog-auto` (was `strix-moe`)
|
||||
to relieve Strix Halo pressure. syslog-auto distributes compression across the
|
||||
weighted pool (55% RTX 3090, 30% Strix Halo, 15% RTX 5070).
|
||||
UPDATED 2026-07-16: Compression model was the stable alias `strix-moe` (NOT `ornith-1.0-35b`,
|
||||
which LiteLLM does not serve). All 3 GPUs verified at 128K (reduced from 256K 2026-07-17 for stability).
|
||||
Added Rule 12 (Context-Issue Diagnostic) + Rule 13 (.env fallback enforcement) from the
|
||||
2026-07-16 Mumuni root-cause investigation (WAL #1300).
|
||||
UPDATED 2026-07-12: GPU workload redistributed. Compression → Strix Halo. RTX 3090 context verified at 128K. Infisical .env fallback required (Rule 3/13).
|
||||
UPDATED 2026-07-12: GPU workload redistributed. Compression → Strix Halo (later switched to syslog-auto 2026-07-18). RTX 3090 context verified at 128K. Infisical .env fallback required (Rule 3/13).
|
||||
---
|
||||
|
||||
## Maintains
|
||||
@@ -130,7 +133,9 @@ mcp_servers:
|
||||
# ─── Compression ───
|
||||
compression:
|
||||
enabled: true
|
||||
model: syslog-auto # ⚠️ Must match auxiliary.compression.model. Stable alias (gpu-fleet § Stable Role-Based Aliases). NOT ornith-1.0-35b (LiteLLM does not serve that name).
|
||||
model: syslog-auto # ⚠️ Switched from strix-moe 2026-07-18 to relieve Strix Halo.
|
||||
# syslog-auto distributes across weighted pool (55% RTX 3090,
|
||||
# 30% Strix Halo, 15% RTX 5070). All GPUs at 128K.
|
||||
provider: harness
|
||||
max_context_window: 131072 # MUST match actual GPU capacity. All 3 GPUs are 128K (Jul 17).
|
||||
threshold: 0.65 # Fires at ~170K for 262K window, ~85K for 128K
|
||||
@@ -145,8 +150,9 @@ compression:
|
||||
# model: gpu-light # stable alias (NOT raw "gemma-4-12b")
|
||||
# base_url: http://192.168.68.116/v1
|
||||
# api_key_env: LITELLM_API_KEY
|
||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# Compression uses syslog-auto (switched from strix-moe 2026-07-18) to distribute
|
||||
# load across the weighted pool and relieve Strix Halo pressure.
|
||||
# Vision and web_extract use gpu-light = RTX 5070 (12B).
|
||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||
# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4)
|
||||
# in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
@@ -166,7 +172,7 @@ auxiliary:
|
||||
timeout: 30
|
||||
compression:
|
||||
provider: harness
|
||||
model: syslog-auto # MUST match compression.model above. Stable alias for Strix Halo (weighted pool).
|
||||
model: syslog-auto # Switched from strix-moe 2026-07-18. Relieves Strix Halo pressure.
|
||||
base_url: http://192.168.68.116/v1 # Rule 5: /v1 NOT /litellm/v1
|
||||
api_key_env: LITELLM_API_KEY
|
||||
timeout: 300 # gpu-fleet: 300s for large-history summarization (was 60)
|
||||
@@ -247,34 +253,30 @@ The following MUST be identical across ALL profiles:
|
||||
- Apply to BOTH main config AND all sub-agent profiles
|
||||
- For agents needing longer outputs: raise to 8192, but never omit
|
||||
|
||||
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
|
||||
- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized)
|
||||
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized)
|
||||
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b`
|
||||
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
||||
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
|
||||
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
|
||||
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
|
||||
fall back to other GPUs if Strix gets hot. Both `compression.model` and `auxiliary.compression.model`
|
||||
MUST be `syslog-auto`.
|
||||
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-18)
|
||||
- Vision and web_extract use `gpu-light` (stable alias, RTX 5070 — 12GB, vision-optimized)
|
||||
- Compression now uses `syslog-auto` (switched from `strix-moe` 2026-07-18) to distribute
|
||||
compression load across the weighted pool (55% RTX 3090, 30% Strix Halo, 15% RTX 5070).
|
||||
This relieves Strix Halo pressure while keeping compression functional on all GPUs.
|
||||
- **`syslog-auto` is the valid compression model** — LiteLLM serves it as the weighted pool.
|
||||
Old configs with `strix-moe` for compression should be updated to `syslog-auto`.
|
||||
- All auxiliary services MUST use identical routing:
|
||||
- `base_url: http://192.168.68.116/v1` (Rule 5: `/v1`, NOT `/litellm/v1`)
|
||||
- `api_key_env: LITELLM_API_KEY`
|
||||
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably
|
||||
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
|
||||
(64GB UMA, 128K context) — the designated compression GPU. This frees the
|
||||
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
|
||||
- **Compression via syslog-auto**: Routes through the weighted pool. Strix Halo still handles
|
||||
~30% of compression calls (at 60 RPM via pool vs 40 RPM direct), but the bulk (55%)
|
||||
goes to RTX 3090 which has ample spare capacity.
|
||||
- The `compression:` block's `model` MUST match `auxiliary: compression: model`
|
||||
- The `compression: max_context_window: 131072` MUST match actual GPU capacity (128K)
|
||||
|
||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
|
||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
|
||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
||||
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
|
||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-18)
|
||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations — also handles ~55% of compression via syslog-auto pool
|
||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract — handles ~15% of compression via syslog-auto pool
|
||||
- **Strix Halo (64GB, 128K ctx, Genesis Hermes V3 APEX)**: Agent reasoning, compression (~30% via syslog-auto pool), fallback for other GPUs
|
||||
- Agent profiles MUST route auxiliary tasks to the correct GPU:
|
||||
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
|
||||
- `auxiliary.vision.model: gpu-light` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gpu-light` (RTX 5070)
|
||||
- `auxiliary.compression.model: syslog-auto` (distributed pool, switched from strix-moe 2026-07-18)
|
||||
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
|
||||
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
|
||||
- Do NOT use `threshold: 0.25` — this fires at 65K, causing premature context loss
|
||||
@@ -376,26 +378,6 @@ curl -s -o /dev/null -w '%{http_code}' -H "Authorization: Bearer $K" http://192.
|
||||
- `/etc/environment` is NO LONGER the canonical key source (stale values there caused 401s).
|
||||
- Do NOT leave a hardcoded stale key in `/etc/environment` — it shadows the drop-in/wrapper.
|
||||
|
||||
### Rule 14: Provider Name Must Match custom_providers Name (ADDED 2026-07-19, WAL #1471)
|
||||
|
||||
- `model.provider` MUST be `harness` (the `custom_providers[0].name`), NOT the literal string `custom`
|
||||
- When `provider: custom`, Hermes' `_get_named_custom_provider("custom")` returns None (no provider is
|
||||
named "custom" — it is named "harness"), causing a fall-through to the generic resolution path
|
||||
(`source: env/config`) at `runtime_provider.py:1156`
|
||||
- The generic path builds `api_key_candidates` from `model.api_key` (empty), host-gated
|
||||
OLLAMA/OPENAI/OPENROUTER keys, and `_host_derived_api_key` (returns "" for IP addresses)
|
||||
- **The generic path does NOT resolve `model.api_key_env` or `custom_providers.key_env`** —
|
||||
`LITELLM_API_KEY` is never read, producing `api_key = "no-key-required"` → HTTP 401
|
||||
- The named custom provider path (`source: custom_provider:harness`) DOES read `key_env` —
|
||||
but only triggers when `provider` matches the `custom_providers[0].name`
|
||||
- All sections MUST use `provider: harness`: `model`, `compression`, `auxiliary.vision`,
|
||||
`auxiliary.web_extract`, `auxiliary.compression`, `delegation`
|
||||
- Only `fallback_providers` uses a different provider (`deepseek`) for true fallback diversity
|
||||
- **Diagnostic**: If you see `source: env/config` in a request dump or log, the provider name
|
||||
is wrong. It should be `source: custom_provider:harness`.
|
||||
- **Audit script**: Run `python3 /root/prose-contracts/audit-hermes-config.py <config.yaml>`
|
||||
before and after any config change to catch this and all other rule violations.
|
||||
|
||||
## Execution
|
||||
|
||||
1. **Check current config** — Read the target agent's config.yaml
|
||||
|
||||
@@ -57,7 +57,7 @@ duration.
|
||||
prefill time at 532 tok/s. Fix context first, routing second.
|
||||
|
||||
- **Route by task**: ornith for multi-step reasoning only; qwen for code/standard
|
||||
queries; gemma for compression/auxiliary. Never send simple completion to a
|
||||
queries; gemma for vision/web extraction; syslog-auto for compression (via weighted pool). Never send simple completion to a
|
||||
35B MoE.
|
||||
- **Compress aggressively**: threshold at 40% (not 65%) — a 256K window should
|
||||
compact at 102K, not 166K. Target 15% tail (not 30%).
|
||||
|
||||
@@ -1,71 +0,0 @@
|
||||
---
|
||||
kind: pattern
|
||||
name: memory-fixer
|
||||
description: >
|
||||
Auto-fix low-hanging fruit in the graph. No judgment calls — only deterministic Level 1 operations.
|
||||
Escalate anything that needs Kwame's input.
|
||||
version: 1.1.0
|
||||
---
|
||||
|
||||
# Memory Fixer
|
||||
|
||||
## Purpose
|
||||
Auto-fix low-hanging fruit in the graph. No judgment calls — only deterministic Level 1 operations. Escalate anything that needs Kwame's input.
|
||||
|
||||
## Level 0 Auto-Deletes (Allowed Without Approval)
|
||||
Ephemeral heartbeat and log nodes that violate "Logs NEVER go in the graph":
|
||||
|
||||
- `[LITELLM-HEALTH]`, `[GPU-SELF-HEAL]`, `[PM2-SELF-HEAL]`
|
||||
- `[PROXMOX-MONITOR]`, `[GPU-MONITOR]`, `[INFRA-MONITOR]`, `[AGENT-HEALTH]`, `[DISK-GC]`
|
||||
- `[WAL]` entries older than 30 days
|
||||
|
||||
**Condition:** node must be an orphan (no edges). Deleting a connected node risks breaking other nodes.
|
||||
|
||||
**Method:** direct SQLite on `.65` (MCP has no delete tool):
|
||||
```bash
|
||||
ssh root@192.168.68.65 "sqlite3 /root/.local/share/RA-H/db/rah.sqlite \"
|
||||
DELETE FROM nodes WHERE id IN (
|
||||
SELECT id FROM nodes WHERE id NOT IN (SELECT from_node_id FROM edges)
|
||||
AND id NOT IN (SELECT to_node_id FROM edges)
|
||||
AND title LIKE '[LITELLM-HEALTH]%' -- add more prefixes as needed
|
||||
);\""
|
||||
```
|
||||
|
||||
## Level 1 Auto-Fixes (No Judgment Required)
|
||||
|
||||
### 1. Missing `type` Field
|
||||
For nodes with content but no `metadata.type`:
|
||||
- Title contains "Proxmox" or "infrastructure" → `type: infrastructure`
|
||||
- Title contains "skill" or "how to" or "guide" → `type: skill`
|
||||
- Title contains "doc" or "template" or "brand" → `type: documentation`
|
||||
- Title starts with "WAL:" or "TASK:" → `type: note`
|
||||
- Title starts with "[LEARN]" → `type: documentation`
|
||||
- Otherwise → `type: note` (default)
|
||||
|
||||
### 2. Missing `tenant` / `namespace`
|
||||
For any node with NULL tenant or namespace:
|
||||
```sql
|
||||
UPDATE nodes
|
||||
SET metadata = json_set(
|
||||
COALESCE(metadata, '{}'),
|
||||
'$.tenant', 'syslogsolution',
|
||||
'$.namespace', 'syslogsolution'
|
||||
)
|
||||
WHERE json_extract(metadata, '$.tenant') IS NULL
|
||||
OR json_extract(metadata, '$.namespace') IS NULL;
|
||||
```
|
||||
|
||||
### 3. Staleness State Transitions
|
||||
Using the type-based windows from the memory-monitor contract:
|
||||
- Nodes stale > their window → transition to `state: review_pending`
|
||||
- Nodes in `review_pending` for >7 days → escalate to Kwame (Level 2)
|
||||
|
||||
## Level 2 Escalations (Kwame Decision Required)
|
||||
1. **Nodes in `review_pending` >7 days** — Archive, refresh, or keep?
|
||||
2. **Orphan Nodes >90 days old** — Delete or Connect?
|
||||
3. **Potential Duplicate Nodes** — Same title or >70% overlap. Merge or Keep?
|
||||
4. **Conflicting Metadata** — Content suggests one tenant but metadata says another.
|
||||
|
||||
## Logging
|
||||
Every Level 1 fix logged to `~/.hermes/logs/memory-fixer/YYYY-MM-DD.md`
|
||||
Every Level 2 escalation logged and delivered to Kwame.
|
||||
@@ -409,3 +409,68 @@ Backup v2 before starting: `cp index.js index.js.v2-backup-$(date +%Y%m%d-%H%M%S
|
||||
| Queue expiry handling | Crash | Auto re-register |
|
||||
| Busy worker deadlock | Router death | Worker SIGKILL + error DM |
|
||||
| PM2 restart exhaustion | Yes (max_restarts=10) | No (max_restarts=100 + watchdog) |
|
||||
|
||||
---
|
||||
|
||||
## Incident Log — 2026-07-18 Fleet-Wide Audit
|
||||
|
||||
### Fleet State After Audit
|
||||
|
||||
| Agent | Platform | Zulip State | Issues Found | Fix Applied |
|
||||
|-------|----------|-------------|--------------|-------------|
|
||||
| **Abiba** | pi (CT 100) | ✅ Connected | API key missing from Infisical injection; poll timeout noise | Added .env fallback; AbortError treated as empty poll (no retry); poll timeout 65s→90s |
|
||||
| **Tanko** | Hermes (CT 112) | ✅ Connected | Gateway disconnected since Jul 11; watchdog restart didn't re-establish Zulip | Full gateway restart (kill wrapper, let infisical-gateway.sh respawn) |
|
||||
| **Mumuni** | Hermes (CT 114) | ✅ Connected | No issues found | None needed |
|
||||
|
||||
### Key Fixes Applied
|
||||
|
||||
**1. Abiba — Credential Fallback (L4 Pattern)**
|
||||
- Root cause: `zulip.api_key` in config.yaml is `""` (expected from Infisical). Infisical vault `ABIBA_ZULIP_API_KEY` wasn't being injected into the process environment.
|
||||
- Fix: Added `.env` file fallback at `/root/.pi/agent/extensions/zulip/.env` with known-working key, sourced before the Infisical `exec`.
|
||||
- Lesson: Per L4 from gpu-self-heal, Infisical is not always available — always keep a local `.env` fallback.
|
||||
|
||||
**2. Abiba — Poll Timeout Handling**
|
||||
- Root cause: Zulip long-poll uses `AbortSignal.timeout(65000)`. Zulip's default `event_queue_longpoll_timeout_seconds` can exceed 65s. When the signal fires, an `AbortError` is thrown and caught by the circuit breaker as a failure.
|
||||
- Fix: Caught `AbortError` inside `poll()` and return empty array (no events) instead of throwing. Extended timeout to 90s to match Zulip server default.
|
||||
- Reference: [Zulip Events System — long-poll timeout](https://zulip.readthedocs.io/en/11.6/subsystems/events-system.html)
|
||||
|
||||
**3. Tanko — Gateway Restart**
|
||||
- Root cause: Gateway process was running but Zulip platform stayed in "disconnected" state since Jul 11, 2026. The wrapper script (`infisical-gateway.sh`) restarts on crash but the gateway wasn't re-establishing Zulip on restart.
|
||||
- Fix: Killed gateway PID to trigger wrapper restart. New gateway (PID 331991) established Zulip connection successfully.
|
||||
|
||||
### Fleet-Wide Zulip Health Metrics (as of 2026-07-18)
|
||||
|
||||
| Metric | Value |
|
||||
|--------|-------|
|
||||
| Zulip server | ✅ HTTP 200 |
|
||||
| Agents connected | 3/3 (Abiba, Tanko, Mumuni) |
|
||||
| Abiba circuit breaker | CLOSED (0 failures) |
|
||||
| Abiba uptime | 2D (post-restart) |
|
||||
| Tanko gateway uptime | Ongoing |
|
||||
| Mumuni gateway uptime | Ongoing |
|
||||
| Watchdog status | ✅ Online (2D uptime) |
|
||||
|
||||
### Hermes Agent Zulip Plugin Improvements
|
||||
|
||||
Based on the audit, improvements that should be ported to all Hermes Zulip adapters:
|
||||
|
||||
1. **Circuit breaker pattern** — Already in Abiba's pi extension. Hermes adapters should add the same CLOSED→OPEN→HALF_OPEN state machine with exponential backoff.
|
||||
2. **Credential fallback** — All Hermes agents use Infisical for credentials. Add `.env` local fallback per L4 pattern for `ZULIP_API_KEY`.
|
||||
3. **Queue re-registration** — Handle `BAD_EVENT_QUEUE_ID` with automatic re-registration instead of gateway restart.
|
||||
4. **Supervisor watchdog** — Hermes uses PM2 which auto-restarts on crash, but has no health-check watchdog. Add lightweight external health checks.
|
||||
5. **Streaming** — All agents have `streaming: true` in their zulip config. Verify `edit_message()` is implemented in each adapter.
|
||||
|
||||
### Abiba pi Zulip Extension v2 — Implemented Resilience Summary
|
||||
|
||||
| Feature | Status | Notes |
|
||||
|---------|--------|-------|
|
||||
| Circuit breaker | ✅ | CLOSED→OPEN→HALF_OPEN; 50% failure threshold; 30s reset timeout |
|
||||
| Retry with jitter | ✅ | 2 attempts, 200ms base, 50-100% jitter |
|
||||
| Queue lifecycle | ✅ | 10min idle_queue_timeout; BAD_EVENT_QUEUE_ID handling |
|
||||
| Crash prevention | ✅ | uncaughtException + unhandledRejection recovery |
|
||||
| Worker timeout | ✅ | 5min busy timeout → SIGKILL + error DM |
|
||||
| Health endpoint | ✅ | :9200 with circuit breaker metrics |
|
||||
| Echo prevention | ✅ | Dynamic bot user resolution |
|
||||
| Poll timeout (AbortError) | ✅ v2.1 | Normal timeout returns [] instead of error |
|
||||
| Credential fallback | ✅ v2.1 | .env file before Infisical exec |
|
||||
| Provider auto-fix | ✅ | Detects reasoning_content models, switches to compatible |
|
||||
|
||||
Reference in New Issue
Block a user