Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cd479caeec | ||
|
|
1b1de8b0fc | ||
|
|
a74229ee74 | ||
|
|
aebc98ead6 | ||
|
|
17a77e6b3f | ||
|
|
14d27a09b5 | ||
|
|
bddbb22f03 | ||
|
|
31ec70ae36 |
@@ -0,0 +1 @@
|
||||
__pycache__/
|
||||
@@ -0,0 +1,230 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Hermes Config Audit — validates a live config.yaml against the prose contract rules.
|
||||
|
||||
Usage:
|
||||
python3 audit-hermes-config.py <config.yaml>
|
||||
python3 audit-hermes-config.py /root/.hermes/config.yaml
|
||||
|
||||
Exit codes:
|
||||
0 = all checks pass
|
||||
1 = one or more contract violations found
|
||||
|
||||
This script encodes every rule from hermes-config-template.prose.md so config
|
||||
changes can be verified before and after application. It is the single automated
|
||||
enforcement layer for the prose contract.
|
||||
|
||||
Contract: /root/prose-contracts/hermes-config-template.prose.md
|
||||
"""
|
||||
|
||||
import sys
|
||||
import yaml
|
||||
|
||||
VIOLATIONS = []
|
||||
WARNINGS = []
|
||||
PASSES = []
|
||||
|
||||
|
||||
def check(condition, rule, message):
|
||||
if condition:
|
||||
PASSES.append(f"[{rule}] {message}")
|
||||
else:
|
||||
VIOLATIONS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def warn(rule, message):
|
||||
WARNINGS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
def audit(path):
|
||||
with open(path) as f:
|
||||
cfg = yaml.safe_load(f)
|
||||
|
||||
model = cfg.get("model", {})
|
||||
fb = cfg.get("fallback_providers", {})
|
||||
comp = cfg.get("compression", {})
|
||||
aux = cfg.get("auxiliary", {})
|
||||
deleg = cfg.get("delegation", {})
|
||||
cps = cfg.get("custom_providers", [])
|
||||
cp = cps[0] if cps else {}
|
||||
|
||||
# --- Rule 3: API Keys via Environment ---
|
||||
check(
|
||||
model.get("api_key") in ("", None),
|
||||
"Rule 3",
|
||||
f"model.api_key must be empty (got {model.get('api_key')!r}) — keys via env var, not hardcoded",
|
||||
)
|
||||
check(
|
||||
model.get("api_key_env") == "LITELLM_API_KEY",
|
||||
"Rule 3",
|
||||
f"model.api_key_env must be LITELLM_API_KEY (got {model.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- Rule 5: Main Config Base URL ---
|
||||
expected_base = "http://192.168.68.116/v1"
|
||||
check(
|
||||
model.get("base_url") == expected_base,
|
||||
"Rule 5",
|
||||
f"model.base_url must be {expected_base} (got {model.get('base_url')!r}) — /v1 not /litellm/v1",
|
||||
)
|
||||
|
||||
# --- Rule 6: max_tokens Is Required ---
|
||||
check(
|
||||
isinstance(model.get("max_tokens"), int) and model.get("max_tokens") <= 8192,
|
||||
"Rule 6",
|
||||
f"model.max_tokens must be set and <= 8192 (got {model.get('max_tokens')!r}) — thermal safety",
|
||||
)
|
||||
|
||||
# --- Rule 7: Auxiliary Model Consistency ---
|
||||
check(
|
||||
comp.get("model") == "syslog-auto",
|
||||
"Rule 7",
|
||||
f"compression.model must be syslog-auto (got {comp.get('model')!r}) — auto-routing to prevent Strix Halo overload",
|
||||
)
|
||||
aux_comp = aux.get("compression", {})
|
||||
check(
|
||||
aux_comp.get("model") == "syslog-auto",
|
||||
"Rule 7",
|
||||
f"auxiliary.compression.model must be syslog-auto (got {aux_comp.get('model')!r}) — must match compression.model",
|
||||
)
|
||||
|
||||
# --- Rule 8: GPU Workload Distribution ---
|
||||
check(
|
||||
aux.get("vision", {}).get("model") == "gpu-light",
|
||||
"Rule 8",
|
||||
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
check(
|
||||
aux.get("web_extract", {}).get("model") == "gpu-light",
|
||||
"Rule 8",
|
||||
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
|
||||
# --- Rule 9: Compression Threshold ---
|
||||
check(
|
||||
comp.get("threshold") == 0.65,
|
||||
"Rule 9",
|
||||
f"compression.threshold must be 0.65 for 128K models (got {comp.get('threshold')!r})",
|
||||
)
|
||||
check(
|
||||
comp.get("max_context_window") == 131072,
|
||||
"Rule 9",
|
||||
f"compression.max_context_window must be 131072 (got {comp.get('max_context_window')!r}) — matches 128K GPU capacity",
|
||||
)
|
||||
|
||||
# --- Rule 10: Default Model Must Be syslog-auto ---
|
||||
check(
|
||||
model.get("default") == "syslog-auto",
|
||||
"Rule 10",
|
||||
f"model.default must be syslog-auto (got {model.get('default')!r}) — auto-routing default",
|
||||
)
|
||||
|
||||
# --- Rule 14: Provider Name Must Match custom_providers Name ---
|
||||
check(
|
||||
model.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"model.provider must be 'harness' (got {model.get('provider')!r}) — NOT 'custom'. "
|
||||
f"provider: custom causes generic resolution path that ignores key_env → 'no-key-required' → 401",
|
||||
)
|
||||
check(
|
||||
comp.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"compression.provider must be 'harness' (got {comp.get('provider')!r})",
|
||||
)
|
||||
for aux_name in ("vision", "web_extract", "compression"):
|
||||
aux_provider = aux.get(aux_name, {}).get("provider")
|
||||
check(
|
||||
aux_provider == "harness",
|
||||
"Rule 14",
|
||||
f"auxiliary.{aux_name}.provider must be 'harness' (got {aux_provider!r})",
|
||||
)
|
||||
check(
|
||||
deleg.get("provider") == "harness",
|
||||
"Rule 14",
|
||||
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
fb.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- custom_providers sanity ---
|
||||
check(
|
||||
cp.get("name") == "harness",
|
||||
"custom_providers",
|
||||
f"custom_providers[0].name must be 'harness' (got {cp.get('name')!r})",
|
||||
)
|
||||
check(
|
||||
cp.get("key_env") == "LITELLM_API_KEY" or cp.get("api_key_env") == "LITELLM_API_KEY",
|
||||
"custom_providers",
|
||||
f"custom_providers[0] must have key_env or api_key_env = LITELLM_API_KEY "
|
||||
f"(got key_env={cp.get('key_env')!r}, api_key_env={cp.get('api_key_env')!r})",
|
||||
)
|
||||
check(
|
||||
cp.get("base_url", "").endswith("/v1"),
|
||||
"custom_providers",
|
||||
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
||||
)
|
||||
|
||||
# --- No raw model names (Rule 7/8 spirit) ---
|
||||
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
|
||||
for section_path, section_dict in [
|
||||
("model", model), ("compression", comp),
|
||||
("auxiliary.vision", aux.get("vision", {})),
|
||||
("auxiliary.web_extract", aux.get("web_extract", {})),
|
||||
("auxiliary.compression", aux.get("compression", {})),
|
||||
("delegation", deleg),
|
||||
]:
|
||||
m = section_dict.get("model", "")
|
||||
if m in raw_names:
|
||||
warn(
|
||||
"Rule 7/8",
|
||||
f"{section_path}.model = {m!r} — raw model name, use stable alias instead "
|
||||
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
|
||||
)
|
||||
|
||||
# --- Report ---
|
||||
print(f"{'=' * 60}")
|
||||
print(f"Hermes Config Audit: {path}")
|
||||
print(f"{'=' * 60}")
|
||||
print(f"\n✅ PASSED ({len(PASSES)}):")
|
||||
for p in PASSES:
|
||||
print(f" ✅ {p}")
|
||||
|
||||
if WARNINGS:
|
||||
print(f"\n⚠️ WARNINGS ({len(WARNINGS)}):")
|
||||
for w in WARNINGS:
|
||||
print(f" ⚠️ {w}")
|
||||
|
||||
if VIOLATIONS:
|
||||
print(f"\n❌ VIOLATIONS ({len(VIOLATIONS)}):")
|
||||
for v in VIOLATIONS:
|
||||
print(f" ❌ {v}")
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"RESULT: FAIL — {len(VIOLATIONS)} violation(s) must be fixed")
|
||||
print(f"{'=' * 60}")
|
||||
return 1
|
||||
else:
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"RESULT: PASS — all contract rules satisfied")
|
||||
print(f"{'=' * 60}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python3 audit-hermes-config.py <config.yaml>")
|
||||
sys.exit(2)
|
||||
sys.exit(audit(sys.argv[1]))
|
||||
+9
-15
@@ -9,9 +9,9 @@ description: >
|
||||
gpu-dense, gpu-light. These never change — only the underlying model does.
|
||||
Strix Halo: strix-moe → unsloth/Qwen3.6-35B-A3B-MTP (UD-Q4_K_M, 22GB).
|
||||
RTX 5070: gemma-4-12b Q4_K_M → IQ4_NL + MTP draft (122 tok/s, 2x faster).
|
||||
UPDATED 2026-07-17: Strix Halo model swapped to Genesis Hermes V3 APEX (LuffyTheFox, 24GB, uncensored,
|
||||
UPDATED 2026-07-17: Context reduced fleet-wide from 256K to 128K for stability.
|
||||
Strix Halo model swapped to Genesis Hermes V3 APEX (LuffyTheFox, 24GB, uncensored,
|
||||
Hermes agent fine-tune, tensor repair, multimodal with mmproj).
|
||||
RTX 5070 swapped to HauhauCS Gemma4-12B QAT Uncensored Balanced (Q4_K_M, 87 tok/s, 0/465 refusals).
|
||||
Instability observed near 100K at 256K. 128K is the stable ceiling.
|
||||
For larger context needs → fall back to external providers (deepseek).
|
||||
VRAM headroom improved: RTX 3090 ~70%, RTX 5070 ~65%.
|
||||
@@ -93,20 +93,14 @@ When a model is swapped on a GPU, ONLY the infrastructure layer changes — agen
|
||||
**Backward compatibility**: Old model-specific names (qwen3.6-27B-code, gemma-4-12b, qwen3.6-35B-udq4) still work
|
||||
but are deprecated for agent configs. Only the stable aliases survive model swaps.
|
||||
|
||||
## Current Model Assignments (2026-07-17)
|
||||
## Current Model Assignments (2026-07-15)
|
||||
|
||||
| Model | GPU | Host | VRAM | Ctx | KV Cache | Parallel | Batch/Ubatch | Status |
|
||||
|-------|-----|------|------|-----|----------|----------|-------------|--------|
|
||||
| qwen3.6-27B-code (MTP) | RTX 3090 | .8 (llm-gpu) | ~17/24.6GB (70%) | **128K** | turbo4 | 2 | default | ✅ 63 tok/s |
|
||||
| gemma-4-12b (HauhauCS QAT) | RTX 5070 | .110 (ocu-llm) | ~10.0/12.2GB (82%) | 128K | q4_0 | 1 | 2048/1024 | ✅ 87 tok/s |
|
||||
| gemma-4-12b | RTX 5070 | .110 (ocu-llm) | ~7.8/12.2GB (65%) | 128K | q4_0 | 2 | 2048/1024 | ✅ healthy |
|
||||
| Genesis Hermes V3 APEX | Strix Halo Vulkan | .15 (amdpve) | ~10GB/64GB | 128K | q4_0 | 1 | 4096/1024 | ✅ 65 tok/s |
|
||||
|
||||
> **RTX 5070 model swap (2026-07-17)**: Switched from `gemma-4-12b-it-IQ4_NL` (Unsloth, 191 tok/s)
|
||||
> to `HauhauCS/Gemma4-12B-QAT-Uncensored-HauhauCS-Balanced` (Q4_K_M QAT, 87 tok/s).
|
||||
> Trade: 54% slower generation for QAT quality, 0/465 refusals, and agent-optimized tuning.
|
||||
> MTP draft also swapped: Q8_0 (444MB) → tuned draft (242MB), saving 200MB VRAM.
|
||||
> Role unchanged: gpu-light (vision, web extract, light auxiliary tasks).
|
||||
|
||||
## Routing Configuration (LiteLLM — July 2026)
|
||||
|
||||
### syslog-auto Weighted Pool (Direct GPU — bypasses router)
|
||||
@@ -125,7 +119,7 @@ Note: All syslog-auto entries route directly to GPUs with `api_key: not-needed`.
|
||||
|-------|---------|-------|
|
||||
| strix-moe (Hermes V3) | 40 | Tight cap — prevents Strix overload |
|
||||
| qwen3.6-27B-code | 500 | High cap — primary workhorse |
|
||||
| gemma-4-12b | 500 | HauhauCS QAT Uncensored Balanced + MTP, 87 tok/s |
|
||||
| gemma-4-12b | 500 | High cap — IQ4_NL+MTP, 122 tok/s |
|
||||
|
||||
### Stable Aliases (for agent configs — never change)
|
||||
|
||||
@@ -133,7 +127,7 @@ Note: All syslog-auto entries route directly to GPUs with `api_key: not-needed`.
|
||||
|-------|---------|-----------|---------|
|
||||
| `strix-moe` | 40 | Strix Halo | Compression tasks (MoE models) |
|
||||
| `gpu-dense` | 500 | RTX 3090 | Heavy reasoning |
|
||||
| `gpu-light` | 500 | RTX 5070 (HauhauCS QAT) | Vision, web extract, light tasks |
|
||||
| `gpu-light` | 500 | RTX 5070 | Vision, web extract, light tasks |
|
||||
|
||||
### Fallback Chains
|
||||
- gemma → qwen
|
||||
@@ -255,8 +249,8 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions.
|
||||
- **LiteLLM /metrics**: Requires auth. Prometheus uses `/health/liveliness` as workaround.
|
||||
- **VRAM (2026-07-15)**: RTX 3090 at ~17/24.6GB (~70%) with **128K context** (reduced from 256K 2026-07-17). RTX 5070 at ~7.8/12.2GB (~65%) with 128K context + MTP. Strix Halo at ~7GB/64GB.
|
||||
- **RTX 3090 runs `--parallel 2`** with MTP draft (spec-type draft-mtp, spec-draft-n-max 2).
|
||||
- **RTX 3090 config**: `-c 131072 -ctk turbo4 -ctv turbo4 --parallel 1 --flash-attn on --cont-batching --spec-type draft-mtp`. Context 128K, single slot for full 128K per-request. VRAM: ~85%. Service: `/home/llmuser/llama-wrapper.sh`.
|
||||
- **RTX 5070 config (2026-07-17)**: HauhauCS Gemma4-12B QAT Uncensored Balanced (Q4_K_M) + tuned MTP draft (242MB) at 128K context, single slot. Gen speed: 87 tok/s (vs 191 IQ4_NL). VRAM: ~10.0/12.2GB (~82%). Service: `/home/llmuser/llama-wrapper.sh`. Model: `Gemma4-12B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf`, MTP: `mtp-gemma-4-12B-it.gguf`, mmproj: `mmproj-Gemma4-12B-QAT-Uncensored-HauhauCS-Balanced-BF16.gguf`. Recommended sampling: temp 0.6, top_k 64, top_p 0.9, min_p 0.05, repeat_penalty 1.1.
|
||||
- **RTX 3090 config**: `-c 131072 -ctk turbo4 -ctv turbo4 --parallel 2 --flash-attn on --cont-batching --spec-type draft-mtp`. Context reduced to 128K (2026-07-17, was 256K). VRAM: ~70%. Service: `/home/llmuser/llama-wrapper.sh`.
|
||||
- **RTX 5070 config (2026-07-15)**: Switched to IQ4_NL + MTP draft (Q8_0) at 128K context. Gen speed: 122 tok/s. VRAM: ~7.8/12.2GB (~65%). Service: `/home/llmuser/llama-wrapper.sh`. Config: `--model gemma-4-12b-it-IQ4_NL.gguf --spec-draft-model gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --ctx-size 131072`.
|
||||
- **LiteLLM timeout tuning (verified 2026-07-16 against `/opt/inference-harness/litellm_config.yaml` on CT 116)**: gemma-4-12b 120s, qwen3.6-27B-code 300s, qwen3.6-35B-udq4 300s, strix-moe 300s, syslog-auto routes all 300s. Nginx proxy_read_timeout: 600s. Global request_timeout: 300s.
|
||||
- **Strix Halo GPU**: Vulkan is the working backend (ROCm/HIP path abandoned — HSA runtime blocked on Debian 13). Build at `/root/llama.cpp/build-vk/`, commit `4fc4ec5` (2026-07-01), ggml 0.15.3 shared-lib arch. Mesa RADV 25.0.7, KHR_coopmat fast path active. ~70 tok/s gen, 532 tok/s prompt. Service: `strix-server.service` on port 8080, model: `LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-V3-GGUF` (APEX quant), alias `strix-moe`, 128K context, flash-attn + q4 KV, multimodal (mmproj loaded). Hermes agent fine-tune, tensor repair (SSM layers fixed via SVD), uncensored (0/465 refusals).
|
||||
- **Port conflict detection (2026-07-05)**: All 3 GPU wrappers now detect ghost processes squatting port 8080 before starting. `.8` and `.110` use inline pre-start check in `llama-wrapper.sh`; `.15` uses `/usr/local/bin/port-cleanup.sh` ExecStartPre. Replaces the blanket `pkill -9 -x llama-server` on .15 which would kill ALL llama-server instances regardless of port. Ghost detection was the root cause of .8 crash-looping for 27+ restarts (stale pid 25836 squatting 8080 after OOM kill).
|
||||
@@ -273,7 +267,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions.
|
||||
| GPU | Model | Gen tok/s | Prompt tok/s | Baseline | Context |
|
||||
|-----|-------|-----------|--------------|----------|---------|
|
||||
| RTX 3090 (.8) | qwen3.6-27B-code (MTP) | **63** | — | — | **128K** |
|
||||
| RTX 5070 (.110) | HauhauCS QAT Uncensored Balanced | **87** | — | — | **128K** |
|
||||
| RTX 5070 (.110) | gemma-4-12b (IQ4_NL+MTP) | **191** | — | — | **128K** |
|
||||
| Strix Halo (.15) | Genesis Hermes V3 APEX | **65** | 140 | — | **128K** |
|
||||
|
||||
Benchmarks from 2026-07-17. Strix Halo swapped to Genesis Hermes V3 APEX (LuffyTheFox). RTX 5070 MTP provides 2.7x speedup over pre-upgrade 70 tok/s.
|
||||
|
||||
+93
-69
@@ -6,10 +6,15 @@ description: >
|
||||
benchmarks, and predicts failures before they happen. Extends gpu-monitor
|
||||
(v2.1.0) with active remediation rules, Prometheus metrics consumption,
|
||||
VRAM trend analysis, and predictive alerting.
|
||||
UPDATED 2026-07-18: Model assignments synced to 2026-07-17 swaps.
|
||||
Router (port 9000) references replaced with direct GPU routing.
|
||||
Benchmark baselines refreshed to live values.
|
||||
Prometheus exporters removed — not deployed; fall back to direct sidecar probes.
|
||||
Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet.
|
||||
agent: abiba
|
||||
depends_on:
|
||||
- gpu-monitor.prose.md (live data source on .24:9100)
|
||||
- gpu-fleet.prose.md (source of truth for topology)
|
||||
- gpu-fleet.prose.md (source of truth for topology, aliases, model assignments)
|
||||
---
|
||||
|
||||
## Maintains
|
||||
@@ -22,8 +27,8 @@ depends_on:
|
||||
|
||||
## Requires
|
||||
|
||||
- gpu-monitor:function — Live fleet data from .24:9100/gpu-data
|
||||
- Prometheus exporters on all 3 GPUs (:9400/metrics)
|
||||
- gpu-monitor:function — Live fleet data from localhost:9100/gpu-data
|
||||
- Direct sidecar probe access to all GPU hosts (:8080/health)
|
||||
- SSH access to GPU hosts for restart operations
|
||||
|
||||
## Continuity
|
||||
@@ -35,21 +40,37 @@ depends_on:
|
||||
|
||||
---
|
||||
|
||||
## Current Fleet Baseline (2026-07-18)
|
||||
|
||||
| Alias | GPU | Host | Model | VRAM | Ctx | tok/s | Role |
|
||||
|-------|-----|------|-------|------|-----|-------|------|
|
||||
| `gpu-dense` | RTX 3090 24GB | ct8 (.8:8080) | ThinkingCap Qwen3.6-27B Q4_K_M + MTP + vision | 21.6/24.6GB (88%) | 128K | 74.9 | Heavy reasoning, code gen |
|
||||
| `gpu-light` | RTX 5070 12GB | ct110 (.110:8080) | HauhauCS Gemma4-12B QAT Q4_K_M + MTP draft | 10.1/12.2GB (83%) | 128K | 169.6 | Vision, web extract, light tasks |
|
||||
| `strix-moe` | Strix Halo 64GB | ct15 (.15:8080) | Genesis Hermes V3 APEX (LuffyTheFox, 24GB) | ~10/64GB (16%) | 128K | 62.9 | Compression, summarization, long docs |
|
||||
|
||||
Key notes:
|
||||
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) is deprecated and NOT in the inference path.
|
||||
- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated.
|
||||
- RTX 5070 tok/s is 2.3x faster than RTX 3090 for its model — gpu-light is the fastest endpoint. Route vision/web/light work there first.
|
||||
- Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads.
|
||||
- RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom).
|
||||
- RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes).
|
||||
|
||||
## Remediation Rules
|
||||
|
||||
### Rule 1: GPU Temperature Critical (>85°C for >2 min)
|
||||
- **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls
|
||||
- **Fix**:
|
||||
1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control)
|
||||
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains
|
||||
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gemma → qwen, qwen → gemma)
|
||||
3. If all GPUs hot, alert about cooling infrastructure
|
||||
- **Verify**: Temp drops below 80°C within 5 minutes
|
||||
- **Escalate after**: 3 verification failures → Zulip alert
|
||||
|
||||
### Rule 2: VRAM Leak Detection (tiered by GPU capacity)
|
||||
- **Detect**: VRAM growing at sustained rate over 6+ hour window
|
||||
- RTX 3090 (24GB): ≥100MB/hour
|
||||
- RTX 5070 (12GB): ≥50MB/hour
|
||||
- RTX 3090 (24GB): ≥300MB/hour
|
||||
- RTX 5070 (12GB): ≥300MB/hour
|
||||
- Strix Halo (64GB UMA): ≥200MB/hour
|
||||
- **Fix**:
|
||||
1. Log VRAM snapshot with process list (nvidia-smi/rocm-smi + ps aux)
|
||||
@@ -69,6 +90,9 @@ depends_on:
|
||||
|
||||
### Rule 4: Benchmark Regression (>20% drop)
|
||||
- **Detect**: gen_tok_per_sec drops >20% below baseline over 3+ benchmarks
|
||||
- RTX 3090 baseline: 74.8 tok/s → alert at <59.8 tok/s
|
||||
- RTX 5070 baseline: 165.2 tok/s → alert at <132.2 tok/s
|
||||
- Strix Halo baseline: 70.5 tok/s → alert at <56.4 tok/s
|
||||
- **Fix**:
|
||||
1. Check GPU utilization — if >90%, other process is competing
|
||||
2. Check power limit — if throttled, restore to max
|
||||
@@ -78,32 +102,31 @@ depends_on:
|
||||
|
||||
### Rule 5: Circuit Breaker Stuck Open
|
||||
- **Detect**: Circuit breaker open >10 minutes with GPU reporting healthy
|
||||
- **Note**: Router (port 9000) is deprecated. If circuit breakers are reported by gpu-monitor, they come from LiteLLM's internal tracking, not the old router.
|
||||
- **Fix**:
|
||||
1. Verify GPU /health returns 200
|
||||
2. If GPU healthy, send 1 test inference
|
||||
3. If test succeeds → reset circuit breaker via router API
|
||||
4. 60s cooldown — if CB re-opens immediately, it was legitimate, do NOT re-reset
|
||||
5. Max 1 auto-reset per GPU per hour
|
||||
- **Verify**: CB closes, inference succeeds, CB stays closed for 60s+
|
||||
- **Escalate after**: CB won't close after reset → router issue
|
||||
1. Verify GPU /health returns 200 on direct port (:8080)
|
||||
2. If GPU healthy, alert but do NOT reset via router API (deprecated)
|
||||
3. Check LiteLLM health directly: http://192.168.68.116/litellm/health/liveliness
|
||||
4. Restart LiteLLM container on CT 116 if circuit breakers are stuck
|
||||
- **Verify**: LiteLLM returns healthy, circuit breaker clears within 60s
|
||||
- **Escalate after**: LiteLLM restart doesn't clear → human investigation
|
||||
|
||||
### Rule 6: Strix Halo Unreachable
|
||||
- **Detect**: Strix not responding — probe .15:8080 directly (firewall opened .24→.15)
|
||||
- **Fix**:
|
||||
1. SSH to .15 → check llama-server process
|
||||
2. Restart llama-server if not running
|
||||
3. Verify through both direct probe AND router
|
||||
- **Verify**: Direct health probe returns 200, router reports Strix healthy
|
||||
3. Verify through both direct probe AND LiteLLM health
|
||||
- **Verify**: Direct health probe returns 200, LiteLLM reports model healthy
|
||||
- **Escalate**: If host .15 itself is unreachable → infrastructure alert
|
||||
|
||||
### Rule 7: Prometheus Exporter Down
|
||||
- **Detect**: Any GPU :9400/metrics unreachable for >2 polls
|
||||
### Rule 7: GPU Data Source Unreachable (replaces old Prometheus rule)
|
||||
- **Detect**: gpu-monitor endpoint (localhost:9100/gpu-data) or sidecar port (:8080) on any GPU unreachable for >2 polls
|
||||
- **Fix**:
|
||||
1. SSH to GPU host → check prometheus-exporter process
|
||||
2. Restart exporter if dead
|
||||
3. While exporter is down, fall back to nvidia-smi/rocm-smi direct probes
|
||||
4. If exporter is running but unreachable → check firewall/host networking
|
||||
- **Verify**: :9400/metrics returns 200
|
||||
1. If gpu-monitor is down: restart systemd service `gpu-monitor.service` on this host
|
||||
2. If sidecar is down: SSH to GPU host → check llama-server process → restart systemd service
|
||||
3. Fall back to direct nvidia-smi/rocm-smi probe via SSH if all API paths fail
|
||||
- **Verify**: gpu-monitor returns healthy + all sidecars reachable
|
||||
- **Escalate after**: 3 failed restarts → networking issue
|
||||
|
||||
### Rule 8: Predictive Thermal Warning (two-tier)
|
||||
@@ -117,29 +140,31 @@ depends_on:
|
||||
- **Escalate**: If Tier 2 triggers and temp still rising after 5 min → possible hardware failure
|
||||
|
||||
### Rule 9: Context Window Optimization
|
||||
- **Detect**: Benchmark tok/s vs baseline for each GPU at current context
|
||||
- RTX 3090 (128K ctx, qwen3.6-27B-code): target 75+ tok/s — currently at baseline
|
||||
- RTX 5070 (128K ctx, gemma-4-12b): target 76+ tok/s — optimal for vision/web role
|
||||
- Strix Halo (128K ctx, strix-moe / qwen3.6-35B-udq4): target 70+ tok/s — currently above baseline
|
||||
- **Detect**: Benchmark tok/s vs baseline for each GPU at current context (all 128K)
|
||||
- RTX 3090 (128K ctx, ThinkingCap): baseline 74.8 tok/s — currently at 74.9 (100%)
|
||||
- RTX 5070 (128K ctx, HauhauCS QAT): baseline 165.2 tok/s — currently at 169.6 (103%)
|
||||
- Strix Halo (128K ctx, Genesis Hermes V3): baseline 70.5 tok/s — currently at 62.9 (89%)
|
||||
- **Fix**:
|
||||
- If tok/s > baseline → context has headroom, consider increasing
|
||||
- If tok/s < 90% baseline → reduce context by 25% and retest
|
||||
- If tok/s within 10% of baseline → optimal, no change
|
||||
- Strix Halo at 89% of baseline → MONITOR but do not reduce yet (recent model swap may still be settling)
|
||||
- **Verify**: Re-benchmark after context change, confirm within 10% of target
|
||||
- **Escalate**: If context can't be adjusted without significant perf loss
|
||||
|
||||
### Rule 10: Workload Distribution Optimization
|
||||
### Rule 10: Workload Distribution Optimization (updated 2026-07-18)
|
||||
- **Detect**: GPU roles misaligned with hardware capabilities
|
||||
- **Target distribution**:
|
||||
- RTX 3090 (24GB, 128K, 75 tok/s) → Heavy reasoning, code gen, long conversations
|
||||
- RTX 5070 (12GB, 128K, 76 tok/s) → Vision/image, web search, quick lightweight tasks
|
||||
- Strix Halo (64GB, 128K, 72 tok/s) → Context compression, summarization, long docs
|
||||
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM).
|
||||
- RTX 5070 (gpu-light, 12GB, 169.6 tok/s) → Vision/image, web search, lightweight tasks (2.3x faster than 3090 per token). Weight: 0.15 (LiteLLM).
|
||||
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM).
|
||||
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
|
||||
- **Fix**:
|
||||
- Alert if any GPU is handling workload outside its designated role
|
||||
- Recommend Hermes agent profile updates to match workload to GPU
|
||||
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe)
|
||||
- Track per-GPU request distribution via LiteLLM spend logs
|
||||
- **Verify**: Each GPU's request pattern matches its designated role within 24h
|
||||
- **Escalate**: If role mismatch persists >48h → agent profile audit needed
|
||||
- **Escalate**: If role mismatch persists >48h → agent alias audit needed
|
||||
|
||||
---
|
||||
|
||||
@@ -148,7 +173,7 @@ depends_on:
|
||||
```prose
|
||||
-- Phase 1: Fetch live GPU data
|
||||
let fleet = call gpu-monitor
|
||||
endpoint: "http://192.168.68.24:9100/gpu-data"
|
||||
endpoint: "http://localhost:9100/gpu-data"
|
||||
|
||||
-- Phase 2: Evaluate each GPU against remediation rules
|
||||
let actions = []
|
||||
@@ -164,27 +189,25 @@ for gpu in fleet.gpus:
|
||||
|
||||
-- Rule 4: Benchmark regression
|
||||
let bench = fleet.benchmarks[gpu.hostname]
|
||||
if bench.current_tok_sec < bench.baseline_tok_sec * 0.8:
|
||||
if bench.current_tok_s < bench.baseline_tok_s * 0.8:
|
||||
push actions apply-benchmark-fix(gpu, bench)
|
||||
|
||||
-- Rule 3: Model stuck
|
||||
for model in fleet.router.available_models:
|
||||
for model in fleet.summary.available_models:
|
||||
if model.consecutive_timeouts >= 3:
|
||||
push actions apply-model-restart(model)
|
||||
|
||||
-- Rule 5: Circuit breaker
|
||||
for cb in fleet.router.circuit_breaker:
|
||||
if cb.open and cb.open_duration > 600 and gpu_is_healthy(cb.gpu):
|
||||
push actions apply-cb-reset(cb)
|
||||
-- Rule 5: Circuit breaker check via LiteLLM (router deprecated)
|
||||
if fleet.summary.circuit_breakers_open > 0:
|
||||
push actions check-litellm-circuit-breakers()
|
||||
|
||||
-- Rule 6: Strix Halo
|
||||
if not fleet.strix.running and pingable("192.168.68.15"):
|
||||
push actions apply-strix-restart()
|
||||
|
||||
-- Rule 7: Prometheus exporters
|
||||
for gpu in fleet.gpus:
|
||||
if not prometheus_reachable(gpu.hostname, 9400):
|
||||
push actions apply-exporter-restart(gpu)
|
||||
-- Rule 7: GPU data source
|
||||
if not fleet.gpus or len(fleet.gpus) < 2:
|
||||
push actions check-gpu-monitor-service()
|
||||
|
||||
-- Rule 8: Predictive thermal
|
||||
for gpu in fleet.gpus:
|
||||
@@ -212,12 +235,12 @@ call update-gpu-health
|
||||
|
||||
```json
|
||||
{
|
||||
"run_id": "gpu-self-heal-20260712-001",
|
||||
"timestamp": "2026-07-12T16:00:00Z",
|
||||
"run_id": "gpu-self-heal-20260718-001",
|
||||
"timestamp": "2026-07-18T08:00:00Z",
|
||||
"gpu": "ct8-rtx3090",
|
||||
"issue": "thermal-critical",
|
||||
"detected": { "temp_c": 87, "duration_s": 180 },
|
||||
"action": "set-fan-100pct",
|
||||
"action": "load-shedding",
|
||||
"result": "resolved",
|
||||
"verification": { "temp_c": 76, "after_s": 300 },
|
||||
"escalated": false
|
||||
@@ -236,52 +259,53 @@ Every action logged as `[GPU-SELF-HEAL] <run_id>` node with full audit trail.
|
||||
- `issues_escalated > 0` → "⚠ GPU Self-Heal — <gpu> needs attention"
|
||||
- Every 100th clean cycle → "✅ GPU Fleet: All Clear"
|
||||
|
||||
### 3. Prometheus/Grafana Integration
|
||||
- GPU self-heal actions exposed as Prometheus counter metrics
|
||||
- Dashboard panel: "GPU Interventions (24h)" showing count/type/result
|
||||
|
||||
### 4. Weekly Benchmark Report
|
||||
### 3. Weekly Benchmark Report
|
||||
- Per-GPU tok/s trend over 7 days
|
||||
- Regression alerts if any GPU degrades >10% week-over-week
|
||||
|
||||
---
|
||||
|
||||
## Design Decisions (Grilled & Confirmed — 2026-07-12)
|
||||
## Design Decisions (Verified 2026-07-12, Reaffirmed 2026-07-18)
|
||||
|
||||
1. **Fan control**: ❌ NO auto fan control. Load-side cooling only (reduce concurrency, redirect).
|
||||
2. **Model restart**: ✅ Only if >50% failure rate over 60s + 30s grace period. Not on single stuck request.
|
||||
3. **Strix direct access**: ✅ Open firewall .15:8080 → .24 for direct health probe + restart.
|
||||
4. **VRAM thresholds**: Tiered — 100MB/h (RTX 3090), 50MB/h (RTX 5070), 200MB/h (Strix).
|
||||
5. **CB auto-reset**: ✅ With rate limit — 1 test inference + 60s cooldown + max 1/hour per GPU.
|
||||
6. **Benchmark baseline**: Rolling 30-day average, recalculated weekly. Original baseline kept in Grafana.
|
||||
4. **VRAM thresholds**: Tiered — **300MB/h** (RTX 3090), **300MB/h** (RTX 5070), 200MB/h (Strix). Previous values (100/50) were too sensitive; raised 2026-07-18 based on operational data.
|
||||
5. **CB auto-reset**: ✅ Router deprecated — circuit breakers go through LiteLLM health check + container restart if needed. No per-GPU auto-reset.
|
||||
6. **Benchmark baseline**: Rolling 30-day average, recalculated weekly. Current baselines live in gpu-monitor.
|
||||
7. **Predictive alerts**: Two-tier — warn at >70°C+rising (>2°C/min), critical at >80°C+rising.
|
||||
8. **Prometheus**: Primary source. Fall back to nvidia-smi/rocm-smi direct probes if exporter down.
|
||||
8. **Prometheus**: ❌ Not deployed. Use direct sidecar probes (:8080/health) and gpu-monitor API. Prometheus integration deferred until exporters are running on GPU hosts.
|
||||
|
||||
## Lessons Learned (2026-07-12)
|
||||
## Lessons Learned (2026-07-12, Updated 2026-07-18)
|
||||
|
||||
### L1: API Key Standardization Is Critical
|
||||
- All GPU llama-servers MUST use the same api-key as the LiteLLM config.
|
||||
- RTX 5070 had `--api-key sk-loc...5678` while LiteLLM sent `not-needed`.
|
||||
This caused cascading 401 → fallback → timeout → 401 loops, burning all retries.
|
||||
- **Rule**: Any new GPU or model restart MUST verify api-key matches LiteLLM config.
|
||||
This caused cascading 401 → fallback → timeout → 401 loops.
|
||||
- **Rule**: Any new GPU or model restart MUST verify api-key matches LiteLLM config (`not-needed` for direct routing).
|
||||
|
||||
### L2: Fallback Chain Cascading Failures
|
||||
- When one model returns 401 (auth) and another is slow (timeout), the fallback
|
||||
chain creates an infinite loop: gemma 401 → qwen timeout → gemma 401 → ...
|
||||
chain creates an infinite loop.
|
||||
- **Rule**: If a model returns 401 (auth error), do NOT fall back to it again.
|
||||
Mark it as permanently failed for this request.
|
||||
|
||||
### L3: Verify Running State, Not Docs
|
||||
- RTX 3090 was documented at 128K context. Actually running at 256K.
|
||||
- Parallel count wrong (docs said 2, actual is 1 on RTX 3090).
|
||||
- RTX 3090 was documented at 128K context. Running at 128K (verified 2026-07-18).
|
||||
- Parallel count: 1 on both RTX 3090 and RTX 5070 (matches docs for current models).
|
||||
- **Rule**: Before making decisions, check `/proc/PID/cmdline` on GPU hosts.
|
||||
|
||||
### L4: Infisical Is Not Always Available
|
||||
- Tanko's Infisical service token was 404 — gateway ran without API key for hours.
|
||||
- **Rule**: Always keep a local `.env` fallback for `LITELLM_API_KEY`.
|
||||
- Contract hermes-config-template Rule 3 updated.
|
||||
- Keep a local `.env` fallback for `LITELLM_API_KEY`.
|
||||
- **Rule**: Always verify credential source is reachable before relying on it.
|
||||
|
||||
### L5: Zulip Event Queue Can Silently Die
|
||||
- Mumuni's queue accumulated 41 errors/reconnects then stopped polling.
|
||||
Gateway was running but ignoring all messages.
|
||||
- **Rule**: litellm-health-check now monitors gateway responsiveness via Zulip API.
|
||||
### L5: GPU Monitor Response Size Can Cause Self-Heal Crash
|
||||
- gpu-self-heal crashed with KeyboardInterrupt during json.loads() of 20MB response.
|
||||
- Root cause: router poll returns accumulated data → cache balloons.
|
||||
- **Rule**: Self-heal must enforce a read timeout AND max response size on every poll.
|
||||
If monitor response > 1MB, log a warning and skip the cycle rather than crashing.
|
||||
|
||||
### L6: Stable Aliases Replace Model Names
|
||||
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15.
|
||||
- Self-heal must use aliases for reporting and alerting, not model-specific names.
|
||||
- **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier.
|
||||
|
||||
@@ -193,6 +193,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||
"models": [
|
||||
{ "id": "syslog-auto" },
|
||||
{ "id": "strix-moe" },
|
||||
{ "id": "gpu-dense" },
|
||||
{ "id": "gpu-light" },
|
||||
{ "id": "qwen3.6-27B-code" },
|
||||
{ "id": "gemma-4-12b" }
|
||||
]
|
||||
|
||||
@@ -130,7 +130,7 @@ mcp_servers:
|
||||
# ─── Compression ───
|
||||
compression:
|
||||
enabled: true
|
||||
model: strix-moe # ⚠️ Must match auxiliary.compression.model. Stable alias (gpu-fleet § Stable Role-Based Aliases). NOT ornith-1.0-35b (LiteLLM does not serve that name).
|
||||
model: syslog-auto # ⚠️ Must match auxiliary.compression.model. Stable alias (gpu-fleet § Stable Role-Based Aliases). NOT ornith-1.0-35b (LiteLLM does not serve that name).
|
||||
provider: harness
|
||||
max_context_window: 131072 # MUST match actual GPU capacity. All 3 GPUs are 128K (Jul 17).
|
||||
threshold: 0.65 # Fires at ~170K for 262K window, ~85K for 128K
|
||||
@@ -166,7 +166,7 @@ auxiliary:
|
||||
timeout: 30
|
||||
compression:
|
||||
provider: harness
|
||||
model: strix-moe # MUST match compression.model above. Stable alias for Strix Halo.
|
||||
model: syslog-auto # MUST match compression.model above. Stable alias for Strix Halo (weighted pool).
|
||||
base_url: http://192.168.68.116/v1 # Rule 5: /v1 NOT /litellm/v1
|
||||
api_key_env: LITELLM_API_KEY
|
||||
timeout: 300 # gpu-fleet: 300s for large-history summarization (was 60)
|
||||
@@ -252,6 +252,11 @@ The following MUST be identical across ALL profiles:
|
||||
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized)
|
||||
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b`
|
||||
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
||||
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
|
||||
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
|
||||
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
|
||||
fall back to other GPUs if Strix gets hot. Both `compression.model` and `auxiliary.compression.model`
|
||||
MUST be `syslog-auto`.
|
||||
- All auxiliary services MUST use identical routing:
|
||||
- `base_url: http://192.168.68.116/v1` (Rule 5: `/v1`, NOT `/litellm/v1`)
|
||||
- `api_key_env: LITELLM_API_KEY`
|
||||
@@ -265,11 +270,11 @@ The following MUST be identical across ALL profiles:
|
||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
|
||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
|
||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
||||
- **Strix Halo (64GB, 128K ctx, strix-moe)**: Context compression, summarization, long docs
|
||||
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
|
||||
- Agent profiles MUST route auxiliary tasks to the correct GPU:
|
||||
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.compression.model: strix-moe` (Strix Halo)
|
||||
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
|
||||
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
|
||||
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
|
||||
- Do NOT use `threshold: 0.25` — this fires at 65K, causing premature context loss
|
||||
@@ -325,7 +330,8 @@ verify ALL FOUR of these against the live config. They are the only root causes
|
||||
|
||||
One-line agent health check (run on the agent host):
|
||||
```bash
|
||||
PID=$(pgrep -f "python -m hermes_cli.main gateway run" | head -1)
|
||||
# Use grep -v infisical to avoid matching the bash wrapper that contains the same string
|
||||
PID=$(pgrep -f "python -m hermes_cli.main gateway run" | grep -v infisical | head -1)
|
||||
cat /proc/$PID/environ | tr '\0' '\n' | grep ^LITELLM_API_KEY= | sed 's/=.*/<set>/'
|
||||
curl -s -o /dev/null -w 'key_health: %{http_code}\n' -H "Authorization: Bearer $(cat /proc/$PID/environ | tr '\0' '\n' | grep ^LITELLM_API_KEY= | cut -d= -f2)" http://192.168.68.116/v1/models
|
||||
```
|
||||
@@ -351,15 +357,45 @@ directly (no infisical). Apply with `systemctl daemon-reload && systemctl restar
|
||||
The wrapper sources `~/.hermes/.env` then exports `LITELLM_API_KEY="$<AGENT>_LITELLM_API_KEY"`.
|
||||
See litellm-api-keys.prose.md § Machine Identity for Vault Writes for vault sync.
|
||||
|
||||
**⚠️ Vault empty-key guard:** If the vault stores the secret as an empty string,
|
||||
the wrapper will inject an empty key and the gateway will silently get 401 errors
|
||||
on all LiteLLM requests (triggering silent DeepSeek fallback). The `.env` fallback
|
||||
is present but the vault takes precedence when the secret key exists (even if empty).
|
||||
|
||||
**Fix:** The wrapper MUST validate the key length after injection. If LITELLM_API_KEY
|
||||
is empty or shorter than 20 chars, log a warning and either fail with a clear error
|
||||
message or fall back to the `.env` value before starting the gateway.
|
||||
|
||||
**Verification (all agents):**
|
||||
```bash
|
||||
GP=$(pgrep -f "python -m hermes_cli.main gateway run" | head -1)
|
||||
# Use grep -v infisical to avoid matching the bash wrapper that contains the same string
|
||||
GP=$(pgrep -f "python -m hermes_cli.main gateway run" | grep -v infisical | head -1)
|
||||
K=$(cat /proc/$GP/environ | tr '\0' '\n' | grep '^LITELLM_API_KEY=' | cut -d= -f2)
|
||||
curl -s -o /dev/null -w '%{http_code}' -H "Authorization: Bearer $K" http://192.168.68.116/v1/models # must be 200
|
||||
```
|
||||
- `/etc/environment` is NO LONGER the canonical key source (stale values there caused 401s).
|
||||
- Do NOT leave a hardcoded stale key in `/etc/environment` — it shadows the drop-in/wrapper.
|
||||
|
||||
### Rule 14: Provider Name Must Match custom_providers Name (ADDED 2026-07-19, WAL #1471)
|
||||
|
||||
- `model.provider` MUST be `harness` (the `custom_providers[0].name`), NOT the literal string `custom`
|
||||
- When `provider: custom`, Hermes' `_get_named_custom_provider("custom")` returns None (no provider is
|
||||
named "custom" — it is named "harness"), causing a fall-through to the generic resolution path
|
||||
(`source: env/config`) at `runtime_provider.py:1156`
|
||||
- The generic path builds `api_key_candidates` from `model.api_key` (empty), host-gated
|
||||
OLLAMA/OPENAI/OPENROUTER keys, and `_host_derived_api_key` (returns "" for IP addresses)
|
||||
- **The generic path does NOT resolve `model.api_key_env` or `custom_providers.key_env`** —
|
||||
`LITELLM_API_KEY` is never read, producing `api_key = "no-key-required"` → HTTP 401
|
||||
- The named custom provider path (`source: custom_provider:harness`) DOES read `key_env` —
|
||||
but only triggers when `provider` matches the `custom_providers[0].name`
|
||||
- All sections MUST use `provider: harness`: `model`, `compression`, `auxiliary.vision`,
|
||||
`auxiliary.web_extract`, `auxiliary.compression`, `delegation`
|
||||
- Only `fallback_providers` uses a different provider (`deepseek`) for true fallback diversity
|
||||
- **Diagnostic**: If you see `source: env/config` in a request dump or log, the provider name
|
||||
is wrong. It should be `source: custom_provider:harness`.
|
||||
- **Audit script**: Run `python3 /root/prose-contracts/audit-hermes-config.py <config.yaml>`
|
||||
before and after any config change to catch this and all other rule violations.
|
||||
|
||||
## Execution
|
||||
|
||||
1. **Check current config** — Read the target agent's config.yaml
|
||||
|
||||
@@ -8,18 +8,18 @@ description: >
|
||||
Ensures agents never use the master key directly. Rotation is event-driven,
|
||||
not calendar-driven — rotate only on compromise, personnel change, or
|
||||
periodic security hygiene (quarterly/annually).
|
||||
|
||||
|
||||
UPDATED 2026-07-12: Keys are stored in Infisical vault (project=agents, env=production)
|
||||
BUT each agent host MUST keep a local .env fallback. Infisical service tokens can
|
||||
expire/404. The .env fallback prevents agents from running without keys.
|
||||
Tanko incident: token 404 → gateway had no LITELLM_API_KEY for hours.
|
||||
|
||||
|
||||
UPDATED 2026-07-16: Vault is SYNCED (session-13 keys written to vault via abiba service
|
||||
token, all validate 200). Koby/Koonimo migrated from hardcoded drop-ins to the
|
||||
infisical-gateway.sh wrapper (live vault injection). 4/5 agents now vault-backed.
|
||||
Canonical process: see § Production Vault Access Process. Tanko (user jerome) pending.
|
||||
Abiba's key is now a proper agent key (NOT the master key — stale note removed).
|
||||
|
||||
|
||||
UPDATED 2026-07-17: FLEET-WIDE STANDARDIZATION. All 4 agents (Mumuni, Tanko, Koby, Koonimo)
|
||||
standardized on a single pattern: systemd drop-in (ExecStart= reset + wrapper path) →
|
||||
infisical-gateway.sh while-true loop → /usr/bin/infisical run --token → bash -c key
|
||||
@@ -30,7 +30,7 @@ description: >
|
||||
Critical lessons: (1) NEVER use shell variables inside single-quoted bash -c in wrappers
|
||||
— hardcode absolute paths. (2) Drop-ins override unit file ExecStart permanently.
|
||||
(3) Capture /proc/<pid>/environ before gateway restarts to preserve running env set.
|
||||
|
||||
|
||||
Current key inventory and agent list: see gpu-fleet.prose.md § Agent Keys.
|
||||
Source of truth for LiteLLM config: /opt/inference-harness/litellm_config.yaml
|
||||
on CT 116. Last verified: 2026-07-17.
|
||||
@@ -154,7 +154,7 @@ through its agent wrapper.
|
||||
The `ExecStart=` (empty reset) clears any ExecStart from the main unit file,
|
||||
then the second `ExecStart=` sets the wrapper. This drop-in **survives unit file
|
||||
regeneration** by `hermes gateway install` — the drop-in always wins.
|
||||
|
||||
|
||||
**Why a drop-in instead of editing the unit file:** `hermes gateway install`
|
||||
(called during Hermes updates and some self-heal operations) regenerates the
|
||||
systemd unit file with `ExecStart=/path/to/python -m hermes_cli.main gateway run`.
|
||||
@@ -171,7 +171,7 @@ through its agent wrapper.
|
||||
- **Survives gateway crash**: the wrapper's `while true` + systemd `Restart=always` revive the gateway. Two-layer defense.
|
||||
- **Survives Hermes updates**: systemd drop-in overrides unit file ExecStart — `hermes gateway install` cannot break the vault injection.
|
||||
- **Survives reboot**: systemd user service + `loginctl enable-linger` ensures gateway starts at boot without a login session.
|
||||
- **Auditable**: `cat /proc/$(pgrep hermes_cli)/environ` shows all injected keys; `infisical secrets` shows the vault source.
|
||||
- **Auditable**: `cat /proc/$(pgrep -f 'python.*hermes_cli.main.gateway.run' | grep -v infisical | head -1)/environ` shows all injected keys (note: pipe through grep -v infisical to avoid matching the bash wrapper); `infisical secrets` shows the vault source.
|
||||
|
||||
### Migration status (2026-07-17)
|
||||
|
||||
|
||||
@@ -184,8 +184,8 @@ def check_agents():
|
||||
print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check")
|
||||
continue
|
||||
|
||||
# Gateway process
|
||||
pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | head -1", user=user)
|
||||
# Gateway process (exclude the infisical bash wrapper that contains the same string)
|
||||
pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user)
|
||||
if not pid:
|
||||
print(f" ❌ {name}: GATEWAY NOT RUNNING")
|
||||
FAIL.append(f"gateway-down:{name}")
|
||||
|
||||
Reference in New Issue
Block a user