|
|
|
@@ -96,7 +96,7 @@ work immediately after restart.
|
|
|
|
|
model:
|
|
|
|
|
default: <agent_model> # e.g., strix-moe, qwen3.6-27B-code, syslog-auto
|
|
|
|
|
provider: harness
|
|
|
|
|
base_url: http://192.168.68.116/v1
|
|
|
|
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated path; /v1 also OK
|
|
|
|
|
api_key_env: LITELLM_API_KEY # Injected via infisical run -- wrapper
|
|
|
|
|
max_tokens: 4096 # ⚠️ CRITICAL: Prevents unbounded generation
|
|
|
|
|
context_length: 131072 # For syslog-auto (all GPUs at 128K for stability).
|
|
|
|
@@ -146,7 +146,7 @@ compression:
|
|
|
|
|
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
|
|
|
|
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
|
|
|
|
# model: gpu-light # stable alias (NOT raw "gemma-4-12b")
|
|
|
|
|
# base_url: http://192.168.68.116/v1
|
|
|
|
|
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
|
|
|
|
# api_key_env: LITELLM_API_KEY
|
|
|
|
|
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
|
|
|
|
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
|
|
|
@@ -157,20 +157,20 @@ auxiliary:
|
|
|
|
|
vision:
|
|
|
|
|
provider: harness
|
|
|
|
|
model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b)
|
|
|
|
|
base_url: http://192.168.68.116/v1
|
|
|
|
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
|
|
|
|
api_key_env: LITELLM_API_KEY
|
|
|
|
|
timeout: 60
|
|
|
|
|
download_timeout: 30
|
|
|
|
|
web_extract:
|
|
|
|
|
provider: harness
|
|
|
|
|
model: gpu-light # stable alias for RTX 5070
|
|
|
|
|
base_url: http://192.168.68.116/v1
|
|
|
|
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
|
|
|
|
api_key_env: LITELLM_API_KEY
|
|
|
|
|
timeout: 30
|
|
|
|
|
compression:
|
|
|
|
|
provider: harness
|
|
|
|
|
model: syslog-auto # MUST match compression.model above. Stable alias for Strix Halo (weighted pool).
|
|
|
|
|
base_url: http://192.168.68.116/v1 # Rule 5: /v1 NOT /litellm/v1
|
|
|
|
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated path; /v1 also OK
|
|
|
|
|
api_key_env: LITELLM_API_KEY
|
|
|
|
|
timeout: 300 # gpu-fleet: 300s for large-history summarization (was 60)
|
|
|
|
|
|
|
|
|
@@ -180,14 +180,14 @@ auxiliary:
|
|
|
|
|
delegation:
|
|
|
|
|
model: gpu-dense # stable alias for RTX 3090 (was raw qwen3.6-27B-code)
|
|
|
|
|
provider: harness
|
|
|
|
|
base_url: http://192.168.68.116/v1
|
|
|
|
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
|
|
|
|
api_key_env: LITELLM_API_KEY
|
|
|
|
|
|
|
|
|
|
# ─── Custom Provider ───
|
|
|
|
|
custom_providers:
|
|
|
|
|
- name: harness
|
|
|
|
|
model: syslog-auto # weighted pool (default)
|
|
|
|
|
base_url: http://192.168.68.116/v1
|
|
|
|
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
|
|
|
|
api_key_env: LITELLM_API_KEY
|
|
|
|
|
api_mode: chat_completions
|
|
|
|
|
```
|
|
|
|
@@ -237,10 +237,13 @@ The following MUST be identical across ALL profiles:
|
|
|
|
|
- When main config uses `api_key_env`, sub-agents automatically use it
|
|
|
|
|
- This means key rotation only touches ONE vault secret (`LITELLM_API_KEY`)
|
|
|
|
|
|
|
|
|
|
### Rule 5: Main Config Base URL
|
|
|
|
|
|- Use direct IP: `http://192.168.68.116/v1`
|
|
|
|
|
### Rule 5: Main Config Base URL (UPDATED 2026-08-09)
|
|
|
|
|
|- Use the authenticated LiteLLM path: `http://192.168.68.116/litellm/v1` (canonical, captain-approved migration)
|
|
|
|
|
|- Legacy `http://192.68.68.116/v1` also works — nginx fronts BOTH paths with key auth
|
|
|
|
|
(verified 2026-08-09: 401 without key, 200 with key, on both /v1 and /litellm/v1)
|
|
|
|
|
|- Both locations have `proxy_read_timeout 600s` (verified in harness-nginx nginx.conf) —
|
|
|
|
|
the old "60s timeout on /litellm/" claim was stale and is retracted
|
|
|
|
|
|- NOT the NetBird URL (`litellm.sysloggh.net`) — can cause 502 when NetBird is down
|
|
|
|
|
|- NOT the old path (`/litellm/v1`) — nginx now routes `/v1` directly
|
|
|
|
|
|
|
|
|
|
### Rule 6: max_tokens Is Required (Thermal Safety)
|
|
|
|
|
- **Every Hermes config MUST set `model.max_tokens: 4096`** — this is non-negotiable
|
|
|
|
@@ -261,7 +264,7 @@ The following MUST be identical across ALL profiles:
|
|
|
|
|
fall back to other GPUs if Strix gets hot. Both `compression.model` and `auxiliary.compression.model`
|
|
|
|
|
MUST be `syslog-auto`.
|
|
|
|
|
- All auxiliary services MUST use identical routing:
|
|
|
|
|
- `base_url: http://192.168.68.116/v1` (Rule 5: `/v1`, NOT `/litellm/v1`)
|
|
|
|
|
- `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK)
|
|
|
|
|
- `api_key_env: LITELLM_API_KEY`
|
|
|
|
|
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably
|
|
|
|
|
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
|
|
|
|
@@ -321,10 +324,13 @@ verify ALL FOUR of these against the live config. They are the only root causes
|
|
|
|
|
1. **max_context_window correct?** — BOTH `compression.max_context_window` AND `context.max_context_window`
|
|
|
|
|
MUST be `131072` (all GPUs are 128K). A value of `262144` causes instability near 100K and must NOT be used.
|
|
|
|
|
~83K instead of ~170K. Check: `grep -n max_context_window ~/.hermes/config.yaml`
|
|
|
|
|
2. **base_url uses /v1 NOT /litellm/v1?** — `custom_providers[0].base_url`, `delegation.base_url`,
|
|
|
|
|
and ALL `auxiliary.*.base_url` MUST be `http://192.168.68.116/v1` (Rule 5). nginx `/litellm/`
|
|
|
|
|
has a 60s default timeout → 504 on any inference >60s; `/v1/` has 600s.
|
|
|
|
|
Check: `grep -n 'litellm/v1' ~/.hermes/config.yaml` (must return NOTHING)
|
|
|
|
|
2. **base_url uses authenticated path?** — `custom_providers[0].base_url`, `delegation.base_url`,
|
|
|
|
|
and ALL `auxiliary.*.base_url` MUST be `http://192.168.68.116/litellm/v1` (Rule 5, canonical)
|
|
|
|
|
or `http://192.168.68.116/v1` (legacy, still authenticated via nginx). BOTH verified 200 with
|
|
|
|
|
key + 600s proxy_read_timeout on 2026-08-09. Never bare `:4000` direct.
|
|
|
|
|
Check: `grep -nE 'base_url: http://192.168.68.116(:4000)?/v1' ~/.hermes/config.yaml` — the
|
|
|
|
|
ONLY paths allowed are `/v1` or `/litellm/v1` (both via nginx :80).
|
|
|
|
|
`:4000` or missing `litellm/v1`/`v1` prefix = violation.
|
|
|
|
|
3. **LITELLM_API_KEY valid?** — The key must be a real LiteLLM key (`sk-` + 64 hex, 67 chars).
|
|
|
|
|
Malformed values (e.g. `sk-_SWAl_Vu_…`, 47 chars) return 401 → DeepSeek fallback.
|
|
|
|
|
Verify: `curl -s -o /dev/null -w '%{http_code}' -H "Authorization: Bearer $LITELLM_API_KEY" http://192.168.68.116/v1/models` (must be 200)
|
|
|
|
|