Merge pull request 'fix(audit): stop requiring the retired gpu-light alias; derive model fields; sweep retired names' (#80) from fix/retired-alias-sweep-20260912 into master
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 1s

This commit was merged in pull request #80.
This commit is contained in:
2026-09-12 17:15:05 +00:00
13 changed files with 350 additions and 77 deletions
+7 -3
View File
@@ -43,17 +43,21 @@ jobs:
echo "=== Prose Contract Frontmatter Validation ==="
FAILED=0
for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do
# NOTE: use herestrings, not `echo "$FM" | grep ...`. Under the runner's
# `-e -o pipefail`, `grep -q` exits on first match and can SIGPIPE the
# producer, making the pipeline report non-zero and raising a false
# "Missing name/description" whose file set varies run to run.
FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d')
[ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; }
KIND=$(echo "$FM" | grep '^kind:' | awk '{print $2}')
KIND=$(grep '^kind:' <<< "$FM" | awk '{print $2}')
case "$KIND" in
function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;;
*) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;;
esac
echo "$FM" | grep -q '^name:' || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
echo "$FM" | grep -q '^description:' || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
grep -q '^name:' <<< "$FM" || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
grep -q '^description:' <<< "$FM" || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
done
[ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; }
echo "✅ Frontmatter validation passed"
+85 -17
View File
@@ -36,6 +36,59 @@ def warn(rule, message):
WARNINGS.append(f"[{rule}] {message}")
# Derivation rule: a model name is any scalar under a mapping key named `model` or
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
# specially. The only other exception is key `models` (litellm key-generation params carry a
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
# hand.
MODEL_KEYS = ("model", "model_name")
MODEL_SECTION_KEYS = ("default", "model", "model_name")
MODEL_LIST_KEYS = ("models",)
def _iter_model_values(node, path=""):
"""Yield (path, value) for every model-name-bearing scalar in a config."""
if isinstance(node, dict):
for key, value in node.items():
child = f"{path}.{key}" if path else key
if key in MODEL_KEYS:
if isinstance(value, dict):
for subkey in MODEL_SECTION_KEYS:
subvalue = value.get(subkey)
if isinstance(subvalue, str):
yield (f"{child}.{subkey}", subvalue)
for subkey, subvalue in value.items():
if isinstance(subvalue, (dict, list)):
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
elif isinstance(value, list):
yield from _iter_model_values(value, child)
else:
yield (child, value)
elif key in MODEL_LIST_KEYS:
yield from _iter_model_list(value, child)
elif isinstance(value, (dict, list)):
yield from _iter_model_values(value, child)
elif isinstance(node, list):
for i, item in enumerate(node):
yield from _iter_model_values(item, f"{path}[{i}]")
def _iter_model_list(node, path):
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
if isinstance(node, list):
for i, item in enumerate(node):
yield from _iter_model_list(item, f"{path}[{i}]")
elif isinstance(node, dict):
for key, value in node.items():
if key in MODEL_KEYS and isinstance(value, str):
yield (f"{path}.{key}", value)
elif isinstance(value, (dict, list)):
yield from _iter_model_list(value, f"{path}.{key}")
else:
yield (path, node)
def audit(path):
with open(path) as f:
cfg = yaml.safe_load(f)
@@ -89,15 +142,16 @@ def audit(path):
)
# --- Rule 8: GPU Workload Distribution ---
# gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision.
check(
aux.get("vision", {}).get("model") == "gpu-light",
aux.get("vision", {}).get("model") == "gpu-vision",
"Rule 8",
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
)
check(
aux.get("web_extract", {}).get("model") == "gpu-light",
aux.get("web_extract", {}).get("model") == "gpu-vision",
"Rule 8",
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
)
# --- Rule 9: Compression Threshold ---
@@ -178,21 +232,35 @@ def audit(path):
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
)
# --- No raw model names (Rule 7/8 spirit) ---
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
for section_path, section_dict in [
("model", model), ("compression", comp),
("auxiliary.vision", aux.get("vision", {})),
("auxiliary.web_extract", aux.get("web_extract", {})),
("auxiliary.compression", aux.get("compression", {})),
("delegation", deleg),
]:
m = section_dict.get("model", "")
if m in raw_names:
# --- Retired/raw model names (Rule 7/8 spirit) ---
# The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
# NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
# gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
# crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
# RESOLVING names (verified 200) are discouraged but working, so they only WARN:
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
# Failing a working alias would reject valid configs - the exact defect this change fixes.
non_resolving = {
"gpu-light": "gpu-vision",
"gemma-4-12b": "gpu-vision",
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
"ornith-1.0-35b": "strix-moe",
}
raw_but_live = {
"qwen3.6-27B-code": "gpu-dense",
"qwen3.6-35B-udq4": "strix-moe",
}
for field_path, value in _iter_model_values(cfg):
if value in non_resolving:
check(
False,
"Rule 7/8",
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
)
elif value in raw_but_live:
warn(
"Rule 7/8",
f"{section_path}.model = {m!r} — raw model name, use stable alias instead "
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
)
# --- Report ---
+8 -8
View File
@@ -216,7 +216,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions.
- **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled.
- **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds.
- **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down.
- **Alias-retirement follow-up (2026-09-12)**: the agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, and the executable `audit-hermes-config.py` still reference the retired `gpu-light`/`gemma-4-12b` names, and `audit-hermes-config.py` Rule 8 currently fails a config whose vision/web_extract model is `gpu-vision`. These are tracked separately and are intentionally NOT updated in this change.
- **Alias-retirement sweep (2026-09-12)**: `gemma-4-12b`, `gpu-light` and `crew-auto` are retired and replaced by `gpu-vision` / no cap respectively. The agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, `gpu-self-heal.prose.md`, `hermes-key-enforcement.prose.md`, `inference-optimization.prose.md`, `litellm-client-timeouts.prose.md` and the executable `audit-hermes-config.py` were all updated to the live canonical alias in the same change. **koby's config on .129 still names `gpu-light` (and `gemma-4-E4B`); .129 is report-only, so that is recorded for its owner and NOT edited here.**
## GPU Inference Benchmarks (Current)
@@ -239,7 +239,7 @@ History stored at `/root/data/toks-history.json` with 7-day rolling window.
### Stable Aliases — CRITICAL
All agent configs MUST use stable role-based aliases, never model-specific names:
- `compression.model: strix-moe`
- `compression.model: syslog-auto`
- `auxiliary.vision.model: gpu-vision`
- `delegation.model: gpu-dense`
- `auxiliary.web_extract.model: gpu-vision`
@@ -249,25 +249,25 @@ When the underlying model is swapped, only the LiteLLM config changes — agent
### Context Windows
- RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K**
- **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek)
- Compression threshold 0.60: fires at ~77K (~51K headroom before 128K ceiling)
- Compression threshold 0.65 (audit Rule 9): fires at ~85K (~43K headroom before 128K ceiling)
- **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K)
- Mumuni compression model alias: `strix-moe`
- Mumuni compression model alias: `syslog-auto`
### Mumuni Agent Profile
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs:
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs. The compression values below are the current required values per template Rules 7/9 and `audit-hermes-config.py`; whether Mumuni's LIVE config currently complies is a separate operational question.
| Setting | Value | Notes |
|---------|-------|-------|
| `model.default` | `syslog-auto` | Balanced default (pool router) |
| `model.provider` | `custom:litellm` | LiteLLM on CT116 |
| `compression.model` | `strix-moe` | Stable alias — survives model swaps |
| `aux.compression.model` | `strix-moe` | Compression auxiliary model |
| `compression.model` | `syslog-auto` | Rule 7: auto-routing, prevents Strix Halo overload |
| `aux.compression.model` | `syslog-auto` | Must match `compression.model` (Rule 7) |
| `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) |
| `aux.web_extract.model` | `gpu-vision` | Web extraction |
| `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) |
| `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling |
| `compression.threshold` | 0.60 | Triggers at ~77K (~60% of 128K) — optimized for 128K context |
| `compression.threshold` | 0.65 | Rule 9: triggers at ~85K for a 128K window |
| `compression.target_ratio` | 0.3 | Compresses to ~38K |
| `compression.protect_last_n` | 40 | Preserves last 40 messages |
| `memory.memory_char_limit` | 800 | Brief memory entries |
+2 -2
View File
@@ -27,8 +27,8 @@ agent: abiba
┌──────┐ ┌──────┐ ┌────────┐
│.8:8080│ │.110 │ │.116:80 │
│RTX3090│ │:8080 │ │nginx │
│gemma │ │RTX5070│ │router │
└──────┘ │qwen27B│ │LiteLLM │
│qwen │ │RTX5070│ │router │
└──────┘ │vision │ │LiteLLM │
└──────┘ │dashboard│
└────────┘
```
+10 -9
View File
@@ -12,7 +12,7 @@ description: >
Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing.
Benchmark baselines refreshed to live values.
Prometheus exporters removed — not deployed; fall back to direct sidecar probes.
Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet.
Stable role-based aliases (strix-moe, gpu-dense, gpu-vision) from gpu-fleet.
agent: abiba
depends_on:
- gpu-monitor.prose.md (live data source on .24:9100)
@@ -52,8 +52,8 @@ depends_on:
Key notes:
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path.
- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated.
- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was.
- Stable aliases (gpu-dense, gpu-vision, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. The retired names `gpu-light` and `gemma-4-12b` were superseded by `gpu-vision` on 2026-09-12 and no longer resolve (400 `Invalid model name`).
- The RTX 5070 is the fastest endpoint per token — `gpu-vision` is its canonical alias. Route vision/web/light work there first. The RTX 5070 model is multimodal (image+text).
- Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads.
- RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom).
- RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes).
@@ -64,7 +64,7 @@ Key notes:
- **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls
- **Fix**:
1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control)
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gemma → qwen, qwen → gemma)
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gpu-vision → gpu-dense, gpu-dense → gpu-vision)
3. If all GPUs hot, alert about cooling infrastructure
- **Verify**: Temp drops below 80°C within 5 minutes
- **Escalate after**: 3 verification failures → Zulip alert
@@ -157,13 +157,14 @@ Key notes:
### Rule 10: Workload Distribution Optimization (updated 2026-07-18)
- **Detect**: GPU roles misaligned with hardware capabilities
- **Target distribution**:
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM).
- RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM).
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM).
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity).
- RTX 5070 (gpu-vision, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks.
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model).
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
- **Weights are not restated here** — the live `syslog-auto` pool weights and rpm caps live in CT 116 `/opt/inference-harness/litellm_config.yaml`, the single source of truth.
- **Fix**:
- Alert if any GPU is handling workload outside its designated role
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe)
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-vision, strix-moe)
- Track per-GPU request distribution via LiteLLM spend logs
- **Verify**: Each GPU's request pattern matches its designated role within 24h
- **Escalate**: If role mismatch persists >48h → agent alias audit needed
@@ -311,6 +312,6 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab
If monitor response > 1MB, log a warning and skip the cycle rather than crashing.
### L6: Stable Aliases Replace Model Names
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15.
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15; `gpu-light` was superseded by `gpu-vision` on 2026-09-12.
- Self-heal must use aliases for reporting and alerting, not model-specific names.
- **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier.
+7 -6
View File
@@ -80,7 +80,7 @@ custom_providers:
auxiliary:
vision:
provider: harness
model: gemma-4-12b # or syslog-auto
model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux)
base_url: http://192.168.68.116/litellm/v1
api_key_env: LITELLM_API_KEY
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
@@ -95,7 +95,7 @@ auxiliary:
threshold: 0.65
target_ratio: 0.3
provider: harness
model: syslog-auto # or gemma-4-12b
model: syslog-auto # Rule 7: compression must be syslog-auto
base_url: http://192.168.68.116/litellm/v1
api_key_env: LITELLM_API_KEY
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
@@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension.
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
**models.json** — Must only list models authorized for the agent's LiteLLM key.
Key is injected via `infisical run --` wrapper at PM2 startup:
`/v1/models` is key-scoped and the live registry is CT 116
`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read
the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup:
```json
{
"providers": {
@@ -197,9 +199,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup:
{ "id": "syslog-auto" },
{ "id": "strix-moe" },
{ "id": "gpu-dense" },
{ "id": "gpu-light" },
{ "id": "qwen3.6-27B-code" },
{ "id": "gemma-4-12b" }
{ "id": "gpu-vision" },
{ "id": "qwen3.6-27B-code" }
]
}
}
+18 -17
View File
@@ -102,7 +102,7 @@ model:
# and falls back to 256K when /v1/models lacks a context
# field (llama-server does). Without this override, agents
# silently run syslog-auto at 256K (verified 2026-08-09).
# Set 65536 if using gemma-4-12b directly (tight VRAM).
# Set 65536 if pinning a single model directly (tight VRAM).
fallback_providers:
provider: deepseek
@@ -143,25 +143,25 @@ compression:
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
# All auxiliary services MUST use identical model, base_url, and api_key_env:
# model: gpu-light # stable alias (NOT raw "gemma-4-12b")
# model: gpu-vision # stable alias (NOT a raw model name)
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
# api_key_env: LITELLM_API_KEY
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4)
# in agent configs — use the stable aliases so model swaps don't break agents.
# NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
auxiliary:
vision:
provider: harness
model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b)
model: gpu-vision # stable alias for RTX 5070
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
api_key_env: LITELLM_API_KEY
timeout: 60
download_timeout: 30
web_extract:
provider: harness
model: gpu-light # stable alias for RTX 5070
model: gpu-vision # stable alias for RTX 5070
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
api_key_env: LITELLM_API_KEY
timeout: 30
@@ -252,10 +252,11 @@ The following MUST be identical across ALL profiles:
- For agents needing longer outputs: raise to 8192, but never omit
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized)
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized)
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b`
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
- Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized)
- Compression uses `syslog-auto` — the Strix Halo weighted pool (64GB, 128K ctx, compression-optimized); do NOT pin `compression.model` to `strix-moe` (audit Rule 7 rejects it)
- **`ornith-1.0-35b` is NOT a valid compression model name** — LiteLLM does not serve it
(do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml`
is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
@@ -264,7 +265,7 @@ The following MUST be identical across ALL profiles:
- All auxiliary services MUST use identical routing:
- `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK)
- `api_key_env: LITELLM_API_KEY`
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably
- **Do NOT use `syslog-auto` for `vision`/`web_extract`** — it routes unpredictably; compression is the deliberate exception (see the OPERATIONAL DECISION above)
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
(64GB UMA, 128K context) — the designated compression GPU. This frees the
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
@@ -273,11 +274,11 @@ The following MUST be identical across ALL profiles:
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
- **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
- Agent profiles MUST route auxiliary tasks to the correct GPU:
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070)
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070)
- `auxiliary.vision.model: gpu-vision` (RTX 5070)
- `auxiliary.web_extract.model: gpu-vision` (RTX 5070)
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
@@ -297,8 +298,8 @@ The following MUST be identical across ALL profiles:
- **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10.
- **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto`
- **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json`
- `syslog-auto` is the LiteLLM routing model — it load-balances between strix-moe
and qwen3.6-27B-code, with gemma-4-12b as fallback. Using it protects against:
- `syslog-auto` is the LiteLLM routing model — it load-balances across the live pool
(see CT 116 `/opt/inference-harness/litellm_config.yaml` for the current members and weights). Using it protects against:
- Model name typos that cause 403 errors and silent worker failures
- Single GPU downtime (routing falls back automatically)
- Key/model authorization mismatches
+9 -5
View File
@@ -169,16 +169,20 @@ The agent picks up the new key via `infisical run --` at gateway startup.
**Keys are permanent and use bare agent name aliases.**
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`.
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
- **Max budget**: $100 per key (config default).
```yaml
# In litellm_config.yaml — ensures all future keys inherit these defaults:
# NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no
# default_key_generate_params block today, and a key generated with no explicit models comes back
# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to
# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and
# re-verify before applying.
litellm_settings:
default_key_generate_params:
models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b"]
models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
duration: null # ← permanent
max_budget: 100
metadata:
@@ -286,13 +290,13 @@ auxiliary:
api_key: sk-<agent-key-from-vault> # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain)
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1
model: gemma-4-12b
model: gpu-vision
provider: harness
compression:
api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1
model: gemma-4-12b
model: syslog-auto
provider: harness
```
+6 -4
View File
@@ -23,7 +23,7 @@ management, and prompt caching — without sacrificing agent capability.
.123, any others on .129/.122) including compression, model, context_window,
prompt_caching, memory settings
- `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080,
qwen .8:8080, gemma .110:8080)
gpu-dense .8:8080, gpu-vision .110:8080)
### Maintains
@@ -56,8 +56,8 @@ duration.
**Context is the root cause.** Every ~46K prompt token costs ~87s of
prefill time at 532 tok/s. Fix context first, routing second.
- **Route by task**: qwen for code/standard queries; gemma for
compression/auxiliary; strix-moe for compression tasks.
- **Route by task**: gpu-dense for code/standard queries; gpu-vision for
vision/web-auxiliary; syslog-auto for compression.
- **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should
compact at 51K, not 85K. Target 15% tail (not 30%).
- **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these
@@ -96,8 +96,10 @@ call enable-prompt-caching
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
-- Phase 5: Verify end-to-end latency
-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is
-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use.
call verify-latency
host: 192.168.68.116
models: [syslog-auto, qwen3.6-27B-code, gemma-4-12b]
models: [syslog-auto, qwen3.6-27B-code, gpu-vision]
```
+1 -1
View File
@@ -222,7 +222,7 @@ description: >
**Prometheus targets**:
- 192.168.68.8:9400 (RTX 3090 — qwen)
- 192.168.68.110:9400 (RTX 5070 — gemma)
- 192.168.68.110:9400 (RTX 5070 — gpu-vision)
- 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4)
- harness-litellm:4000 (LiteLLM health)
+4 -2
View File
@@ -67,8 +67,10 @@ description: >
4. **If action == "create"**:
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
- Duration is null (permanent) — inherited from litellm default_key_generate_params
- Set models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b", "strix-moe", "gpu-dense", "gpu-light", "qwen3.6-35B-udq4"]
- Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
- Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
- Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed).
- Return the new key
5. **If action == "rotate"**:
+2 -3
View File
@@ -28,7 +28,7 @@ description: >
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
| strix-moe | 7.5s | — | Strix Halo, healthy |
| gemma-4-12b | 2.6s | — | RTX 5070, healthy |
| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
backend), full recovery 07:00-08:00 with ZERO client failures once requests
@@ -53,8 +53,7 @@ proxy queuing.
### 2. Auxiliary tasks — keep template timeouts, one correction
- vision: 60s (keep), web_extract: 30s (keep) — gemma-4-12b averages 2.6s;
these are fine.
- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
+191
View File
@@ -0,0 +1,191 @@
"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
observable behaviour — exit code and the emitted rule message — for the live alias and
for both retired names. No network, vault, or SSH access is required.
"""
from __future__ import annotations
import pathlib
import subprocess
import sys
ROOT = pathlib.Path(__file__).resolve().parent.parent
AUDIT = ROOT / "audit-hermes-config.py"
BASE = """
model:
api_key: ""
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/v1
max_tokens: 4096
default: syslog-auto
provider: harness
fallback_providers:
provider: deepseek
model: deepseek-v4-flash
api_key_env: DEEPSEEK_API_KEY
compression:
model: syslog-auto
provider: harness
threshold: 0.65
max_context_window: 131072
auxiliary:
vision:
model: {alias}
provider: harness
web_extract:
model: {alias}
provider: harness
compression:
model: syslog-auto
provider: harness
delegation:
provider: harness
custom_providers:
- name: harness
key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/v1
"""
def _run_config(tmp_path, name, text):
cfg = tmp_path / name
cfg.write_text(text)
proc = subprocess.run(
[sys.executable, str(AUDIT), str(cfg)],
capture_output=True, text=True,
)
return proc.returncode, proc.stdout
def _run(tmp_path, alias):
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
def test_live_canonical_alias_passes(tmp_path):
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "RESULT: PASS" in out
def test_retired_gpu_light_is_rejected(tmp_path):
"""A config pinned to the retired alias must fail, not pass."""
code, out = _run(tmp_path, "gpu-light")
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_retired_gemma_is_rejected(tmp_path):
"""The retired raw model name must fail Rule 8 as well."""
code, out = _run(tmp_path, "gemma-4-12b")
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_corrected_compression_example_passes(tmp_path):
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "RESULT: PASS" in out
def test_retired_alias_in_delegation_is_rejected(tmp_path):
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
code, out = _run_config(
tmp_path,
"delegation-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: gpu-light",
),
)
assert code == 1, out
assert "delegation.model = 'gpu-light' is retired" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"custom-provider-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" - name: harness\n key_env: LITELLM_API_KEY",
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
),
)
assert code == 1, out
assert "custom_providers[0].model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_raw_but_live_alias_warns_but_passes(tmp_path):
"""Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs."""
code, out = _run_config(
tmp_path,
"raw-qwen.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
),
)
assert code == 0, out
assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out
assert "prefer the stable alias gpu-dense" in out
assert "RESULT: PASS" in out
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
"""fallback_providers.model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"fallback-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
)
assert code == 1, out
assert "fallback_providers.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_x_search_is_rejected(tmp_path):
"""x_search.model was previously not enumerated; the derivation must catch it."""
code, out = _run_config(
tmp_path,
"x-search-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
),
)
assert code == 1, out
assert "x_search.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
"""A nested auxiliary sub-block outside the named three must still be derived."""
code, out = _run_config(
tmp_path,
"nested-aux-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
" compression:\n model: syslog-auto\n provider: harness\n"
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
),
)
assert code == 1, out
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out