fix(audit): stop requiring the retired gpu-light alias; derive model fields; sweep retired names #80
@@ -43,17 +43,21 @@ jobs:
|
||||
echo "=== Prose Contract Frontmatter Validation ==="
|
||||
FAILED=0
|
||||
for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do
|
||||
# NOTE: use herestrings, not `echo "$FM" | grep ...`. Under the runner's
|
||||
# `-e -o pipefail`, `grep -q` exits on first match and can SIGPIPE the
|
||||
# producer, making the pipeline report non-zero and raising a false
|
||||
# "Missing name/description" whose file set varies run to run.
|
||||
FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d')
|
||||
[ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; }
|
||||
|
||||
KIND=$(echo "$FM" | grep '^kind:' | awk '{print $2}')
|
||||
KIND=$(grep '^kind:' <<< "$FM" | awk '{print $2}')
|
||||
case "$KIND" in
|
||||
function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;;
|
||||
*) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;;
|
||||
esac
|
||||
|
||||
echo "$FM" | grep -q '^name:' || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
|
||||
echo "$FM" | grep -q '^description:' || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
|
||||
grep -q '^name:' <<< "$FM" || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
|
||||
grep -q '^description:' <<< "$FM" || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
|
||||
done
|
||||
[ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; }
|
||||
echo "✅ Frontmatter validation passed"
|
||||
|
||||
+85
-17
@@ -36,6 +36,59 @@ def warn(rule, message):
|
||||
WARNINGS.append(f"[{rule}] {message}")
|
||||
|
||||
|
||||
# Derivation rule: a model name is any scalar under a mapping key named `model` or
|
||||
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
|
||||
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
|
||||
# specially. The only other exception is key `models` (litellm key-generation params carry a
|
||||
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
|
||||
# hand.
|
||||
MODEL_KEYS = ("model", "model_name")
|
||||
MODEL_SECTION_KEYS = ("default", "model", "model_name")
|
||||
MODEL_LIST_KEYS = ("models",)
|
||||
|
||||
|
||||
def _iter_model_values(node, path=""):
|
||||
"""Yield (path, value) for every model-name-bearing scalar in a config."""
|
||||
if isinstance(node, dict):
|
||||
for key, value in node.items():
|
||||
child = f"{path}.{key}" if path else key
|
||||
if key in MODEL_KEYS:
|
||||
if isinstance(value, dict):
|
||||
for subkey in MODEL_SECTION_KEYS:
|
||||
subvalue = value.get(subkey)
|
||||
if isinstance(subvalue, str):
|
||||
yield (f"{child}.{subkey}", subvalue)
|
||||
for subkey, subvalue in value.items():
|
||||
if isinstance(subvalue, (dict, list)):
|
||||
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
|
||||
elif isinstance(value, list):
|
||||
yield from _iter_model_values(value, child)
|
||||
else:
|
||||
yield (child, value)
|
||||
elif key in MODEL_LIST_KEYS:
|
||||
yield from _iter_model_list(value, child)
|
||||
elif isinstance(value, (dict, list)):
|
||||
yield from _iter_model_values(value, child)
|
||||
elif isinstance(node, list):
|
||||
for i, item in enumerate(node):
|
||||
yield from _iter_model_values(item, f"{path}[{i}]")
|
||||
|
||||
|
||||
def _iter_model_list(node, path):
|
||||
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
|
||||
if isinstance(node, list):
|
||||
for i, item in enumerate(node):
|
||||
yield from _iter_model_list(item, f"{path}[{i}]")
|
||||
elif isinstance(node, dict):
|
||||
for key, value in node.items():
|
||||
if key in MODEL_KEYS and isinstance(value, str):
|
||||
yield (f"{path}.{key}", value)
|
||||
elif isinstance(value, (dict, list)):
|
||||
yield from _iter_model_list(value, f"{path}.{key}")
|
||||
else:
|
||||
yield (path, node)
|
||||
|
||||
|
||||
def audit(path):
|
||||
with open(path) as f:
|
||||
cfg = yaml.safe_load(f)
|
||||
@@ -89,15 +142,16 @@ def audit(path):
|
||||
)
|
||||
|
||||
# --- Rule 8: GPU Workload Distribution ---
|
||||
# gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision.
|
||||
check(
|
||||
aux.get("vision", {}).get("model") == "gpu-light",
|
||||
aux.get("vision", {}).get("model") == "gpu-vision",
|
||||
"Rule 8",
|
||||
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
check(
|
||||
aux.get("web_extract", {}).get("model") == "gpu-light",
|
||||
aux.get("web_extract", {}).get("model") == "gpu-vision",
|
||||
"Rule 8",
|
||||
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||
)
|
||||
|
||||
# --- Rule 9: Compression Threshold ---
|
||||
@@ -178,21 +232,35 @@ def audit(path):
|
||||
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
||||
)
|
||||
|
||||
# --- No raw model names (Rule 7/8 spirit) ---
|
||||
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
|
||||
for section_path, section_dict in [
|
||||
("model", model), ("compression", comp),
|
||||
("auxiliary.vision", aux.get("vision", {})),
|
||||
("auxiliary.web_extract", aux.get("web_extract", {})),
|
||||
("auxiliary.compression", aux.get("compression", {})),
|
||||
("delegation", deleg),
|
||||
]:
|
||||
m = section_dict.get("model", "")
|
||||
if m in raw_names:
|
||||
# --- Retired/raw model names (Rule 7/8 spirit) ---
|
||||
# The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
|
||||
# NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
|
||||
# gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
|
||||
# crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
|
||||
# RESOLVING names (verified 200) are discouraged but working, so they only WARN:
|
||||
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
|
||||
# Failing a working alias would reject valid configs - the exact defect this change fixes.
|
||||
non_resolving = {
|
||||
"gpu-light": "gpu-vision",
|
||||
"gemma-4-12b": "gpu-vision",
|
||||
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
|
||||
"ornith-1.0-35b": "strix-moe",
|
||||
}
|
||||
raw_but_live = {
|
||||
"qwen3.6-27B-code": "gpu-dense",
|
||||
"qwen3.6-35B-udq4": "strix-moe",
|
||||
}
|
||||
for field_path, value in _iter_model_values(cfg):
|
||||
if value in non_resolving:
|
||||
check(
|
||||
False,
|
||||
"Rule 7/8",
|
||||
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
|
||||
)
|
||||
elif value in raw_but_live:
|
||||
warn(
|
||||
"Rule 7/8",
|
||||
f"{section_path}.model = {m!r} — raw model name, use stable alias instead "
|
||||
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
|
||||
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
|
||||
)
|
||||
|
||||
# --- Report ---
|
||||
|
||||
+8
-8
@@ -216,7 +216,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions.
|
||||
- **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled.
|
||||
- **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds.
|
||||
- **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down.
|
||||
- **Alias-retirement follow-up (2026-09-12)**: the agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, and the executable `audit-hermes-config.py` still reference the retired `gpu-light`/`gemma-4-12b` names, and `audit-hermes-config.py` Rule 8 currently fails a config whose vision/web_extract model is `gpu-vision`. These are tracked separately and are intentionally NOT updated in this change.
|
||||
- **Alias-retirement sweep (2026-09-12)**: `gemma-4-12b`, `gpu-light` and `crew-auto` are retired and replaced by `gpu-vision` / no cap respectively. The agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, `gpu-self-heal.prose.md`, `hermes-key-enforcement.prose.md`, `inference-optimization.prose.md`, `litellm-client-timeouts.prose.md` and the executable `audit-hermes-config.py` were all updated to the live canonical alias in the same change. **koby's config on .129 still names `gpu-light` (and `gemma-4-E4B`); .129 is report-only, so that is recorded for its owner and NOT edited here.**
|
||||
|
||||
## GPU Inference Benchmarks (Current)
|
||||
|
||||
@@ -239,7 +239,7 @@ History stored at `/root/data/toks-history.json` with 7-day rolling window.
|
||||
### Stable Aliases — CRITICAL
|
||||
|
||||
All agent configs MUST use stable role-based aliases, never model-specific names:
|
||||
- `compression.model: strix-moe`
|
||||
- `compression.model: syslog-auto`
|
||||
- `auxiliary.vision.model: gpu-vision`
|
||||
- `delegation.model: gpu-dense`
|
||||
- `auxiliary.web_extract.model: gpu-vision`
|
||||
@@ -249,25 +249,25 @@ When the underlying model is swapped, only the LiteLLM config changes — agent
|
||||
### Context Windows
|
||||
- RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K**
|
||||
- **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek)
|
||||
- Compression threshold 0.60: fires at ~77K (~51K headroom before 128K ceiling)
|
||||
- Compression threshold 0.65 (audit Rule 9): fires at ~85K (~43K headroom before 128K ceiling)
|
||||
- **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K)
|
||||
- Mumuni compression model alias: `strix-moe`
|
||||
- Mumuni compression model alias: `syslog-auto`
|
||||
|
||||
### Mumuni Agent Profile
|
||||
|
||||
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs:
|
||||
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs. The compression values below are the current required values per template Rules 7/9 and `audit-hermes-config.py`; whether Mumuni's LIVE config currently complies is a separate operational question.
|
||||
|
||||
| Setting | Value | Notes |
|
||||
|---------|-------|-------|
|
||||
| `model.default` | `syslog-auto` | Balanced default (pool router) |
|
||||
| `model.provider` | `custom:litellm` | LiteLLM on CT116 |
|
||||
| `compression.model` | `strix-moe` | Stable alias — survives model swaps |
|
||||
| `aux.compression.model` | `strix-moe` | Compression auxiliary model |
|
||||
| `compression.model` | `syslog-auto` | Rule 7: auto-routing, prevents Strix Halo overload |
|
||||
| `aux.compression.model` | `syslog-auto` | Must match `compression.model` (Rule 7) |
|
||||
| `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) |
|
||||
| `aux.web_extract.model` | `gpu-vision` | Web extraction |
|
||||
| `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) |
|
||||
| `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling |
|
||||
| `compression.threshold` | 0.60 | Triggers at ~77K (~60% of 128K) — optimized for 128K context |
|
||||
| `compression.threshold` | 0.65 | Rule 9: triggers at ~85K for a 128K window |
|
||||
| `compression.target_ratio` | 0.3 | Compresses to ~38K |
|
||||
| `compression.protect_last_n` | 40 | Preserves last 40 messages |
|
||||
| `memory.memory_char_limit` | 800 | Brief memory entries |
|
||||
|
||||
@@ -27,8 +27,8 @@ agent: abiba
|
||||
┌──────┐ ┌──────┐ ┌────────┐
|
||||
│.8:8080│ │.110 │ │.116:80 │
|
||||
│RTX3090│ │:8080 │ │nginx │
|
||||
│gemma │ │RTX5070│ │router │
|
||||
└──────┘ │qwen27B│ │LiteLLM │
|
||||
│qwen │ │RTX5070│ │router │
|
||||
└──────┘ │vision │ │LiteLLM │
|
||||
└──────┘ │dashboard│
|
||||
└────────┘
|
||||
```
|
||||
|
||||
+10
-9
@@ -12,7 +12,7 @@ description: >
|
||||
Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing.
|
||||
Benchmark baselines refreshed to live values.
|
||||
Prometheus exporters removed — not deployed; fall back to direct sidecar probes.
|
||||
Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet.
|
||||
Stable role-based aliases (strix-moe, gpu-dense, gpu-vision) from gpu-fleet.
|
||||
agent: abiba
|
||||
depends_on:
|
||||
- gpu-monitor.prose.md (live data source on .24:9100)
|
||||
@@ -52,8 +52,8 @@ depends_on:
|
||||
|
||||
Key notes:
|
||||
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path.
|
||||
- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated.
|
||||
- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was.
|
||||
- Stable aliases (gpu-dense, gpu-vision, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. The retired names `gpu-light` and `gemma-4-12b` were superseded by `gpu-vision` on 2026-09-12 and no longer resolve (400 `Invalid model name`).
|
||||
- The RTX 5070 is the fastest endpoint per token — `gpu-vision` is its canonical alias. Route vision/web/light work there first. The RTX 5070 model is multimodal (image+text).
|
||||
- Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads.
|
||||
- RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom).
|
||||
- RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes).
|
||||
@@ -64,7 +64,7 @@ Key notes:
|
||||
- **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls
|
||||
- **Fix**:
|
||||
1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control)
|
||||
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gemma → qwen, qwen → gemma)
|
||||
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gpu-vision → gpu-dense, gpu-dense → gpu-vision)
|
||||
3. If all GPUs hot, alert about cooling infrastructure
|
||||
- **Verify**: Temp drops below 80°C within 5 minutes
|
||||
- **Escalate after**: 3 verification failures → Zulip alert
|
||||
@@ -157,13 +157,14 @@ Key notes:
|
||||
### Rule 10: Workload Distribution Optimization (updated 2026-07-18)
|
||||
- **Detect**: GPU roles misaligned with hardware capabilities
|
||||
- **Target distribution**:
|
||||
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM).
|
||||
- RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM).
|
||||
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM).
|
||||
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity).
|
||||
- RTX 5070 (gpu-vision, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks.
|
||||
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model).
|
||||
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
|
||||
- **Weights are not restated here** — the live `syslog-auto` pool weights and rpm caps live in CT 116 `/opt/inference-harness/litellm_config.yaml`, the single source of truth.
|
||||
- **Fix**:
|
||||
- Alert if any GPU is handling workload outside its designated role
|
||||
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe)
|
||||
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-vision, strix-moe)
|
||||
- Track per-GPU request distribution via LiteLLM spend logs
|
||||
- **Verify**: Each GPU's request pattern matches its designated role within 24h
|
||||
- **Escalate**: If role mismatch persists >48h → agent alias audit needed
|
||||
@@ -311,6 +312,6 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab
|
||||
If monitor response > 1MB, log a warning and skip the cycle rather than crashing.
|
||||
|
||||
### L6: Stable Aliases Replace Model Names
|
||||
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15.
|
||||
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15; `gpu-light` was superseded by `gpu-vision` on 2026-09-12.
|
||||
- Self-heal must use aliases for reporting and alerting, not model-specific names.
|
||||
- **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier.
|
||||
|
||||
@@ -80,7 +80,7 @@ custom_providers:
|
||||
auxiliary:
|
||||
vision:
|
||||
provider: harness
|
||||
model: gemma-4-12b # or syslog-auto
|
||||
model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux)
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
api_key_env: LITELLM_API_KEY
|
||||
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
||||
@@ -95,7 +95,7 @@ auxiliary:
|
||||
threshold: 0.65
|
||||
target_ratio: 0.3
|
||||
provider: harness
|
||||
model: syslog-auto # or gemma-4-12b
|
||||
model: syslog-auto # Rule 7: compression must be syslog-auto
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
api_key_env: LITELLM_API_KEY
|
||||
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
||||
@@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension.
|
||||
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
|
||||
|
||||
**models.json** — Must only list models authorized for the agent's LiteLLM key.
|
||||
Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||
`/v1/models` is key-scoped and the live registry is CT 116
|
||||
`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read
|
||||
the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||
```json
|
||||
{
|
||||
"providers": {
|
||||
@@ -197,9 +199,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||
{ "id": "syslog-auto" },
|
||||
{ "id": "strix-moe" },
|
||||
{ "id": "gpu-dense" },
|
||||
{ "id": "gpu-light" },
|
||||
{ "id": "qwen3.6-27B-code" },
|
||||
{ "id": "gemma-4-12b" }
|
||||
{ "id": "gpu-vision" },
|
||||
{ "id": "qwen3.6-27B-code" }
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -102,7 +102,7 @@ model:
|
||||
# and falls back to 256K when /v1/models lacks a context
|
||||
# field (llama-server does). Without this override, agents
|
||||
# silently run syslog-auto at 256K (verified 2026-08-09).
|
||||
# Set 65536 if using gemma-4-12b directly (tight VRAM).
|
||||
# Set 65536 if pinning a single model directly (tight VRAM).
|
||||
|
||||
fallback_providers:
|
||||
provider: deepseek
|
||||
@@ -143,25 +143,25 @@ compression:
|
||||
|
||||
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
||||
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
||||
# model: gpu-light # stable alias (NOT raw "gemma-4-12b")
|
||||
# model: gpu-vision # stable alias (NOT a raw model name)
|
||||
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||
# api_key_env: LITELLM_API_KEY
|
||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||
# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4)
|
||||
# in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
# NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
|
||||
# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
auxiliary:
|
||||
vision:
|
||||
provider: harness
|
||||
model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b)
|
||||
model: gpu-vision # stable alias for RTX 5070
|
||||
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||
api_key_env: LITELLM_API_KEY
|
||||
timeout: 60
|
||||
download_timeout: 30
|
||||
web_extract:
|
||||
provider: harness
|
||||
model: gpu-light # stable alias for RTX 5070
|
||||
model: gpu-vision # stable alias for RTX 5070
|
||||
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||
api_key_env: LITELLM_API_KEY
|
||||
timeout: 30
|
||||
@@ -252,10 +252,11 @@ The following MUST be identical across ALL profiles:
|
||||
- For agents needing longer outputs: raise to 8192, but never omit
|
||||
|
||||
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
|
||||
- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized)
|
||||
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized)
|
||||
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b`
|
||||
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
||||
- Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized)
|
||||
- Compression uses `syslog-auto` — the Strix Halo weighted pool (64GB, 128K ctx, compression-optimized); do NOT pin `compression.model` to `strix-moe` (audit Rule 7 rejects it)
|
||||
- **`ornith-1.0-35b` is NOT a valid compression model name** — LiteLLM does not serve it
|
||||
(do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml`
|
||||
is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
||||
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
|
||||
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
|
||||
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
|
||||
@@ -264,7 +265,7 @@ The following MUST be identical across ALL profiles:
|
||||
- All auxiliary services MUST use identical routing:
|
||||
- `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK)
|
||||
- `api_key_env: LITELLM_API_KEY`
|
||||
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably
|
||||
- **Do NOT use `syslog-auto` for `vision`/`web_extract`** — it routes unpredictably; compression is the deliberate exception (see the OPERATIONAL DECISION above)
|
||||
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
|
||||
(64GB UMA, 128K context) — the designated compression GPU. This frees the
|
||||
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
|
||||
@@ -273,11 +274,11 @@ The following MUST be identical across ALL profiles:
|
||||
|
||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
|
||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
|
||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
||||
- **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
||||
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
|
||||
- Agent profiles MUST route auxiliary tasks to the correct GPU:
|
||||
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070)
|
||||
- `auxiliary.vision.model: gpu-vision` (RTX 5070)
|
||||
- `auxiliary.web_extract.model: gpu-vision` (RTX 5070)
|
||||
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
|
||||
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
|
||||
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
|
||||
@@ -297,8 +298,8 @@ The following MUST be identical across ALL profiles:
|
||||
- **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10.
|
||||
- **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto`
|
||||
- **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json`
|
||||
- `syslog-auto` is the LiteLLM routing model — it load-balances between strix-moe
|
||||
and qwen3.6-27B-code, with gemma-4-12b as fallback. Using it protects against:
|
||||
- `syslog-auto` is the LiteLLM routing model — it load-balances across the live pool
|
||||
(see CT 116 `/opt/inference-harness/litellm_config.yaml` for the current members and weights). Using it protects against:
|
||||
- Model name typos that cause 403 errors and silent worker failures
|
||||
- Single GPU downtime (routing falls back automatically)
|
||||
- Key/model authorization mismatches
|
||||
|
||||
@@ -169,16 +169,20 @@ The agent picks up the new key via `infisical run --` at gateway startup.
|
||||
|
||||
**Keys are permanent and use bare agent name aliases.**
|
||||
|
||||
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`.
|
||||
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
|
||||
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
|
||||
- **Max budget**: $100 per key (config default).
|
||||
|
||||
```yaml
|
||||
# In litellm_config.yaml — ensures all future keys inherit these defaults:
|
||||
# NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no
|
||||
# default_key_generate_params block today, and a key generated with no explicit models comes back
|
||||
# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to
|
||||
# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and
|
||||
# re-verify before applying.
|
||||
litellm_settings:
|
||||
default_key_generate_params:
|
||||
models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b"]
|
||||
models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
|
||||
duration: null # ← permanent
|
||||
max_budget: 100
|
||||
metadata:
|
||||
@@ -286,13 +290,13 @@ auxiliary:
|
||||
api_key: sk-<agent-key-from-vault> # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain)
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
model: gemma-4-12b
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
model: gemma-4-12b
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
```
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ management, and prompt caching — without sacrificing agent capability.
|
||||
.123, any others on .129/.122) including compression, model, context_window,
|
||||
prompt_caching, memory settings
|
||||
- `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080,
|
||||
qwen .8:8080, gemma .110:8080)
|
||||
gpu-dense .8:8080, gpu-vision .110:8080)
|
||||
|
||||
### Maintains
|
||||
|
||||
@@ -56,8 +56,8 @@ duration.
|
||||
**Context is the root cause.** Every ~46K prompt token costs ~87s of
|
||||
prefill time at 532 tok/s. Fix context first, routing second.
|
||||
|
||||
- **Route by task**: qwen for code/standard queries; gemma for
|
||||
compression/auxiliary; strix-moe for compression tasks.
|
||||
- **Route by task**: gpu-dense for code/standard queries; gpu-vision for
|
||||
vision/web-auxiliary; syslog-auto for compression.
|
||||
- **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should
|
||||
compact at 51K, not 85K. Target 15% tail (not 30%).
|
||||
- **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these
|
||||
@@ -96,8 +96,10 @@ call enable-prompt-caching
|
||||
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
|
||||
|
||||
-- Phase 5: Verify end-to-end latency
|
||||
-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is
|
||||
-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use.
|
||||
|
||||
call verify-latency
|
||||
host: 192.168.68.116
|
||||
models: [syslog-auto, qwen3.6-27B-code, gemma-4-12b]
|
||||
models: [syslog-auto, qwen3.6-27B-code, gpu-vision]
|
||||
```
|
||||
|
||||
@@ -222,7 +222,7 @@ description: >
|
||||
|
||||
**Prometheus targets**:
|
||||
- 192.168.68.8:9400 (RTX 3090 — qwen)
|
||||
- 192.168.68.110:9400 (RTX 5070 — gemma)
|
||||
- 192.168.68.110:9400 (RTX 5070 — gpu-vision)
|
||||
- 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4)
|
||||
- harness-litellm:4000 (LiteLLM health)
|
||||
|
||||
|
||||
@@ -67,8 +67,10 @@ description: >
|
||||
4. **If action == "create"**:
|
||||
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
|
||||
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
|
||||
- Duration is null (permanent) — inherited from litellm default_key_generate_params
|
||||
- Set models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b", "strix-moe", "gpu-dense", "gpu-light", "qwen3.6-35B-udq4"]
|
||||
- Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||
- Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
|
||||
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
|
||||
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
|
||||
- Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed).
|
||||
- Return the new key
|
||||
5. **If action == "rotate"**:
|
||||
|
||||
@@ -28,7 +28,7 @@ description: >
|
||||
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
|
||||
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
|
||||
| strix-moe | 7.5s | — | Strix Halo, healthy |
|
||||
| gemma-4-12b | 2.6s | — | RTX 5070, healthy |
|
||||
| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
|
||||
|
||||
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
|
||||
backend), full recovery 07:00-08:00 with ZERO client failures once requests
|
||||
@@ -53,8 +53,7 @@ proxy queuing.
|
||||
|
||||
### 2. Auxiliary tasks — keep template timeouts, one correction
|
||||
|
||||
- vision: 60s (keep), web_extract: 30s (keep) — gemma-4-12b averages 2.6s;
|
||||
these are fine.
|
||||
- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
|
||||
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
|
||||
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
|
||||
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
|
||||
|
||||
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
|
||||
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
|
||||
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
|
||||
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
|
||||
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
|
||||
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
|
||||
|
||||
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
|
||||
observable behaviour — exit code and the emitted rule message — for the live alias and
|
||||
for both retired names. No network, vault, or SSH access is required.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||
AUDIT = ROOT / "audit-hermes-config.py"
|
||||
|
||||
BASE = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: {alias}
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: {alias}
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/v1
|
||||
"""
|
||||
|
||||
|
||||
def _run_config(tmp_path, name, text):
|
||||
cfg = tmp_path / name
|
||||
cfg.write_text(text)
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(AUDIT), str(cfg)],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
return proc.returncode, proc.stdout
|
||||
|
||||
|
||||
def _run(tmp_path, alias):
|
||||
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
|
||||
|
||||
|
||||
def test_live_canonical_alias_passes(tmp_path):
|
||||
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
|
||||
code, out = _run(tmp_path, "gpu-vision")
|
||||
assert code == 0, out
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_retired_gpu_light_is_rejected(tmp_path):
|
||||
"""A config pinned to the retired alias must fail, not pass."""
|
||||
code, out = _run(tmp_path, "gpu-light")
|
||||
assert code == 1, out
|
||||
assert "auxiliary.vision.model must be gpu-vision" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_gemma_is_rejected(tmp_path):
|
||||
"""The retired raw model name must fail Rule 8 as well."""
|
||||
code, out = _run(tmp_path, "gemma-4-12b")
|
||||
assert code == 1, out
|
||||
assert "auxiliary.vision.model must be gpu-vision" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_corrected_compression_example_passes(tmp_path):
|
||||
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
|
||||
code, out = _run(tmp_path, "gpu-vision")
|
||||
assert code == 0, out
|
||||
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
|
||||
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_delegation_is_rejected(tmp_path):
|
||||
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"delegation-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
"delegation:\n provider: harness",
|
||||
"delegation:\n provider: harness\n model: gpu-light",
|
||||
),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "delegation.model = 'gpu-light' is retired" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
|
||||
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"custom-provider-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
" - name: harness\n key_env: LITELLM_API_KEY",
|
||||
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
|
||||
),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "custom_providers[0].model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_raw_but_live_alias_warns_but_passes(tmp_path):
|
||||
"""Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"raw-qwen.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
"delegation:\n provider: harness",
|
||||
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
|
||||
),
|
||||
)
|
||||
assert code == 0, out
|
||||
assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out
|
||||
assert "prefer the stable alias gpu-dense" in out
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
||||
"""fallback_providers.model is model-bearing; a retired name there must fail."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"fallback-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "fallback_providers.model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_x_search_is_rejected(tmp_path):
|
||||
"""x_search.model was previously not enumerated; the derivation must catch it."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"x-search-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
"delegation:\n provider: harness",
|
||||
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
|
||||
),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "x_search.model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
|
||||
|
||||
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
|
||||
"""A nested auxiliary sub-block outside the named three must still be derived."""
|
||||
code, out = _run_config(
|
||||
tmp_path,
|
||||
"nested-aux-gpu-light.yaml",
|
||||
BASE.format(alias="gpu-vision").replace(
|
||||
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
|
||||
" compression:\n model: syslog-auto\n provider: harness\n"
|
||||
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
|
||||
),
|
||||
)
|
||||
assert code == 1, out
|
||||
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
|
||||
assert "RESULT: FAIL" in out
|
||||
Reference in New Issue
Block a user