Merge pull request 'fix(audit): stop requiring the retired gpu-light alias; derive model fields; sweep retired names' (#80) from fix/retired-alias-sweep-20260912 into master
PR Pipeline — Authorize → Validate → Review → Merge / auth (push) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / validate (push) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / lint (push) Successful in 2s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (push) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (push) Successful in 1s

This commit was merged in pull request #80.
This commit is contained in:
2026-09-12 17:15:05 +00:00
13 changed files with 350 additions and 77 deletions
+7 -3
View File
@@ -43,17 +43,21 @@ jobs:
echo "=== Prose Contract Frontmatter Validation ===" echo "=== Prose Contract Frontmatter Validation ==="
FAILED=0 FAILED=0
for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do
# NOTE: use herestrings, not `echo "$FM" | grep ...`. Under the runner's
# `-e -o pipefail`, `grep -q` exits on first match and can SIGPIPE the
# producer, making the pipeline report non-zero and raising a false
# "Missing name/description" whose file set varies run to run.
FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d') FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d')
[ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; } [ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; }
KIND=$(echo "$FM" | grep '^kind:' | awk '{print $2}') KIND=$(grep '^kind:' <<< "$FM" | awk '{print $2}')
case "$KIND" in case "$KIND" in
function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;; function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;;
*) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;; *) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;;
esac esac
echo "$FM" | grep -q '^name:' || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); } grep -q '^name:' <<< "$FM" || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
echo "$FM" | grep -q '^description:' || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); } grep -q '^description:' <<< "$FM" || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
done done
[ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; } [ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; }
echo "✅ Frontmatter validation passed" echo "✅ Frontmatter validation passed"
+85 -17
View File
@@ -36,6 +36,59 @@ def warn(rule, message):
WARNINGS.append(f"[{rule}] {message}") WARNINGS.append(f"[{rule}] {message}")
# Derivation rule: a model name is any scalar under a mapping key named `model` or
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
# specially. The only other exception is key `models` (litellm key-generation params carry a
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
# hand.
MODEL_KEYS = ("model", "model_name")
MODEL_SECTION_KEYS = ("default", "model", "model_name")
MODEL_LIST_KEYS = ("models",)
def _iter_model_values(node, path=""):
"""Yield (path, value) for every model-name-bearing scalar in a config."""
if isinstance(node, dict):
for key, value in node.items():
child = f"{path}.{key}" if path else key
if key in MODEL_KEYS:
if isinstance(value, dict):
for subkey in MODEL_SECTION_KEYS:
subvalue = value.get(subkey)
if isinstance(subvalue, str):
yield (f"{child}.{subkey}", subvalue)
for subkey, subvalue in value.items():
if isinstance(subvalue, (dict, list)):
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
elif isinstance(value, list):
yield from _iter_model_values(value, child)
else:
yield (child, value)
elif key in MODEL_LIST_KEYS:
yield from _iter_model_list(value, child)
elif isinstance(value, (dict, list)):
yield from _iter_model_values(value, child)
elif isinstance(node, list):
for i, item in enumerate(node):
yield from _iter_model_values(item, f"{path}[{i}]")
def _iter_model_list(node, path):
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
if isinstance(node, list):
for i, item in enumerate(node):
yield from _iter_model_list(item, f"{path}[{i}]")
elif isinstance(node, dict):
for key, value in node.items():
if key in MODEL_KEYS and isinstance(value, str):
yield (f"{path}.{key}", value)
elif isinstance(value, (dict, list)):
yield from _iter_model_list(value, f"{path}.{key}")
else:
yield (path, node)
def audit(path): def audit(path):
with open(path) as f: with open(path) as f:
cfg = yaml.safe_load(f) cfg = yaml.safe_load(f)
@@ -89,15 +142,16 @@ def audit(path):
) )
# --- Rule 8: GPU Workload Distribution --- # --- Rule 8: GPU Workload Distribution ---
# gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision.
check( check(
aux.get("vision", {}).get("model") == "gpu-light", aux.get("vision", {}).get("model") == "gpu-vision",
"Rule 8", "Rule 8",
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias", f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
) )
check( check(
aux.get("web_extract", {}).get("model") == "gpu-light", aux.get("web_extract", {}).get("model") == "gpu-vision",
"Rule 8", "Rule 8",
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias", f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
) )
# --- Rule 9: Compression Threshold --- # --- Rule 9: Compression Threshold ---
@@ -178,21 +232,35 @@ def audit(path):
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})", f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
) )
# --- No raw model names (Rule 7/8 spirit) --- # --- Retired/raw model names (Rule 7/8 spirit) ---
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"} # The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
for section_path, section_dict in [ # NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
("model", model), ("compression", comp), # gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
("auxiliary.vision", aux.get("vision", {})), # crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
("auxiliary.web_extract", aux.get("web_extract", {})), # RESOLVING names (verified 200) are discouraged but working, so they only WARN:
("auxiliary.compression", aux.get("compression", {})), # qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
("delegation", deleg), # Failing a working alias would reject valid configs - the exact defect this change fixes.
]: non_resolving = {
m = section_dict.get("model", "") "gpu-light": "gpu-vision",
if m in raw_names: "gemma-4-12b": "gpu-vision",
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
"ornith-1.0-35b": "strix-moe",
}
raw_but_live = {
"qwen3.6-27B-code": "gpu-dense",
"qwen3.6-35B-udq4": "strix-moe",
}
for field_path, value in _iter_model_values(cfg):
if value in non_resolving:
check(
False,
"Rule 7/8",
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
)
elif value in raw_but_live:
warn( warn(
"Rule 7/8", "Rule 7/8",
f"{section_path}.model = {m!r} — raw model name, use stable alias instead " f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
) )
# --- Report --- # --- Report ---
+8 -8
View File
@@ -216,7 +216,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions.
- **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled. - **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled.
- **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds. - **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds.
- **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down. - **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down.
- **Alias-retirement follow-up (2026-09-12)**: the agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, and the executable `audit-hermes-config.py` still reference the retired `gpu-light`/`gemma-4-12b` names, and `audit-hermes-config.py` Rule 8 currently fails a config whose vision/web_extract model is `gpu-vision`. These are tracked separately and are intentionally NOT updated in this change. - **Alias-retirement sweep (2026-09-12)**: `gemma-4-12b`, `gpu-light` and `crew-auto` are retired and replaced by `gpu-vision` / no cap respectively. The agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, `gpu-self-heal.prose.md`, `hermes-key-enforcement.prose.md`, `inference-optimization.prose.md`, `litellm-client-timeouts.prose.md` and the executable `audit-hermes-config.py` were all updated to the live canonical alias in the same change. **koby's config on .129 still names `gpu-light` (and `gemma-4-E4B`); .129 is report-only, so that is recorded for its owner and NOT edited here.**
## GPU Inference Benchmarks (Current) ## GPU Inference Benchmarks (Current)
@@ -239,7 +239,7 @@ History stored at `/root/data/toks-history.json` with 7-day rolling window.
### Stable Aliases — CRITICAL ### Stable Aliases — CRITICAL
All agent configs MUST use stable role-based aliases, never model-specific names: All agent configs MUST use stable role-based aliases, never model-specific names:
- `compression.model: strix-moe` - `compression.model: syslog-auto`
- `auxiliary.vision.model: gpu-vision` - `auxiliary.vision.model: gpu-vision`
- `delegation.model: gpu-dense` - `delegation.model: gpu-dense`
- `auxiliary.web_extract.model: gpu-vision` - `auxiliary.web_extract.model: gpu-vision`
@@ -249,25 +249,25 @@ When the underlying model is swapped, only the LiteLLM config changes — agent
### Context Windows ### Context Windows
- RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K** - RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K**
- **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek) - **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek)
- Compression threshold 0.60: fires at ~77K (~51K headroom before 128K ceiling) - Compression threshold 0.65 (audit Rule 9): fires at ~85K (~43K headroom before 128K ceiling)
- **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K) - **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K)
- Mumuni compression model alias: `strix-moe` - Mumuni compression model alias: `syslog-auto`
### Mumuni Agent Profile ### Mumuni Agent Profile
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs: Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs. The compression values below are the current required values per template Rules 7/9 and `audit-hermes-config.py`; whether Mumuni's LIVE config currently complies is a separate operational question.
| Setting | Value | Notes | | Setting | Value | Notes |
|---------|-------|-------| |---------|-------|-------|
| `model.default` | `syslog-auto` | Balanced default (pool router) | | `model.default` | `syslog-auto` | Balanced default (pool router) |
| `model.provider` | `custom:litellm` | LiteLLM on CT116 | | `model.provider` | `custom:litellm` | LiteLLM on CT116 |
| `compression.model` | `strix-moe` | Stable alias — survives model swaps | | `compression.model` | `syslog-auto` | Rule 7: auto-routing, prevents Strix Halo overload |
| `aux.compression.model` | `strix-moe` | Compression auxiliary model | | `aux.compression.model` | `syslog-auto` | Must match `compression.model` (Rule 7) |
| `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) | | `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) |
| `aux.web_extract.model` | `gpu-vision` | Web extraction | | `aux.web_extract.model` | `gpu-vision` | Web extraction |
| `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) | | `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) |
| `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling | | `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling |
| `compression.threshold` | 0.60 | Triggers at ~77K (~60% of 128K) — optimized for 128K context | | `compression.threshold` | 0.65 | Rule 9: triggers at ~85K for a 128K window |
| `compression.target_ratio` | 0.3 | Compresses to ~38K | | `compression.target_ratio` | 0.3 | Compresses to ~38K |
| `compression.protect_last_n` | 40 | Preserves last 40 messages | | `compression.protect_last_n` | 40 | Preserves last 40 messages |
| `memory.memory_char_limit` | 800 | Brief memory entries | | `memory.memory_char_limit` | 800 | Brief memory entries |
+2 -2
View File
@@ -27,8 +27,8 @@ agent: abiba
┌──────┐ ┌──────┐ ┌────────┐ ┌──────┐ ┌──────┐ ┌────────┐
│.8:8080│ │.110 │ │.116:80 │ │.8:8080│ │.110 │ │.116:80 │
│RTX3090│ │:8080 │ │nginx │ │RTX3090│ │:8080 │ │nginx │
│gemma │ │RTX5070│ │router │ │qwen │ │RTX5070│ │router │
└──────┘ │qwen27B│ │LiteLLM │ └──────┘ │vision │ │LiteLLM │
└──────┘ │dashboard│ └──────┘ │dashboard│
└────────┘ └────────┘
``` ```
+10 -9
View File
@@ -12,7 +12,7 @@ description: >
Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing. Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing.
Benchmark baselines refreshed to live values. Benchmark baselines refreshed to live values.
Prometheus exporters removed — not deployed; fall back to direct sidecar probes. Prometheus exporters removed — not deployed; fall back to direct sidecar probes.
Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet. Stable role-based aliases (strix-moe, gpu-dense, gpu-vision) from gpu-fleet.
agent: abiba agent: abiba
depends_on: depends_on:
- gpu-monitor.prose.md (live data source on .24:9100) - gpu-monitor.prose.md (live data source on .24:9100)
@@ -52,8 +52,8 @@ depends_on:
Key notes: Key notes:
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path. - All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path.
- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. - Stable aliases (gpu-dense, gpu-vision, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. The retired names `gpu-light` and `gemma-4-12b` were superseded by `gpu-vision` on 2026-09-12 and no longer resolve (400 `Invalid model name`).
- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was. - The RTX 5070 is the fastest endpoint per token — `gpu-vision` is its canonical alias. Route vision/web/light work there first. The RTX 5070 model is multimodal (image+text).
- Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads. - Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads.
- RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom). - RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom).
- RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes). - RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes).
@@ -64,7 +64,7 @@ Key notes:
- **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls - **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls
- **Fix**: - **Fix**:
1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control) 1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control)
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gemma → qwen, qwen → gemma) 2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gpu-vision → gpu-dense, gpu-dense → gpu-vision)
3. If all GPUs hot, alert about cooling infrastructure 3. If all GPUs hot, alert about cooling infrastructure
- **Verify**: Temp drops below 80°C within 5 minutes - **Verify**: Temp drops below 80°C within 5 minutes
- **Escalate after**: 3 verification failures → Zulip alert - **Escalate after**: 3 verification failures → Zulip alert
@@ -157,13 +157,14 @@ Key notes:
### Rule 10: Workload Distribution Optimization (updated 2026-07-18) ### Rule 10: Workload Distribution Optimization (updated 2026-07-18)
- **Detect**: GPU roles misaligned with hardware capabilities - **Detect**: GPU roles misaligned with hardware capabilities
- **Target distribution**: - **Target distribution**:
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM). - RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity).
- RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM). - RTX 5070 (gpu-vision, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks.
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM). - Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model).
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first. - **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
- **Weights are not restated here** — the live `syslog-auto` pool weights and rpm caps live in CT 116 `/opt/inference-harness/litellm_config.yaml`, the single source of truth.
- **Fix**: - **Fix**:
- Alert if any GPU is handling workload outside its designated role - Alert if any GPU is handling workload outside its designated role
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe) - Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-vision, strix-moe)
- Track per-GPU request distribution via LiteLLM spend logs - Track per-GPU request distribution via LiteLLM spend logs
- **Verify**: Each GPU's request pattern matches its designated role within 24h - **Verify**: Each GPU's request pattern matches its designated role within 24h
- **Escalate**: If role mismatch persists >48h → agent alias audit needed - **Escalate**: If role mismatch persists >48h → agent alias audit needed
@@ -311,6 +312,6 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab
If monitor response > 1MB, log a warning and skip the cycle rather than crashing. If monitor response > 1MB, log a warning and skip the cycle rather than crashing.
### L6: Stable Aliases Replace Model Names ### L6: Stable Aliases Replace Model Names
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15. - gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15; `gpu-light` was superseded by `gpu-vision` on 2026-09-12.
- Self-heal must use aliases for reporting and alerting, not model-specific names. - Self-heal must use aliases for reporting and alerting, not model-specific names.
- **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier. - **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier.
+7 -6
View File
@@ -80,7 +80,7 @@ custom_providers:
auxiliary: auxiliary:
vision: vision:
provider: harness provider: harness
model: gemma-4-12b # or syslog-auto model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux)
base_url: http://192.168.68.116/litellm/v1 base_url: http://192.168.68.116/litellm/v1
api_key_env: LITELLM_API_KEY api_key_env: LITELLM_API_KEY
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
@@ -95,7 +95,7 @@ auxiliary:
threshold: 0.65 threshold: 0.65
target_ratio: 0.3 target_ratio: 0.3
provider: harness provider: harness
model: syslog-auto # or gemma-4-12b model: syslog-auto # Rule 7: compression must be syslog-auto
base_url: http://192.168.68.116/litellm/v1 base_url: http://192.168.68.116/litellm/v1
api_key_env: LITELLM_API_KEY api_key_env: LITELLM_API_KEY
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
@@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension.
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`. Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
**models.json** — Must only list models authorized for the agent's LiteLLM key. **models.json** — Must only list models authorized for the agent's LiteLLM key.
Key is injected via `infisical run --` wrapper at PM2 startup: `/v1/models` is key-scoped and the live registry is CT 116
`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read
the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup:
```json ```json
{ {
"providers": { "providers": {
@@ -197,9 +199,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup:
{ "id": "syslog-auto" }, { "id": "syslog-auto" },
{ "id": "strix-moe" }, { "id": "strix-moe" },
{ "id": "gpu-dense" }, { "id": "gpu-dense" },
{ "id": "gpu-light" }, { "id": "gpu-vision" },
{ "id": "qwen3.6-27B-code" }, { "id": "qwen3.6-27B-code" }
{ "id": "gemma-4-12b" }
] ]
} }
} }
+18 -17
View File
@@ -102,7 +102,7 @@ model:
# and falls back to 256K when /v1/models lacks a context # and falls back to 256K when /v1/models lacks a context
# field (llama-server does). Without this override, agents # field (llama-server does). Without this override, agents
# silently run syslog-auto at 256K (verified 2026-08-09). # silently run syslog-auto at 256K (verified 2026-08-09).
# Set 65536 if using gemma-4-12b directly (tight VRAM). # Set 65536 if pinning a single model directly (tight VRAM).
fallback_providers: fallback_providers:
provider: deepseek provider: deepseek
@@ -143,25 +143,25 @@ compression:
# ─── Auxiliary Tasks (CONSISTENCY RULE) ─── # ─── Auxiliary Tasks (CONSISTENCY RULE) ───
# All auxiliary services MUST use identical model, base_url, and api_key_env: # All auxiliary services MUST use identical model, base_url, and api_key_env:
# model: gpu-light # stable alias (NOT raw "gemma-4-12b") # model: gpu-vision # stable alias (NOT a raw model name)
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK # base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
# api_key_env: LITELLM_API_KEY # api_key_env: LITELLM_API_KEY
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU. # Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning. # gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead. # Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4) # NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
# in agent configs — use the stable aliases so model swaps don't break agents. # and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
auxiliary: auxiliary:
vision: vision:
provider: harness provider: harness
model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b) model: gpu-vision # stable alias for RTX 5070
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
api_key_env: LITELLM_API_KEY api_key_env: LITELLM_API_KEY
timeout: 60 timeout: 60
download_timeout: 30 download_timeout: 30
web_extract: web_extract:
provider: harness provider: harness
model: gpu-light # stable alias for RTX 5070 model: gpu-vision # stable alias for RTX 5070
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
api_key_env: LITELLM_API_KEY api_key_env: LITELLM_API_KEY
timeout: 30 timeout: 30
@@ -252,10 +252,11 @@ The following MUST be identical across ALL profiles:
- For agents needing longer outputs: raise to 8192, but never omit - For agents needing longer outputs: raise to 8192, but never omit
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16) ### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized) - Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized)
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized) - Compression uses `syslog-auto` — the Strix Halo weighted pool (64GB, 128K ctx, compression-optimized); do NOT pin `compression.model` to `strix-moe` (audit Rule 7 rejects it)
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b` - **`ornith-1.0-35b` is NOT a valid compression model name** — LiteLLM does not serve it
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls. (do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml`
is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.** - **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
@@ -264,7 +265,7 @@ The following MUST be identical across ALL profiles:
- All auxiliary services MUST use identical routing: - All auxiliary services MUST use identical routing:
- `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK) - `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK)
- `api_key_env: LITELLM_API_KEY` - `api_key_env: LITELLM_API_KEY`
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably - **Do NOT use `syslog-auto` for `vision`/`web_extract`** — it routes unpredictably; compression is the deliberate exception (see the OPERATIONAL DECISION above)
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo - **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
(64GB UMA, 128K context) — the designated compression GPU. This frees the (64GB UMA, 128K context) — the designated compression GPU. This frees the
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning. RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
@@ -273,11 +274,11 @@ The following MUST be identical across ALL profiles:
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16) ### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations - **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K) - **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs - **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
- Agent profiles MUST route auxiliary tasks to the correct GPU: - Agent profiles MUST route auxiliary tasks to the correct GPU:
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070) - `auxiliary.vision.model: gpu-vision` (RTX 5070)
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070) - `auxiliary.web_extract.model: gpu-vision` (RTX 5070)
- `auxiliary.compression.model: syslog-auto` (Strix Halo) - `auxiliary.compression.model: syslog-auto` (Strix Halo)
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing - Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens) - For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
@@ -297,8 +298,8 @@ The following MUST be identical across ALL profiles:
- **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10. - **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10.
- **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto` - **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto`
- **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json` - **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json`
- `syslog-auto` is the LiteLLM routing model — it load-balances between strix-moe - `syslog-auto` is the LiteLLM routing model — it load-balances across the live pool
and qwen3.6-27B-code, with gemma-4-12b as fallback. Using it protects against: (see CT 116 `/opt/inference-harness/litellm_config.yaml` for the current members and weights). Using it protects against:
- Model name typos that cause 403 errors and silent worker failures - Model name typos that cause 403 errors and silent worker failures
- Single GPU downtime (routing falls back automatically) - Single GPU downtime (routing falls back automatically)
- Key/model authorization mismatches - Key/model authorization mismatches
+9 -5
View File
@@ -169,16 +169,20 @@ The agent picks up the new key via `infisical run --` at gateway startup.
**Keys are permanent and use bare agent name aliases.** **Keys are permanent and use bare agent name aliases.**
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`. - **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity. - **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven. - **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
- **Max budget**: $100 per key (config default). - **Max budget**: $100 per key (config default).
```yaml ```yaml
# In litellm_config.yaml — ensures all future keys inherit these defaults: # NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no
# default_key_generate_params block today, and a key generated with no explicit models comes back
# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to
# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and
# re-verify before applying.
litellm_settings: litellm_settings:
default_key_generate_params: default_key_generate_params:
models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b"] models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
duration: null # ← permanent duration: null # ← permanent
max_budget: 100 max_budget: 100
metadata: metadata:
@@ -286,13 +290,13 @@ auxiliary:
api_key: sk-<agent-key-from-vault> # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain) api_key: sk-<agent-key-from-vault> # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain)
api_key_env: LITELLM_API_KEY api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1 base_url: http://192.168.68.116/litellm/v1
model: gemma-4-12b model: gpu-vision
provider: harness provider: harness
compression: compression:
api_key: sk-<agent-key-from-vault> # ← workaround (same as above) api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
api_key_env: LITELLM_API_KEY api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/litellm/v1 base_url: http://192.168.68.116/litellm/v1
model: gemma-4-12b model: syslog-auto
provider: harness provider: harness
``` ```
+6 -4
View File
@@ -23,7 +23,7 @@ management, and prompt caching — without sacrificing agent capability.
.123, any others on .129/.122) including compression, model, context_window, .123, any others on .129/.122) including compression, model, context_window,
prompt_caching, memory settings prompt_caching, memory settings
- `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080, - `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080,
qwen .8:8080, gemma .110:8080) gpu-dense .8:8080, gpu-vision .110:8080)
### Maintains ### Maintains
@@ -56,8 +56,8 @@ duration.
**Context is the root cause.** Every ~46K prompt token costs ~87s of **Context is the root cause.** Every ~46K prompt token costs ~87s of
prefill time at 532 tok/s. Fix context first, routing second. prefill time at 532 tok/s. Fix context first, routing second.
- **Route by task**: qwen for code/standard queries; gemma for - **Route by task**: gpu-dense for code/standard queries; gpu-vision for
compression/auxiliary; strix-moe for compression tasks. vision/web-auxiliary; syslog-auto for compression.
- **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should - **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should
compact at 51K, not 85K. Target 15% tail (not 30%). compact at 51K, not 85K. Target 15% tail (not 30%).
- **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these - **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these
@@ -96,8 +96,10 @@ call enable-prompt-caching
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110] hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
-- Phase 5: Verify end-to-end latency -- Phase 5: Verify end-to-end latency
-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is
-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use.
call verify-latency call verify-latency
host: 192.168.68.116 host: 192.168.68.116
models: [syslog-auto, qwen3.6-27B-code, gemma-4-12b] models: [syslog-auto, qwen3.6-27B-code, gpu-vision]
``` ```
+1 -1
View File
@@ -222,7 +222,7 @@ description: >
**Prometheus targets**: **Prometheus targets**:
- 192.168.68.8:9400 (RTX 3090 — qwen) - 192.168.68.8:9400 (RTX 3090 — qwen)
- 192.168.68.110:9400 (RTX 5070 — gemma) - 192.168.68.110:9400 (RTX 5070 — gpu-vision)
- 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4) - 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4)
- harness-litellm:4000 (LiteLLM health) - harness-litellm:4000 (LiteLLM health)
+4 -2
View File
@@ -67,8 +67,10 @@ description: >
4. **If action == "create"**: 4. **If action == "create"**:
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date) - Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" } - Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
- Duration is null (permanent) — inherited from litellm default_key_generate_params - Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
- Set models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b", "strix-moe", "gpu-dense", "gpu-light", "qwen3.6-35B-udq4"] - Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
- Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed). - Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed).
- Return the new key - Return the new key
5. **If action == "rotate"**: 5. **If action == "rotate"**:
+2 -3
View File
@@ -28,7 +28,7 @@ description: >
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load | | syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto | | qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
| strix-moe | 7.5s | — | Strix Halo, healthy | | strix-moe | 7.5s | — | Strix Halo, healthy |
| gemma-4-12b | 2.6s | — | RTX 5070, healthy | | gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
backend), full recovery 07:00-08:00 with ZERO client failures once requests backend), full recovery 07:00-08:00 with ZERO client failures once requests
@@ -53,8 +53,7 @@ proxy queuing.
### 2. Auxiliary tasks — keep template timeouts, one correction ### 2. Auxiliary tasks — keep template timeouts, one correction
- vision: 60s (keep), web_extract: 30s (keep) — gemma-4-12b averages 2.6s; - vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
these are fine.
- compression: 300s (keep — this was already raised from 60 per gpu-fleet). - compression: 300s (keep — this was already raised from 60 per gpu-fleet).
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090 - **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as (qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
+191
View File
@@ -0,0 +1,191 @@
"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
observable behaviour — exit code and the emitted rule message — for the live alias and
for both retired names. No network, vault, or SSH access is required.
"""
from __future__ import annotations
import pathlib
import subprocess
import sys
ROOT = pathlib.Path(__file__).resolve().parent.parent
AUDIT = ROOT / "audit-hermes-config.py"
BASE = """
model:
api_key: ""
api_key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/v1
max_tokens: 4096
default: syslog-auto
provider: harness
fallback_providers:
provider: deepseek
model: deepseek-v4-flash
api_key_env: DEEPSEEK_API_KEY
compression:
model: syslog-auto
provider: harness
threshold: 0.65
max_context_window: 131072
auxiliary:
vision:
model: {alias}
provider: harness
web_extract:
model: {alias}
provider: harness
compression:
model: syslog-auto
provider: harness
delegation:
provider: harness
custom_providers:
- name: harness
key_env: LITELLM_API_KEY
base_url: http://192.168.68.116/v1
"""
def _run_config(tmp_path, name, text):
cfg = tmp_path / name
cfg.write_text(text)
proc = subprocess.run(
[sys.executable, str(AUDIT), str(cfg)],
capture_output=True, text=True,
)
return proc.returncode, proc.stdout
def _run(tmp_path, alias):
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
def test_live_canonical_alias_passes(tmp_path):
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "RESULT: PASS" in out
def test_retired_gpu_light_is_rejected(tmp_path):
"""A config pinned to the retired alias must fail, not pass."""
code, out = _run(tmp_path, "gpu-light")
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_retired_gemma_is_rejected(tmp_path):
"""The retired raw model name must fail Rule 8 as well."""
code, out = _run(tmp_path, "gemma-4-12b")
assert code == 1, out
assert "auxiliary.vision.model must be gpu-vision" in out
assert "RESULT: FAIL" in out
def test_corrected_compression_example_passes(tmp_path):
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
code, out = _run(tmp_path, "gpu-vision")
assert code == 0, out
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
assert "RESULT: PASS" in out
def test_retired_alias_in_delegation_is_rejected(tmp_path):
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
code, out = _run_config(
tmp_path,
"delegation-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: gpu-light",
),
)
assert code == 1, out
assert "delegation.model = 'gpu-light' is retired" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"custom-provider-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" - name: harness\n key_env: LITELLM_API_KEY",
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
),
)
assert code == 1, out
assert "custom_providers[0].model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_raw_but_live_alias_warns_but_passes(tmp_path):
"""Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs."""
code, out = _run_config(
tmp_path,
"raw-qwen.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
),
)
assert code == 0, out
assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out
assert "prefer the stable alias gpu-dense" in out
assert "RESULT: PASS" in out
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
"""fallback_providers.model is model-bearing; a retired name there must fail."""
code, out = _run_config(
tmp_path,
"fallback-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
)
assert code == 1, out
assert "fallback_providers.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_x_search_is_rejected(tmp_path):
"""x_search.model was previously not enumerated; the derivation must catch it."""
code, out = _run_config(
tmp_path,
"x-search-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
"delegation:\n provider: harness",
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
),
)
assert code == 1, out
assert "x_search.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
"""A nested auxiliary sub-block outside the named three must still be derived."""
code, out = _run_config(
tmp_path,
"nested-aux-gpu-light.yaml",
BASE.format(alias="gpu-vision").replace(
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
" compression:\n model: syslog-auto\n provider: harness\n"
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
),
)
assert code == 1, out
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
assert "RESULT: FAIL" in out