fix(audit): stop requiring the retired gpu-light alias; derive model fields; sweep retired names #80
@@ -43,17 +43,21 @@ jobs:
|
|||||||
echo "=== Prose Contract Frontmatter Validation ==="
|
echo "=== Prose Contract Frontmatter Validation ==="
|
||||||
FAILED=0
|
FAILED=0
|
||||||
for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do
|
for f in $(find . -name "*.prose.md" -not -path "./.git/*" -not -path "./runs/*"); do
|
||||||
|
# NOTE: use herestrings, not `echo "$FM" | grep ...`. Under the runner's
|
||||||
|
# `-e -o pipefail`, `grep -q` exits on first match and can SIGPIPE the
|
||||||
|
# producer, making the pipeline report non-zero and raising a false
|
||||||
|
# "Missing name/description" whose file set varies run to run.
|
||||||
FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d')
|
FM=$(sed -n '/^---$/,/^---$/p' "$f" | sed '1d;$d')
|
||||||
[ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; }
|
[ -z "$FM" ] && { echo " ❌ $f: No YAML frontmatter"; FAILED=$((FAILED+1)); continue; }
|
||||||
|
|
||||||
KIND=$(echo "$FM" | grep '^kind:' | awk '{print $2}')
|
KIND=$(grep '^kind:' <<< "$FM" | awk '{print $2}')
|
||||||
case "$KIND" in
|
case "$KIND" in
|
||||||
function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;;
|
function|responsibility|gateway|pattern|test|template|architecture|enforcement) echo " ✅ $f: kind=$KIND" ;;
|
||||||
*) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;;
|
*) echo " ❌ $f: Invalid kind='$KIND'"; FAILED=$((FAILED+1)) ;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
echo "$FM" | grep -q '^name:' || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
|
grep -q '^name:' <<< "$FM" || { echo " ❌ $f: Missing name"; FAILED=$((FAILED+1)); }
|
||||||
echo "$FM" | grep -q '^description:' || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
|
grep -q '^description:' <<< "$FM" || { echo " ❌ $f: Missing description"; FAILED=$((FAILED+1)); }
|
||||||
done
|
done
|
||||||
[ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; }
|
[ $FAILED -gt 0 ] && { echo "❌ FRONTMATTER FAILED ($FAILED error(s))"; exit 1; }
|
||||||
echo "✅ Frontmatter validation passed"
|
echo "✅ Frontmatter validation passed"
|
||||||
|
|||||||
+85
-17
@@ -36,6 +36,59 @@ def warn(rule, message):
|
|||||||
WARNINGS.append(f"[{rule}] {message}")
|
WARNINGS.append(f"[{rule}] {message}")
|
||||||
|
|
||||||
|
|
||||||
|
# Derivation rule: a model name is any scalar under a mapping key named `model` or
|
||||||
|
# `model_name`, at any depth. The top-level `model:` SECTION is the exception where the model
|
||||||
|
# name lives under `default`/`model`/`model_name` inside that section, so it is descended
|
||||||
|
# specially. The only other exception is key `models` (litellm key-generation params carry a
|
||||||
|
# list of model names). EXTEND THE ALLOWLIST for a new exception; do NOT add another field by
|
||||||
|
# hand.
|
||||||
|
MODEL_KEYS = ("model", "model_name")
|
||||||
|
MODEL_SECTION_KEYS = ("default", "model", "model_name")
|
||||||
|
MODEL_LIST_KEYS = ("models",)
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_model_values(node, path=""):
|
||||||
|
"""Yield (path, value) for every model-name-bearing scalar in a config."""
|
||||||
|
if isinstance(node, dict):
|
||||||
|
for key, value in node.items():
|
||||||
|
child = f"{path}.{key}" if path else key
|
||||||
|
if key in MODEL_KEYS:
|
||||||
|
if isinstance(value, dict):
|
||||||
|
for subkey in MODEL_SECTION_KEYS:
|
||||||
|
subvalue = value.get(subkey)
|
||||||
|
if isinstance(subvalue, str):
|
||||||
|
yield (f"{child}.{subkey}", subvalue)
|
||||||
|
for subkey, subvalue in value.items():
|
||||||
|
if isinstance(subvalue, (dict, list)):
|
||||||
|
yield from _iter_model_values(subvalue, f"{child}.{subkey}")
|
||||||
|
elif isinstance(value, list):
|
||||||
|
yield from _iter_model_values(value, child)
|
||||||
|
else:
|
||||||
|
yield (child, value)
|
||||||
|
elif key in MODEL_LIST_KEYS:
|
||||||
|
yield from _iter_model_list(value, child)
|
||||||
|
elif isinstance(value, (dict, list)):
|
||||||
|
yield from _iter_model_values(value, child)
|
||||||
|
elif isinstance(node, list):
|
||||||
|
for i, item in enumerate(node):
|
||||||
|
yield from _iter_model_values(item, f"{path}[{i}]")
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_model_list(node, path):
|
||||||
|
"""Yield scalars under an allowlisted `models` key (list of names or list of dicts)."""
|
||||||
|
if isinstance(node, list):
|
||||||
|
for i, item in enumerate(node):
|
||||||
|
yield from _iter_model_list(item, f"{path}[{i}]")
|
||||||
|
elif isinstance(node, dict):
|
||||||
|
for key, value in node.items():
|
||||||
|
if key in MODEL_KEYS and isinstance(value, str):
|
||||||
|
yield (f"{path}.{key}", value)
|
||||||
|
elif isinstance(value, (dict, list)):
|
||||||
|
yield from _iter_model_list(value, f"{path}.{key}")
|
||||||
|
else:
|
||||||
|
yield (path, node)
|
||||||
|
|
||||||
|
|
||||||
def audit(path):
|
def audit(path):
|
||||||
with open(path) as f:
|
with open(path) as f:
|
||||||
cfg = yaml.safe_load(f)
|
cfg = yaml.safe_load(f)
|
||||||
@@ -89,15 +142,16 @@ def audit(path):
|
|||||||
)
|
)
|
||||||
|
|
||||||
# --- Rule 8: GPU Workload Distribution ---
|
# --- Rule 8: GPU Workload Distribution ---
|
||||||
|
# gpu-light (and gemma-4-12b) were retired 2026-09-12; the RTX 5070 stable alias is gpu-vision.
|
||||||
check(
|
check(
|
||||||
aux.get("vision", {}).get("model") == "gpu-light",
|
aux.get("vision", {}).get("model") == "gpu-vision",
|
||||||
"Rule 8",
|
"Rule 8",
|
||||||
f"auxiliary.vision.model must be gpu-light (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
f"auxiliary.vision.model must be gpu-vision (got {aux.get('vision', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||||
)
|
)
|
||||||
check(
|
check(
|
||||||
aux.get("web_extract", {}).get("model") == "gpu-light",
|
aux.get("web_extract", {}).get("model") == "gpu-vision",
|
||||||
"Rule 8",
|
"Rule 8",
|
||||||
f"auxiliary.web_extract.model must be gpu-light (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
f"auxiliary.web_extract.model must be gpu-vision (got {aux.get('web_extract', {}).get('model')!r}) — RTX 5070 stable alias",
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- Rule 9: Compression Threshold ---
|
# --- Rule 9: Compression Threshold ---
|
||||||
@@ -178,21 +232,35 @@ def audit(path):
|
|||||||
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
f"custom_providers[0].base_url must end with /v1 (got {cp.get('base_url')!r})",
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- No raw model names (Rule 7/8 spirit) ---
|
# --- Retired/raw model names (Rule 7/8 spirit) ---
|
||||||
raw_names = {"gemma-4-12b", "qwen3.6-27B-code", "qwen3.6-35B-udq4", "ornith-1.0-35b"}
|
# The audit's job is to catch configs that are BROKEN, not to enforce a style preference.
|
||||||
for section_path, section_dict in [
|
# NON-RESOLVING names (removed 2026-09-12, verified 400/403 via live LiteLLM) must hard-FAIL:
|
||||||
("model", model), ("compression", comp),
|
# gpu-light -> gpu-vision ; gemma-4-12b -> gpu-vision
|
||||||
("auxiliary.vision", aux.get("vision", {})),
|
# crew-auto -> syslog-auto (its 64K cap is retired; no cap in force) ; ornith-1.0-35b -> strix-moe
|
||||||
("auxiliary.web_extract", aux.get("web_extract", {})),
|
# RESOLVING names (verified 200) are discouraged but working, so they only WARN:
|
||||||
("auxiliary.compression", aux.get("compression", {})),
|
# qwen3.6-27B-code -> gpu-dense ; qwen3.6-35B-udq4 -> strix-moe
|
||||||
("delegation", deleg),
|
# Failing a working alias would reject valid configs - the exact defect this change fixes.
|
||||||
]:
|
non_resolving = {
|
||||||
m = section_dict.get("model", "")
|
"gpu-light": "gpu-vision",
|
||||||
if m in raw_names:
|
"gemma-4-12b": "gpu-vision",
|
||||||
|
"crew-auto": "syslog-auto (its 64K cap is retired; no cap in force)",
|
||||||
|
"ornith-1.0-35b": "strix-moe",
|
||||||
|
}
|
||||||
|
raw_but_live = {
|
||||||
|
"qwen3.6-27B-code": "gpu-dense",
|
||||||
|
"qwen3.6-35B-udq4": "strix-moe",
|
||||||
|
}
|
||||||
|
for field_path, value in _iter_model_values(cfg):
|
||||||
|
if value in non_resolving:
|
||||||
|
check(
|
||||||
|
False,
|
||||||
|
"Rule 7/8",
|
||||||
|
f"{field_path} = {value!r} is retired and no longer resolves (2026-09-12) — use {non_resolving[value]}",
|
||||||
|
)
|
||||||
|
elif value in raw_but_live:
|
||||||
warn(
|
warn(
|
||||||
"Rule 7/8",
|
"Rule 7/8",
|
||||||
f"{section_path}.model = {m!r} — raw model name, use stable alias instead "
|
f"{field_path} = {value!r} is a raw-but-live model name — prefer the stable alias {raw_but_live[value]}",
|
||||||
f"(gpu-light, gpu-dense, strix-moe, syslog-auto)",
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- Report ---
|
# --- Report ---
|
||||||
|
|||||||
+8
-8
@@ -216,7 +216,7 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions.
|
|||||||
- **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled.
|
- **Alert migration**: All alerts now go to `#agent-hub` topics (`alerts-gpu`, `alerts-pm2`, `alerts-infra`) instead of DMs. Cross-agent visibility enabled.
|
||||||
- **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds.
|
- **tok/s benchmarks**: Measured every 5 min via LiteLLM proxy. Baselines tracked with 30%/50% degradation thresholds.
|
||||||
- **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down.
|
- **NetBird 502**: Tanko routes through NetBird for litellm.sysloggh.net. Use direct IP if NetBird down.
|
||||||
- **Alias-retirement follow-up (2026-09-12)**: the agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, and the executable `audit-hermes-config.py` still reference the retired `gpu-light`/`gemma-4-12b` names, and `audit-hermes-config.py` Rule 8 currently fails a config whose vision/web_extract model is `gpu-vision`. These are tracked separately and are intentionally NOT updated in this change.
|
- **Alias-retirement sweep (2026-09-12)**: `gemma-4-12b`, `gpu-light` and `crew-auto` are retired and replaced by `gpu-vision` / no cap respectively. The agent-facing templates (`hermes-config-template.prose.md`, `hermes-agent-baseline.prose.md`), `litellm-api-keys.prose.md`, `gpu-self-heal.prose.md`, `hermes-key-enforcement.prose.md`, `inference-optimization.prose.md`, `litellm-client-timeouts.prose.md` and the executable `audit-hermes-config.py` were all updated to the live canonical alias in the same change. **koby's config on .129 still names `gpu-light` (and `gemma-4-E4B`); .129 is report-only, so that is recorded for its owner and NOT edited here.**
|
||||||
|
|
||||||
## GPU Inference Benchmarks (Current)
|
## GPU Inference Benchmarks (Current)
|
||||||
|
|
||||||
@@ -239,7 +239,7 @@ History stored at `/root/data/toks-history.json` with 7-day rolling window.
|
|||||||
### Stable Aliases — CRITICAL
|
### Stable Aliases — CRITICAL
|
||||||
|
|
||||||
All agent configs MUST use stable role-based aliases, never model-specific names:
|
All agent configs MUST use stable role-based aliases, never model-specific names:
|
||||||
- `compression.model: strix-moe`
|
- `compression.model: syslog-auto`
|
||||||
- `auxiliary.vision.model: gpu-vision`
|
- `auxiliary.vision.model: gpu-vision`
|
||||||
- `delegation.model: gpu-dense`
|
- `delegation.model: gpu-dense`
|
||||||
- `auxiliary.web_extract.model: gpu-vision`
|
- `auxiliary.web_extract.model: gpu-vision`
|
||||||
@@ -249,25 +249,25 @@ When the underlying model is swapped, only the LiteLLM config changes — agent
|
|||||||
### Context Windows
|
### Context Windows
|
||||||
- RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K**
|
- RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K**
|
||||||
- **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek)
|
- **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek)
|
||||||
- Compression threshold 0.60: fires at ~77K (~51K headroom before 128K ceiling)
|
- Compression threshold 0.65 (audit Rule 9): fires at ~85K (~43K headroom before 128K ceiling)
|
||||||
- **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K)
|
- **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K)
|
||||||
- Mumuni compression model alias: `strix-moe`
|
- Mumuni compression model alias: `syslog-auto`
|
||||||
|
|
||||||
### Mumuni Agent Profile
|
### Mumuni Agent Profile
|
||||||
|
|
||||||
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs:
|
Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the primary business assistant. This profile is the reference for all agent configs. The compression values below are the current required values per template Rules 7/9 and `audit-hermes-config.py`; whether Mumuni's LIVE config currently complies is a separate operational question.
|
||||||
|
|
||||||
| Setting | Value | Notes |
|
| Setting | Value | Notes |
|
||||||
|---------|-------|-------|
|
|---------|-------|-------|
|
||||||
| `model.default` | `syslog-auto` | Balanced default (pool router) |
|
| `model.default` | `syslog-auto` | Balanced default (pool router) |
|
||||||
| `model.provider` | `custom:litellm` | LiteLLM on CT116 |
|
| `model.provider` | `custom:litellm` | LiteLLM on CT116 |
|
||||||
| `compression.model` | `strix-moe` | Stable alias — survives model swaps |
|
| `compression.model` | `syslog-auto` | Rule 7: auto-routing, prevents Strix Halo overload |
|
||||||
| `aux.compression.model` | `strix-moe` | Compression auxiliary model |
|
| `aux.compression.model` | `syslog-auto` | Must match `compression.model` (Rule 7) |
|
||||||
| `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) |
|
| `aux.vision.model` | `gpu-vision` | Vision tasks (RTX 5070) |
|
||||||
| `aux.web_extract.model` | `gpu-vision` | Web extraction |
|
| `aux.web_extract.model` | `gpu-vision` | Web extraction |
|
||||||
| `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) |
|
| `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) |
|
||||||
| `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling |
|
| `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling |
|
||||||
| `compression.threshold` | 0.60 | Triggers at ~77K (~60% of 128K) — optimized for 128K context |
|
| `compression.threshold` | 0.65 | Rule 9: triggers at ~85K for a 128K window |
|
||||||
| `compression.target_ratio` | 0.3 | Compresses to ~38K |
|
| `compression.target_ratio` | 0.3 | Compresses to ~38K |
|
||||||
| `compression.protect_last_n` | 40 | Preserves last 40 messages |
|
| `compression.protect_last_n` | 40 | Preserves last 40 messages |
|
||||||
| `memory.memory_char_limit` | 800 | Brief memory entries |
|
| `memory.memory_char_limit` | 800 | Brief memory entries |
|
||||||
|
|||||||
@@ -27,8 +27,8 @@ agent: abiba
|
|||||||
┌──────┐ ┌──────┐ ┌────────┐
|
┌──────┐ ┌──────┐ ┌────────┐
|
||||||
│.8:8080│ │.110 │ │.116:80 │
|
│.8:8080│ │.110 │ │.116:80 │
|
||||||
│RTX3090│ │:8080 │ │nginx │
|
│RTX3090│ │:8080 │ │nginx │
|
||||||
│gemma │ │RTX5070│ │router │
|
│qwen │ │RTX5070│ │router │
|
||||||
└──────┘ │qwen27B│ │LiteLLM │
|
└──────┘ │vision │ │LiteLLM │
|
||||||
└──────┘ │dashboard│
|
└──────┘ │dashboard│
|
||||||
└────────┘
|
└────────┘
|
||||||
```
|
```
|
||||||
|
|||||||
+10
-9
@@ -12,7 +12,7 @@ description: >
|
|||||||
Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing.
|
Router (port 9000) DECOMMISSIONED 2026-09-11; references replaced with direct GPU routing.
|
||||||
Benchmark baselines refreshed to live values.
|
Benchmark baselines refreshed to live values.
|
||||||
Prometheus exporters removed — not deployed; fall back to direct sidecar probes.
|
Prometheus exporters removed — not deployed; fall back to direct sidecar probes.
|
||||||
Stable role-based aliases (strix-moe, gpu-dense, gpu-light) from gpu-fleet.
|
Stable role-based aliases (strix-moe, gpu-dense, gpu-vision) from gpu-fleet.
|
||||||
agent: abiba
|
agent: abiba
|
||||||
depends_on:
|
depends_on:
|
||||||
- gpu-monitor.prose.md (live data source on .24:9100)
|
- gpu-monitor.prose.md (live data source on .24:9100)
|
||||||
@@ -52,8 +52,8 @@ depends_on:
|
|||||||
|
|
||||||
Key notes:
|
Key notes:
|
||||||
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path.
|
- All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) was decommissioned 2026-09-11 and is NOT in the inference path.
|
||||||
- Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated.
|
- Stable aliases (gpu-dense, gpu-vision, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. The retired names `gpu-light` and `gemma-4-12b` were superseded by `gpu-vision` on 2026-09-12 and no longer resolve (400 `Invalid model name`).
|
||||||
- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was.
|
- The RTX 5070 is the fastest endpoint per token — `gpu-vision` is its canonical alias. Route vision/web/light work there first. The RTX 5070 model is multimodal (image+text).
|
||||||
- Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads.
|
- Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads.
|
||||||
- RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom).
|
- RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom).
|
||||||
- RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes).
|
- RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes).
|
||||||
@@ -64,7 +64,7 @@ Key notes:
|
|||||||
- **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls
|
- **Detect**: Any GPU temp >85°C sustained for 2+ consecutive polls
|
||||||
- **Fix**:
|
- **Fix**:
|
||||||
1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control)
|
1. Reduce inference concurrency on that GPU (load-side cooling only — NO fan control)
|
||||||
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gemma → qwen, qwen → gemma)
|
2. Redirect new requests to cooler GPUs via LiteLLM fallback chains (gpu-vision → gpu-dense, gpu-dense → gpu-vision)
|
||||||
3. If all GPUs hot, alert about cooling infrastructure
|
3. If all GPUs hot, alert about cooling infrastructure
|
||||||
- **Verify**: Temp drops below 80°C within 5 minutes
|
- **Verify**: Temp drops below 80°C within 5 minutes
|
||||||
- **Escalate after**: 3 verification failures → Zulip alert
|
- **Escalate after**: 3 verification failures → Zulip alert
|
||||||
@@ -157,13 +157,14 @@ Key notes:
|
|||||||
### Rule 10: Workload Distribution Optimization (updated 2026-07-18)
|
### Rule 10: Workload Distribution Optimization (updated 2026-07-18)
|
||||||
- **Detect**: GPU roles misaligned with hardware capabilities
|
- **Detect**: GPU roles misaligned with hardware capabilities
|
||||||
- **Target distribution**:
|
- **Target distribution**:
|
||||||
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM).
|
- RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity).
|
||||||
- RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM).
|
- RTX 5070 (gpu-vision, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks.
|
||||||
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM).
|
- Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model).
|
||||||
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
|
- **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first.
|
||||||
|
- **Weights are not restated here** — the live `syslog-auto` pool weights and rpm caps live in CT 116 `/opt/inference-harness/litellm_config.yaml`, the single source of truth.
|
||||||
- **Fix**:
|
- **Fix**:
|
||||||
- Alert if any GPU is handling workload outside its designated role
|
- Alert if any GPU is handling workload outside its designated role
|
||||||
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-light, strix-moe)
|
- Recommend agent alias updates to match workload to GPU role (use stable aliases: gpu-dense, gpu-vision, strix-moe)
|
||||||
- Track per-GPU request distribution via LiteLLM spend logs
|
- Track per-GPU request distribution via LiteLLM spend logs
|
||||||
- **Verify**: Each GPU's request pattern matches its designated role within 24h
|
- **Verify**: Each GPU's request pattern matches its designated role within 24h
|
||||||
- **Escalate**: If role mismatch persists >48h → agent alias audit needed
|
- **Escalate**: If role mismatch persists >48h → agent alias audit needed
|
||||||
@@ -311,6 +312,6 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab
|
|||||||
If monitor response > 1MB, log a warning and skip the cycle rather than crashing.
|
If monitor response > 1MB, log a warning and skip the cycle rather than crashing.
|
||||||
|
|
||||||
### L6: Stable Aliases Replace Model Names
|
### L6: Stable Aliases Replace Model Names
|
||||||
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15.
|
- gpu-fleet introduced stable aliases (strix-moe, gpu-dense, gpu-light) on 2026-07-15; `gpu-light` was superseded by `gpu-vision` on 2026-09-12.
|
||||||
- Self-heal must use aliases for reporting and alerting, not model-specific names.
|
- Self-heal must use aliases for reporting and alerting, not model-specific names.
|
||||||
- **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier.
|
- **Rule**: All alert messages and KG nodes use the stable alias as the GPU identifier.
|
||||||
|
|||||||
@@ -80,7 +80,7 @@ custom_providers:
|
|||||||
auxiliary:
|
auxiliary:
|
||||||
vision:
|
vision:
|
||||||
provider: harness
|
provider: harness
|
||||||
model: gemma-4-12b # or syslog-auto
|
model: gpu-vision # RTX 5070 stable alias (Rule 8; do not use syslog-auto for aux)
|
||||||
base_url: http://192.168.68.116/litellm/v1
|
base_url: http://192.168.68.116/litellm/v1
|
||||||
api_key_env: LITELLM_API_KEY
|
api_key_env: LITELLM_API_KEY
|
||||||
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
||||||
@@ -95,7 +95,7 @@ auxiliary:
|
|||||||
threshold: 0.65
|
threshold: 0.65
|
||||||
target_ratio: 0.3
|
target_ratio: 0.3
|
||||||
provider: harness
|
provider: harness
|
||||||
model: syslog-auto # or gemma-4-12b
|
model: syslog-auto # Rule 7: compression must be syslog-auto
|
||||||
base_url: http://192.168.68.116/litellm/v1
|
base_url: http://192.168.68.116/litellm/v1
|
||||||
api_key_env: LITELLM_API_KEY
|
api_key_env: LITELLM_API_KEY
|
||||||
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
api_key: <value from: infisical secrets get LITELLM_API_KEY --project=agents --env=production> # ← MANDATORY workaround
|
||||||
@@ -185,7 +185,9 @@ Abiba (CT100) runs pi via PM2 with the Zulip extension.
|
|||||||
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
|
Config files: `~/.pi/agent/models.json`, `~/.pi/agent/settings.json`.
|
||||||
|
|
||||||
**models.json** — Must only list models authorized for the agent's LiteLLM key.
|
**models.json** — Must only list models authorized for the agent's LiteLLM key.
|
||||||
Key is injected via `infisical run --` wrapper at PM2 startup:
|
`/v1/models` is key-scoped and the live registry is CT 116
|
||||||
|
`/opt/inference-harness/litellm_config.yaml`; treat the list below as a snapshot and re-read
|
||||||
|
the registry before applying. Key is injected via `infisical run --` wrapper at PM2 startup:
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"providers": {
|
"providers": {
|
||||||
@@ -197,9 +199,8 @@ Key is injected via `infisical run --` wrapper at PM2 startup:
|
|||||||
{ "id": "syslog-auto" },
|
{ "id": "syslog-auto" },
|
||||||
{ "id": "strix-moe" },
|
{ "id": "strix-moe" },
|
||||||
{ "id": "gpu-dense" },
|
{ "id": "gpu-dense" },
|
||||||
{ "id": "gpu-light" },
|
{ "id": "gpu-vision" },
|
||||||
{ "id": "qwen3.6-27B-code" },
|
{ "id": "qwen3.6-27B-code" }
|
||||||
{ "id": "gemma-4-12b" }
|
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -102,7 +102,7 @@ model:
|
|||||||
# and falls back to 256K when /v1/models lacks a context
|
# and falls back to 256K when /v1/models lacks a context
|
||||||
# field (llama-server does). Without this override, agents
|
# field (llama-server does). Without this override, agents
|
||||||
# silently run syslog-auto at 256K (verified 2026-08-09).
|
# silently run syslog-auto at 256K (verified 2026-08-09).
|
||||||
# Set 65536 if using gemma-4-12b directly (tight VRAM).
|
# Set 65536 if pinning a single model directly (tight VRAM).
|
||||||
|
|
||||||
fallback_providers:
|
fallback_providers:
|
||||||
provider: deepseek
|
provider: deepseek
|
||||||
@@ -143,25 +143,25 @@ compression:
|
|||||||
|
|
||||||
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
||||||
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
||||||
# model: gpu-light # stable alias (NOT raw "gemma-4-12b")
|
# model: gpu-vision # stable alias (NOT a raw model name)
|
||||||
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||||
# api_key_env: LITELLM_API_KEY
|
# api_key_env: LITELLM_API_KEY
|
||||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||||
# gpu-light = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||||
# NEVER use raw model names (gemma-4-12b, qwen3.6-27B-code, qwen3.6-35B-udq4)
|
# NEVER use raw model names (e.g. qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
|
||||||
# in agent configs — use the stable aliases so model swaps don't break agents.
|
# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
|
||||||
auxiliary:
|
auxiliary:
|
||||||
vision:
|
vision:
|
||||||
provider: harness
|
provider: harness
|
||||||
model: gpu-light # stable alias for RTX 5070 (was raw gemma-4-12b)
|
model: gpu-vision # stable alias for RTX 5070
|
||||||
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||||
api_key_env: LITELLM_API_KEY
|
api_key_env: LITELLM_API_KEY
|
||||||
timeout: 60
|
timeout: 60
|
||||||
download_timeout: 30
|
download_timeout: 30
|
||||||
web_extract:
|
web_extract:
|
||||||
provider: harness
|
provider: harness
|
||||||
model: gpu-light # stable alias for RTX 5070
|
model: gpu-vision # stable alias for RTX 5070
|
||||||
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||||
api_key_env: LITELLM_API_KEY
|
api_key_env: LITELLM_API_KEY
|
||||||
timeout: 30
|
timeout: 30
|
||||||
@@ -252,10 +252,11 @@ The following MUST be identical across ALL profiles:
|
|||||||
- For agents needing longer outputs: raise to 8192, but never omit
|
- For agents needing longer outputs: raise to 8192, but never omit
|
||||||
|
|
||||||
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
|
### Rule 7: Auxiliary Model Consistency (UPDATED 2026-07-16)
|
||||||
- Vision and web_extract use `gemma-4-12b` (RTX 5070 — 12GB, vision-optimized)
|
- Vision and web_extract use `gpu-vision` (RTX 5070 — 12GB, vision-optimized)
|
||||||
- Compression uses `strix-moe` (stable alias for Strix Halo — 64GB, 128K ctx, compression-optimized)
|
- Compression uses `syslog-auto` — the Strix Halo weighted pool (64GB, 128K ctx, compression-optimized); do NOT pin `compression.model` to `strix-moe` (audit Rule 7 rejects it)
|
||||||
- **`strix-moe` is the only valid compression model name** — LiteLLM does NOT serve `ornith-1.0-35b`
|
- **`ornith-1.0-35b` is NOT a valid compression model name** — LiteLLM does not serve it
|
||||||
(it serves `strix-moe`, `qwen3.6-35B-udq4`, `gpu-dense`, `gpu-light`, `syslog-auto`, `gemma-4-12b`, `qwen3.6-27B-code`). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
(do not restate the served model list here — CT 116 `/opt/inference-harness/litellm_config.yaml`
|
||||||
|
is the single source of truth for models, aliases, weights and fallbacks). Old configs with `ornith-1.0-35b` cause 403/model-not-found on compression calls.
|
||||||
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
|
- **OPERATIONAL DECISION (2026-07-23): Use `syslog-auto` for compression across all agents.**
|
||||||
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
|
The `syslog-auto` alias routes to the Strix Halo, but uses the weighted pool instead of pinning
|
||||||
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
|
to `strix-moe` directly. This prevents sustained Strix Halo thermal load because the pool can
|
||||||
@@ -264,7 +265,7 @@ The following MUST be identical across ALL profiles:
|
|||||||
- All auxiliary services MUST use identical routing:
|
- All auxiliary services MUST use identical routing:
|
||||||
- `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK)
|
- `base_url: http://192.168.68.116/litellm/v1` (Rule 5, 2026-08-09: canonical authenticated; `/v1` also OK)
|
||||||
- `api_key_env: LITELLM_API_KEY`
|
- `api_key_env: LITELLM_API_KEY`
|
||||||
- **Do NOT use `syslog-auto`** for auxiliary tasks — it routes unpredictably
|
- **Do NOT use `syslog-auto` for `vision`/`web_extract`** — it routes unpredictably; compression is the deliberate exception (see the OPERATIONAL DECISION above)
|
||||||
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
|
- **Compression on Strix Halo**: The strix-moe alias routes to Strix Halo
|
||||||
(64GB UMA, 128K context) — the designated compression GPU. This frees the
|
(64GB UMA, 128K context) — the designated compression GPU. This frees the
|
||||||
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
|
RTX 5070 for vision and web search, and the RTX 3090 for heavy reasoning.
|
||||||
@@ -273,11 +274,11 @@ The following MUST be identical across ALL profiles:
|
|||||||
|
|
||||||
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
|
### Rule 8: GPU Workload Distribution (UPDATED 2026-07-16)
|
||||||
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
|
- **RTX 3090 (24GB, 128K ctx, qwen3.6-27B-code)**: Heavy reasoning, code gen, long conversations
|
||||||
- **RTX 5070 (12GB, 128K ctx, gemma-4-12b)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
- **RTX 5070 (12GB, 128K ctx, gpu-vision)**: Vision, web search, quick tasks, web_extract (IQ4_NL+MTP, ~65% VRAM at 128K)
|
||||||
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
|
- **Strix Halo (64GB, 128K ctx, syslog-auto)**: Context compression, summarization, long docs
|
||||||
- Agent profiles MUST route auxiliary tasks to the correct GPU:
|
- Agent profiles MUST route auxiliary tasks to the correct GPU:
|
||||||
- `auxiliary.vision.model: gemma-4-12b` (RTX 5070)
|
- `auxiliary.vision.model: gpu-vision` (RTX 5070)
|
||||||
- `auxiliary.web_extract.model: gemma-4-12b` (RTX 5070)
|
- `auxiliary.web_extract.model: gpu-vision` (RTX 5070)
|
||||||
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
|
- `auxiliary.compression.model: syslog-auto` (Strix Halo)
|
||||||
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
|
- Default model (`model.default`) and custom_provider remain `syslog-auto` for auto-routing
|
||||||
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
|
- For 128K context window: `threshold: 0.65` (fires at ~85K tokens)
|
||||||
@@ -297,8 +298,8 @@ The following MUST be identical across ALL profiles:
|
|||||||
- **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10.
|
- **Koby Exception**: Per captain ruling 2026-08-11, Koby is a DeepSeek-primary external agent; its primary model remains `deepseek-v4-flash` (via api.deepseek.com to preserve DeepSeek-specific reasoning, while other sections follow Rule 10.
|
||||||
- **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto`
|
- **Hermes agents**: `model.default: syslog-auto`, `custom_providers[0].model: syslog-auto`
|
||||||
- **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json`
|
- **pi agents**: `defaultModel: syslog-auto` in `settings.json`, first model in `models.json`
|
||||||
- `syslog-auto` is the LiteLLM routing model — it load-balances between strix-moe
|
- `syslog-auto` is the LiteLLM routing model — it load-balances across the live pool
|
||||||
and qwen3.6-27B-code, with gemma-4-12b as fallback. Using it protects against:
|
(see CT 116 `/opt/inference-harness/litellm_config.yaml` for the current members and weights). Using it protects against:
|
||||||
- Model name typos that cause 403 errors and silent worker failures
|
- Model name typos that cause 403 errors and silent worker failures
|
||||||
- Single GPU downtime (routing falls back automatically)
|
- Single GPU downtime (routing falls back automatically)
|
||||||
- Key/model authorization mismatches
|
- Key/model authorization mismatches
|
||||||
|
|||||||
@@ -169,16 +169,20 @@ The agent picks up the new key via `infisical run --` at gateway startup.
|
|||||||
|
|
||||||
**Keys are permanent and use bare agent name aliases.**
|
**Keys are permanent and use bare agent name aliases.**
|
||||||
|
|
||||||
- **Duration**: `null` — keys never expire. This is enforced by `default_key_generate_params` in `litellm_config.yaml`.
|
- **Duration**: `null` — keys never expire. NOT enforced today: CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key generated with no explicit models comes back with an empty models list. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||||
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
|
- **Alias convention**: bare agent name only (e.g., `tanko`, `mumuni`, `koby`, `koonimo`). No dates, no versions. The alias IS the identity.
|
||||||
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
|
- **Rotation triggers**: compromise, personnel departure, or quarterly security hygiene. NOT calendar-driven.
|
||||||
- **Max budget**: $100 per key (config default).
|
- **Max budget**: $100 per key (config default).
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
# In litellm_config.yaml — ensures all future keys inherit these defaults:
|
# NOT currently set in the authority; recommended value. CT 116 litellm_config.yaml has no
|
||||||
|
# default_key_generate_params block today, and a key generated with no explicit models comes back
|
||||||
|
# with an EMPTY models list. `models` is a literal key-generation parameter, so this is a value to
|
||||||
|
# ADD — re-read the live registry at CT 116 /opt/inference-harness/litellm_config.yaml and
|
||||||
|
# re-verify before applying.
|
||||||
litellm_settings:
|
litellm_settings:
|
||||||
default_key_generate_params:
|
default_key_generate_params:
|
||||||
models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b"]
|
models: ["syslog-auto", "qwen3.6-27B-code", "gpu-vision"]
|
||||||
duration: null # ← permanent
|
duration: null # ← permanent
|
||||||
max_budget: 100
|
max_budget: 100
|
||||||
metadata:
|
metadata:
|
||||||
@@ -286,13 +290,13 @@ auxiliary:
|
|||||||
api_key: sk-<agent-key-from-vault> # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain)
|
api_key: sk-<agent-key-from-vault> # ← workaround (get via: infisical secrets get LITELLM_API_KEY --project=agents --env=production --plain)
|
||||||
api_key_env: LITELLM_API_KEY
|
api_key_env: LITELLM_API_KEY
|
||||||
base_url: http://192.168.68.116/litellm/v1
|
base_url: http://192.168.68.116/litellm/v1
|
||||||
model: gemma-4-12b
|
model: gpu-vision
|
||||||
provider: harness
|
provider: harness
|
||||||
compression:
|
compression:
|
||||||
api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
|
api_key: sk-<agent-key-from-vault> # ← workaround (same as above)
|
||||||
api_key_env: LITELLM_API_KEY
|
api_key_env: LITELLM_API_KEY
|
||||||
base_url: http://192.168.68.116/litellm/v1
|
base_url: http://192.168.68.116/litellm/v1
|
||||||
model: gemma-4-12b
|
model: syslog-auto
|
||||||
provider: harness
|
provider: harness
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -23,7 +23,7 @@ management, and prompt caching — without sacrificing agent capability.
|
|||||||
.123, any others on .129/.122) including compression, model, context_window,
|
.123, any others on .129/.122) including compression, model, context_window,
|
||||||
prompt_caching, memory settings
|
prompt_caching, memory settings
|
||||||
- `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080,
|
- `gpu-health`: health check response from all 3 GPU backends (strix-moe .15:8080,
|
||||||
qwen .8:8080, gemma .110:8080)
|
gpu-dense .8:8080, gpu-vision .110:8080)
|
||||||
|
|
||||||
### Maintains
|
### Maintains
|
||||||
|
|
||||||
@@ -56,8 +56,8 @@ duration.
|
|||||||
**Context is the root cause.** Every ~46K prompt token costs ~87s of
|
**Context is the root cause.** Every ~46K prompt token costs ~87s of
|
||||||
prefill time at 532 tok/s. Fix context first, routing second.
|
prefill time at 532 tok/s. Fix context first, routing second.
|
||||||
|
|
||||||
- **Route by task**: qwen for code/standard queries; gemma for
|
- **Route by task**: gpu-dense for code/standard queries; gpu-vision for
|
||||||
compression/auxiliary; strix-moe for compression tasks.
|
vision/web-auxiliary; syslog-auto for compression.
|
||||||
- **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should
|
- **Compress aggressively**: threshold at 40% (not 65%) — a 128K window should
|
||||||
compact at 51K, not 85K. Target 15% tail (not 30%).
|
compact at 51K, not 85K. Target 15% tail (not 30%).
|
||||||
- **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these
|
- **Cache everything repeated**: system prompts, skill docs, AGENTS.md — these
|
||||||
@@ -96,8 +96,10 @@ call enable-prompt-caching
|
|||||||
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
|
hosts: [192.168.68.15, 192.168.68.8, 192.168.68.110]
|
||||||
|
|
||||||
-- Phase 5: Verify end-to-end latency
|
-- Phase 5: Verify end-to-end latency
|
||||||
|
-- `models` is a literal verification parameter (a snapshot only): the authoritative registry is
|
||||||
|
-- CT 116 /opt/inference-harness/litellm_config.yaml; re-read it before use.
|
||||||
|
|
||||||
call verify-latency
|
call verify-latency
|
||||||
host: 192.168.68.116
|
host: 192.168.68.116
|
||||||
models: [syslog-auto, qwen3.6-27B-code, gemma-4-12b]
|
models: [syslog-auto, qwen3.6-27B-code, gpu-vision]
|
||||||
```
|
```
|
||||||
|
|||||||
@@ -222,7 +222,7 @@ description: >
|
|||||||
|
|
||||||
**Prometheus targets**:
|
**Prometheus targets**:
|
||||||
- 192.168.68.8:9400 (RTX 3090 — qwen)
|
- 192.168.68.8:9400 (RTX 3090 — qwen)
|
||||||
- 192.168.68.110:9400 (RTX 5070 — gemma)
|
- 192.168.68.110:9400 (RTX 5070 — gpu-vision)
|
||||||
- 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4)
|
- 192.168.68.15:9400 (Strix Halo — qwen3.6-35B-udq4)
|
||||||
- harness-litellm:4000 (LiteLLM health)
|
- harness-litellm:4000 (LiteLLM health)
|
||||||
|
|
||||||
|
|||||||
@@ -67,8 +67,10 @@ description: >
|
|||||||
4. **If action == "create"**:
|
4. **If action == "create"**:
|
||||||
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
|
- Generate new key with key_alias: "{agent_name}" (e.g., "tanko" — bare name, no date)
|
||||||
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
|
- Set metadata: { "agent": "{agent_name}", "purpose": "agent-inference" }
|
||||||
- Duration is null (permanent) — inherited from litellm default_key_generate_params
|
- Duration is whatever the caller passes; NO default enforcement exists today (CT 116 `litellm_config.yaml` has no `default_key_generate_params` block, and a key with no explicit models returns an empty models list). Agent keys are permanent by policy, not by that block. OPEN policy question: should agent keys expire by default? (captain security-policy decision, raised separately.)
|
||||||
- Set models: ["syslog-auto", "qwen3.6-27B-code", "gemma-4-12b", "strix-moe", "gpu-dense", "gpu-light", "qwen3.6-35B-udq4"]
|
- Set models: read the live key-scoped set rather than hardcoding one — `/v1/models` is key-scoped,
|
||||||
|
and the authoritative registry is CT 116 `/opt/inference-harness/litellm_config.yaml`. Do not add
|
||||||
|
retired names (`gemma-4-12b`, `gpu-light`, `crew-auto` — all retired 2026-09-12).
|
||||||
- Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed).
|
- Note: `ornith-1.0-35b` is NOT a valid LiteLLM model name (use `strix-moe`, the stable alias). qwen3.6-35B-A3B removed from fleet (was never deployed).
|
||||||
- Return the new key
|
- Return the new key
|
||||||
5. **If action == "rotate"**:
|
5. **If action == "rotate"**:
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ description: >
|
|||||||
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
|
| syslog-auto | 28.8s | 25.4s | 68 calls took 30-120s; tail to ~300s under load |
|
||||||
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
|
| qwen3.6-27B-code | 23.0s | — | same backend class as syslog-auto |
|
||||||
| strix-moe | 7.5s | — | Strix Halo, healthy |
|
| strix-moe | 7.5s | — | Strix Halo, healthy |
|
||||||
| gemma-4-12b | 2.6s | — | RTX 5070, healthy |
|
| gemma-4-12b (retired 2026-09-12; RTX 5070 now `gpu-vision`) | 2.6s | — | RTX 5070, healthy |
|
||||||
|
|
||||||
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
|
Sep 6 incident timeline: failures 04:00-07:00 EDT (0% GPU util = wedged
|
||||||
backend), full recovery 07:00-08:00 with ZERO client failures once requests
|
backend), full recovery 07:00-08:00 with ZERO client failures once requests
|
||||||
@@ -53,8 +53,7 @@ proxy queuing.
|
|||||||
|
|
||||||
### 2. Auxiliary tasks — keep template timeouts, one correction
|
### 2. Auxiliary tasks — keep template timeouts, one correction
|
||||||
|
|
||||||
- vision: 60s (keep), web_extract: 30s (keep) — gemma-4-12b averages 2.6s;
|
- vision: 60s (keep), web_extract: 30s (keep) — the 2.6s average was measured on `gemma-4-12b` (retired 2026-09-12); the live RTX 5070 alias is `gpu-vision`.
|
||||||
these are fine.
|
|
||||||
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
|
- compression: 300s (keep — this was already raised from 60 per gpu-fleet).
|
||||||
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
|
- **gpu-dense delegation/x_search: set timeout >= 120s.** The RTX 3090
|
||||||
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
|
(qwen3.6-27B-code backend, 23.0s avg) is the same speed class as
|
||||||
|
|||||||
@@ -0,0 +1,191 @@
|
|||||||
|
"""Regression tests for the 2026-09-12 retired-alias sweep in audit-hermes-config.py.
|
||||||
|
|
||||||
|
WHY THIS FILE EXISTS: the executable audit pushed agent configs toward a DEAD alias.
|
||||||
|
Rule 8 required `auxiliary.vision.model == "gpu-light"` and
|
||||||
|
`auxiliary.web_extract.model == "gpu-light"`, but `gpu-light` (and its raw predecessor
|
||||||
|
`gemma-4-12b`) were retired on 2026-09-12 and now return 400 `Invalid model name`; the
|
||||||
|
live RTX 5070 alias is `gpu-vision`. A config that adopted the correct canonical alias
|
||||||
|
therefore FAILED our own audit, so the audit was actively enforcing a broken config.
|
||||||
|
|
||||||
|
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert
|
||||||
|
observable behaviour — exit code and the emitted rule message — for the live alias and
|
||||||
|
for both retired names. No network, vault, or SSH access is required.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import pathlib
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||||
|
AUDIT = ROOT / "audit-hermes-config.py"
|
||||||
|
|
||||||
|
BASE = """
|
||||||
|
model:
|
||||||
|
api_key: ""
|
||||||
|
api_key_env: LITELLM_API_KEY
|
||||||
|
base_url: http://192.168.68.116/v1
|
||||||
|
max_tokens: 4096
|
||||||
|
default: syslog-auto
|
||||||
|
provider: harness
|
||||||
|
fallback_providers:
|
||||||
|
provider: deepseek
|
||||||
|
model: deepseek-v4-flash
|
||||||
|
api_key_env: DEEPSEEK_API_KEY
|
||||||
|
compression:
|
||||||
|
model: syslog-auto
|
||||||
|
provider: harness
|
||||||
|
threshold: 0.65
|
||||||
|
max_context_window: 131072
|
||||||
|
auxiliary:
|
||||||
|
vision:
|
||||||
|
model: {alias}
|
||||||
|
provider: harness
|
||||||
|
web_extract:
|
||||||
|
model: {alias}
|
||||||
|
provider: harness
|
||||||
|
compression:
|
||||||
|
model: syslog-auto
|
||||||
|
provider: harness
|
||||||
|
delegation:
|
||||||
|
provider: harness
|
||||||
|
custom_providers:
|
||||||
|
- name: harness
|
||||||
|
key_env: LITELLM_API_KEY
|
||||||
|
base_url: http://192.168.68.116/v1
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def _run_config(tmp_path, name, text):
|
||||||
|
cfg = tmp_path / name
|
||||||
|
cfg.write_text(text)
|
||||||
|
proc = subprocess.run(
|
||||||
|
[sys.executable, str(AUDIT), str(cfg)],
|
||||||
|
capture_output=True, text=True,
|
||||||
|
)
|
||||||
|
return proc.returncode, proc.stdout
|
||||||
|
|
||||||
|
|
||||||
|
def _run(tmp_path, alias):
|
||||||
|
return _run_config(tmp_path, f"{alias}.yaml", BASE.format(alias=alias))
|
||||||
|
|
||||||
|
|
||||||
|
def test_live_canonical_alias_passes(tmp_path):
|
||||||
|
"""The RTX 5070 alias that actually resolves must satisfy Rule 8."""
|
||||||
|
code, out = _run(tmp_path, "gpu-vision")
|
||||||
|
assert code == 0, out
|
||||||
|
assert "RESULT: PASS" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_gpu_light_is_rejected(tmp_path):
|
||||||
|
"""A config pinned to the retired alias must fail, not pass."""
|
||||||
|
code, out = _run(tmp_path, "gpu-light")
|
||||||
|
assert code == 1, out
|
||||||
|
assert "auxiliary.vision.model must be gpu-vision" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_gemma_is_rejected(tmp_path):
|
||||||
|
"""The retired raw model name must fail Rule 8 as well."""
|
||||||
|
code, out = _run(tmp_path, "gemma-4-12b")
|
||||||
|
assert code == 1, out
|
||||||
|
assert "auxiliary.vision.model must be gpu-vision" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_corrected_compression_example_passes(tmp_path):
|
||||||
|
"""The corrected workaround (vision=gpu-vision, compression=syslog-auto) must PASS."""
|
||||||
|
code, out = _run(tmp_path, "gpu-vision")
|
||||||
|
assert code == 0, out
|
||||||
|
assert "[Rule 7] compression.model must be syslog-auto (got 'syslog-auto')" in out
|
||||||
|
assert "[Rule 7] auxiliary.compression.model must be syslog-auto (got 'syslog-auto')" in out
|
||||||
|
assert "RESULT: PASS" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_delegation_is_rejected(tmp_path):
|
||||||
|
"""delegation.model has no dedicated value rule, so a retired name there used to PASS."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"delegation-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
"delegation:\n provider: harness",
|
||||||
|
"delegation:\n provider: harness\n model: gpu-light",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "delegation.model = 'gpu-light' is retired" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_custom_providers_is_rejected(tmp_path):
|
||||||
|
"""custom_providers[*].model is model-bearing; a retired name there must fail."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"custom-provider-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
" - name: harness\n key_env: LITELLM_API_KEY",
|
||||||
|
" - name: harness\n model: gpu-light\n key_env: LITELLM_API_KEY",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "custom_providers[0].model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_raw_but_live_alias_warns_but_passes(tmp_path):
|
||||||
|
"""Raw-but-live names resolve (200), so they warn only; failing them rejects valid configs."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"raw-qwen.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
"delegation:\n provider: harness",
|
||||||
|
"delegation:\n provider: harness\n model: qwen3.6-27B-code",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 0, out
|
||||||
|
assert "delegation.model = 'qwen3.6-27B-code' is a raw-but-live model name" in out
|
||||||
|
assert "prefer the stable alias gpu-dense" in out
|
||||||
|
assert "RESULT: PASS" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_fallback_providers_is_rejected(tmp_path):
|
||||||
|
"""fallback_providers.model is model-bearing; a retired name there must fail."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"fallback-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(" model: deepseek-v4-flash", " model: gpu-light"),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "fallback_providers.model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_x_search_is_rejected(tmp_path):
|
||||||
|
"""x_search.model was previously not enumerated; the derivation must catch it."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"x-search-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
"delegation:\n provider: harness",
|
||||||
|
"delegation:\n provider: harness\nx_search:\n model: gpu-light",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "x_search.model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
|
|
||||||
|
|
||||||
|
def test_retired_alias_in_nested_auxiliary_block_is_rejected(tmp_path):
|
||||||
|
"""A nested auxiliary sub-block outside the named three must still be derived."""
|
||||||
|
code, out = _run_config(
|
||||||
|
tmp_path,
|
||||||
|
"nested-aux-gpu-light.yaml",
|
||||||
|
BASE.format(alias="gpu-vision").replace(
|
||||||
|
" compression:\n model: syslog-auto\n provider: harness\ndelegation:",
|
||||||
|
" compression:\n model: syslog-auto\n provider: harness\n"
|
||||||
|
" tasks:\n summarize:\n model: gpu-light\ndelegation:",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
assert code == 1, out
|
||||||
|
assert "auxiliary.tasks.summarize.model = 'gpu-light'" in out
|
||||||
|
assert "RESULT: FAIL" in out
|
||||||
Reference in New Issue
Block a user