diff --git a/abiba-zulip-restore.prose.md b/abiba-zulip-restore.prose.md index a02556d..dd0f8ac 100644 --- a/abiba-zulip-restore.prose.md +++ b/abiba-zulip-restore.prose.md @@ -1,4 +1,6 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 kind: function name: abiba-zulip-restore description: > @@ -11,6 +13,7 @@ version: 1.0.0 status: active runtime_contract: 2 --- +--- # Abiba Zulip Restore — Resume pi Zulip Communication @@ -304,6 +307,7 @@ module.exports = { }; ``` +--- --- **Last verified good state**: 2026-07-13 — Extension v2 running via `pi --mode rpc`, health endpoint :9200 returning `{status:"ok",connected:true}`, queue a669f21e. diff --git a/contract-registry.yaml b/contract-registry.yaml index c7ae3b5..dc27ba8 100644 --- a/contract-registry.yaml +++ b/contract-registry.yaml @@ -1867,3 +1867,73 @@ contracts: last_run: null last_status: null drift_alerts: [] +# Koby Report-Only Registry (2026-08-17 — Captain) +# ⛔ KOBY IS NEVER REPAIRED — detect + report, never fix on .129 +koby_report_only: true +koby_host: "CT 111 (tdunna)" +koby_ip: ".129" +koby_user: "Theo" + +# Contracts that should be Koby-aware (detect only, no heal path) +koby_aware_contracts: + - name: pm2-self-heal + path: pm2-self-heal.prose.md + koby_action: skip_heal + koby_note: "Koby PM2 processes reported to Zulip, never auto-restarted on .129" + + - name: zulip-health + path: zulip-health.prose.md + koby_action: skip_heal + koby_note: "Koby Zulip bridge issues reported to Zulip, never repaired on .129" + + - name: hermes-zulip-restore + path: hermes-zulip-restore.prose.md + koby_action: skip_heal + koby_note: "Koby Zulip restoration skipped, only diagnostic alerts" + + - name: abiba-zulip-restore + path: abiba-zulip-restore.prose.md + koby_action: skip_heal + koby_note: "Abiba-Zulip restoration not applicable to Koby" + + - name: litellm-self-heal + path: litellm-self-heal.prose.md + koby_action: skip_heal + koby_note: "Koby LiteLLM issues reported, never fixed on .129" + + - name: disk-gc-threat-response + path: disk-gc-threat-response.prose.md + koby_action: skip_heal + koby_note: "Koby disk GC threats reported, never executed on .129" + + - name: memory-fixer + path: memory-fixer.prose.md + koby_action: skip_heal + koby_note: "Koby memory issues reported, never fixed on .129" + + - name: memory-audit-maintenance + path: memory-audit-maintenance.prose.md + koby_action: skip_heal + koby_note: "Koby memory audits reported, never performed on .129" + + - name: gpu-self-heal + path: gpu-self-heal.prose.md + koby_action: skip_heal + koby_note: "Koby GPU issues reported, never fixed on .129" + + - name: gpu-monitor + path: gpu-monitor.prose.md + koby_action: skip_heal + koby_note: "Koby GPU monitoring reports only, never repairs on .129" + + - name: agent-health-check + path: agent-health-check.prose.md + koby_action: skip_heal + koby_note: "Koby agent health checks reported, never repairs on .129" + +# Scripts that should skip Koby +koby_aware_scripts: + - name: agent-health-check.py + path: scripts/agent-health-check.py + koby_action: skip_heal + koby_note: "Script should only run diagnostics on Koby, not repairs" diff --git a/disk-gc-threat-response.prose.md b/disk-gc-threat-response.prose.md index fa0cf34..34243d6 100644 --- a/disk-gc-threat-response.prose.md +++ b/disk-gc-threat-response.prose.md @@ -1,4 +1,6 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 kind: responsibility name: disk-gc-threat-response description: > @@ -11,6 +13,7 @@ description: > id: 067NV8KJ03ZG71S44N41F31022 version: 1.0.0 --- +--- # Disk GC & Threat Response @@ -301,4 +304,4 @@ one-off GPU builds. No automated post-migration cleanup was in place. |------|-----|------|--------| | docker-vm | 192.168.68.7 | 16 Docker containers, 4 stacks | ✅ reachable | -> **Note:** CT 118 is now jdownloader (active on storepve). CT 119 (infisical-vault) added on minipve.\n> **Migrated:** CT 101 → .8, CT 103 → .110 (bare metal GPU).\n> **KVM VM:** CT 109 (docker-vm) is a KVM VM, not LXC — access via SSH .7. \ No newline at end of file +> **Note:** CT 118 is now jdownloader (active on storepve). CT 119 (infisical-vault) added on minipve.\n> **Migrated:** CT 101 → .8, CT 103 → .110 (bare metal GPU).\n> **KVM VM:** CT 109 (docker-vm) is a KVM VM, not LXC — access via SSH .7. diff --git a/gpu-fleet.prose.md b/gpu-fleet.prose.md index f16ec95..f3b4aab 100644 --- a/gpu-fleet.prose.md +++ b/gpu-fleet.prose.md @@ -14,9 +14,6 @@ description: > Instability observed near 100K at 256K (now all GPUs at 128K). 128K is the stable ceiling. For larger context needs → fall back to external providers (deepseek). VRAM headroom improved: RTX 3090 ~70%, RTX 5070 ~65%. - UPDATED 2026-08-15: gpu-dense swapped to Qwen3.8-27B-Uncensored-Q4_K_M - (~16.8GB, 128K ctx, --spec-type draft-mtp (v2)). Served on .8:8080 under the legacy - alias qwen3.6-27B-code for LiteLLM routing continuity. Replaces SmartCode-Fable-5-27B. agent: abiba triggers: - on model add/remove @@ -74,7 +71,7 @@ triggers: │ RTX 3090 │ │ RTX 5070 │ │ Strix Halo│ │ GPU Monitor │ │ 24GB │ │ 12GB │ │ 64GB UMA │ │ :9100 │ │ 128K ctx │ │ 128K ctx │ │ 128K ctx │ │ Watchdog │ -│ qwen3.6 │ │ gemma-4-12b │ │ qwen3.6 │ │ Prometheus │ +│ qwen3.6 │ │ Qwen3.5-9B │ │ qwen3.5 │ │ Prometheus │ │ 27B-code │ │ :8080 │ │ -35B-udq4 │ │ exporter │ │ :8080 │ │ :9400 (exp) │ │ :9400(exp)│ │ :9401 │ │ :9400 │ └─────────────┘ └───────────┘ └──────────────┘ @@ -92,16 +89,13 @@ When a model is swapped on a GPU, ONLY the infrastructure layer changes — agen | `gpu-dense` | RTX 3090 (.8) | Qwen3.8-27B-Uncensored-Q4_K_M | Whatever runs on RTX 3090 | | `gpu-light` | RTX 5070 (.110) | gemma-4-12b | Whatever runs on RTX 5070 | -**Backward compatibility**: Old model-specific names (qwen3.6-27B-code, gemma-4-12b, qwen3.6-35B-udq4) still work +**Backward compatibility**: Old model-specific names (qwen3.6-27B-code, gemma-4-12b, qwen3.5-9b-it) still work but are deprecated for agent configs. Only the stable aliases survive model swaps. ## Current Model Assignments (2026-07-15) | Model | GPU | Host | VRAM | Ctx | KV Cache | Parallel | Batch/Ubatch | Status | |-------|-----|------|------|-----|----------|----------|-------------|--------| -| Qwen3.8-27B-Uncensored-Q4_K_M | RTX 3090 | .8 (llm-gpu) | ~16.8/24.6GB | **128K** | turbo4 | 1 | 2048/1024 | ✅ healthy | -| gemma-4-12b | RTX 5070 | .110 (ocu-llm) | ~7.8/12.2GB (65%) | 128K | q4_0 | 2 | 2048/1024 | ✅ healthy | -| qwen3.6-35B-udq4 | Strix Halo Vulkan | .15 (amdpve) | ~22GB/64GB | 128K | q4_0 | 1 | 4096/1024 | ✅ 65 tok/s | ## Routing Configuration (LiteLLM — July 2026) @@ -110,8 +104,6 @@ but are deprecated for agent configs. Only the stable aliases survive model swap | Model | GPU | Weight | RPM Cap | Timeout | |-------|-----|--------|---------|---------| | Qwen3.8-27B-Uncensored-Q4_K_M | RTX 3090 (.8:8080) | **0.55** | 500 | **300s** | -| qwen3.6-35B-udq4 | Strix Halo (.15:8080) | **0.30** | 60 | **300s** | -| gemma-4-12b | RTX 5070 (.110:8080) | **0.15** | 200 | **120s** | Note: All syslog-auto entries route directly to GPUs with `api_key: not-needed`. The router (port 9000) is NOT in the inference path. @@ -119,9 +111,6 @@ Note: All syslog-auto entries route directly to GPUs with `api_key: not-needed`. | Model | RPM Cap | Notes | |-------|---------|-------| -| strix-moe (qwen3.6-35B-udq4) | 40 | Tight cap — prevents Strix overload | -| Qwen3.8-27B-Uncensored-Q4_K_M | 500 | High cap — primary workhorse (replaces qwen3.6-27B-code) | -| gemma-4-12b | 500 | High cap — IQ4_NL+MTP, 122 tok/s | ### Stable Aliases (for agent configs — never change) @@ -192,7 +181,7 @@ Show full fleet status: GPUs, models, VRAM, context windows, parallel slots, act 3. Check LiteLLM: `curl http://192.168.68.116/health` (expect "I'm alive!") 4. Check LiteLLM models: `curl -H "Authorization: Bearer $MASTER_KEY" http://192.168.68.116/v1/models` 5. Check LiteLLM timeouts: `grep -n 'timeout:' /opt/inference-harness/litellm_config.yaml` - - gemma-4-12b: 120s, qwen3.6-27B-code: 300s, qwen3.6-35B-udq4/strix-moe: 300s (strix-moe does NOT exist — legacy name, do not use) + - Qwen3.5-9B: 120s, qwen3.6-27B-code: 300s, Carnice-Qwen3.6-MoE-35B-A3B/strix-moe: 300s (strix-moe alias retained, legacy name qwen3.6-35B-udq4 deprecated) - global request_timeout: 300s, nginx proxy_read_timeout: 600s 6. Check AMD metrics: `curl http://192.168.68.15:9400/metrics` (Radeon 8060S, util%, VRAM, temp, power) 7. Check port conflicts: verify only one llama-server on :8080 per host @@ -249,14 +238,6 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. - **Router startup race**: Compose router.py doesn't call load_roster(). Reload thread sleeps 30s first. Fix: trigger roster reload via SSH after restart, or rebuild image with startup load_roster(). - **LiteLLM /metrics**: Requires auth. Prometheus uses `/health/liveliness` as workaround. -- **VRAM (2026-07-15)**: RTX 3090 at ~17/24.6GB (~70%) with **128K context** (reduced from 256K 2026-07-17). RTX 5070 at ~7.8/12.2GB (~65%) with 128K context + MTP. Strix Halo at ~7GB/64GB. -- **RTX 3090 (2026-08-15)**: Swapped to Qwen3.8-27B-Uncensored-Q4_K_M (~16.8GB, replaced - SmartCode-Fable-5-27B-UD-Q3_K_XL). Service: `/home/llmuser/llama-fable-wrapper.sh`. - Served under alias `qwen3.6-27B-code` for LiteLLM routing continuity. -- **RTX 5070 config (2026-07-15)**: Switched to IQ4_NL + MTP draft (Q8_0) at 128K context. Gen speed: 122 tok/s. VRAM: ~7.8/12.2GB (~65%). Service: `/home/llmuser/llama-wrapper.sh`. Config: `--model gemma-4-12b-it-IQ4_NL.gguf --spec-draft-model gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --ctx-size 131072`. -- **LiteLLM timeout tuning (verified 2026-08-16)**: Qwen3.8-27B (alias qwen3.6-27B-code) - 300s, gemma-4-12b 120s, qwen3.6-35B-udq4 300s, strix-moe 300s, syslog-auto routes all - 300s. Nginx proxy_read_timeout: 600s. Global request_timeout: 300s. - **Strix Halo GPU**: Vulkan is the working backend (ROCm/HIP path abandoned — HSA runtime blocked on Debian 13). Build at `/root/llama.cpp/build-vk/`, commit `4fc4ec5` (2026-07-01), ggml 0.15.3 shared-lib arch. Mesa RADV 25.0.7, KHR_coopmat fast path active. ~70 tok/s gen, 532 tok/s prompt. Service: `strix-server.service` on port 8080, model: `qwen3.6-35B-udq4`, alias `strix-moe`, 128K context, flash-attn + q4 KV, multimodal (mmproj loaded). - **Port conflict detection (2026-07-05)**: All 3 GPU wrappers now detect ghost processes squatting port 8080 before starting. `.8` and `.110` use inline pre-start check in `llama-wrapper.sh`; `.15` uses `/usr/local/bin/port-cleanup.sh` ExecStartPre. Replaces the blanket `pkill -9 -x llama-server` on .15 which would kill ALL llama-server instances regardless of port. Ghost detection was the root cause of .8 crash-looping for 27+ restarts (stale pid 25836 squatting 8080 after OOM kill). - **Strix Halo thermal safeguard (2026-07-02)**: `strix-server.service` has `-n 8192` (hard generation cap per request). Without it, `--predict` defaults to -1 (infinity) — a runaway request from .123 (old Mumuni CT114 — now inside Abiba CT100 at .24) decoded 39,868 tokens over 24 min, pushing Tctl to 98°C (crit 89.8°C) and throttling 70→29 t/s. The cap bounds worst-case generation to ~5 min. Do NOT remove `-n` without a replacement ceiling. Sustained load hits ~84°C even at 92s; the APU is fanless/low-flow. Clients MUST also set `max_tokens`. @@ -272,7 +253,6 @@ If no SSH access, send Zulip DM via abiba-bot with vault update instructions. | GPU | Model | Gen tok/s | Prompt tok/s | Baseline | Context | |-----|-------|-----------|--------------|----------|---------| | RTX 3090 (.8) | Qwen3.8-27B-Uncensored-Q4_K_M | **TBD** | — | — | **128K** | -| RTX 5070 (.110) | gemma-4-12b (IQ4_NL+MTP) | **191** | — | — | **128K** | | Strix Halo (.15) | qwen3.6-35B-udq4 | **65** | 140 | — | **128K** | Benchmarks from 2026-07-17. Strix Halo model: qwen3.6-35B-udq4. RTX 5070 MTP provides 2.7x speedup over pre-upgrade 70 tok/s. @@ -299,7 +279,8 @@ When the underlying model is swapped, only the LiteLLM config changes — agent ### Context Windows - RTX 3090: **128K** (reduced from 256K 2026-07-17) | RTX 5070: **128K** (reduced from 256K) | Strix Halo: **128K** - **All agents**: 128K ceiling — stable margin. For >128K workloads, use external providers (deepseek) -- Compression threshold 0.65: fires at ~85K (~43K headroom before 128K ceiling) +- Compression threshold 0.60: fires at ~77K (~51K headroom before 128K ceiling) +- **Pi agents (Abiba)**: `compaction.reserveTokens: 52739` (≈60% of 128K) - Mumuni compression model alias: `strix-moe` with 300s timeout ### Mumuni Agent Profile @@ -316,7 +297,7 @@ Mumuni (kagentz CT105, 192.168.68.14 — migrated from CT100 2026-08-29) is the | `aux.web_extract.model` | `gpu-light` | Web extraction | | `delegation.model` | `gpu-dense` | Sub-agent reasoning (RTX 3090) | | `context.max_context_window` | 131072 (128K) | Reduced from 256K 2026-07-17 — stable 128K ceiling | -| `compression.threshold` | 0.65 | Triggers at ~85K | +| `compression.threshold` | 0.60 | Triggers at ~77K (~60% of 128K) — optimized for 128K context | | `compression.target_ratio` | 0.3 | Compresses to ~38K | | `compression.protect_last_n` | 40 | Preserves last 40 messages | | `memory.memory_char_limit` | 800 | Brief memory entries | diff --git a/gpu-self-heal.prose.md b/gpu-self-heal.prose.md index 6afe6d5..1161933 100644 --- a/gpu-self-heal.prose.md +++ b/gpu-self-heal.prose.md @@ -1,4 +1,6 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 kind: responsibility name: gpu-self-heal description: > @@ -16,6 +18,7 @@ depends_on: - gpu-monitor.prose.md (live data source on .24:9100) - gpu-fleet.prose.md (source of truth for topology, aliases, model assignments) --- +--- ## Maintains @@ -38,20 +41,19 @@ depends_on: - On fix: verify with benchmark inference test before declaring resolved - Escalate: after 3 failed remediation attempts → Zulip #agent-hub alert +--- --- ## Current Fleet Baseline (2026-07-18) | Alias | GPU | Host | Model | VRAM | Ctx | tok/s | Role | |-------|-----|------|-------|------|-----|-------|------| -| `gpu-dense` | RTX 3090 24GB | ct8 (.8:8080) | Qwen3.8-27B-Uncensored-Q4_K_M (alias qwen3.6-27B-code) | ~16.8/24.6GB | 128K | — | Heavy reasoning, code gen | -| `gpu-light` | RTX 5070 12GB | ct110 (.110:8080) | HauhauCS Gemma4-12B QAT Q4_K_M + MTP draft | 10.1/12.2GB (83%) | 128K | 169.6 | Vision, web extract, light tasks | | `strix-moe` | Strix Halo 64GB | ct15 (.15:8080) | qwen3.6-35B-udq4 | ~10/64GB (16%) | 128K | 62.9 | Compression, summarization, long docs | Key notes: - All models use direct GPU routing via LiteLLM (`api_key: not-needed`). Router (port 9000) is deprecated and NOT in the inference path. - Stable aliases (gpu-dense, gpu-light, strix-moe) from gpu-fleet are the canonical names for agent configs. Model-specific names still work but are deprecated. -- RTX 5070 tok/s is 2.3x faster than RTX 3090 for its model — gpu-light is the fastest endpoint. Route vision/web/light work there first. +- RTX 5070 tok/s is ~145 for Qwen3.5-9B — gpu-light is the fastest endpoint. Route vision/web/light work there first. NOTE: Qwen3.5-9B is multimodal (image+text), NOT text-only like gemma-4-12b was. - Strix Halo is 62.9 tok/s (89% of 70.5 baseline) — below optimal but stable. Check for competing workloads. - RTX 3090 VRAM at 88% — within role-appropriate range (role = heavy reasoning, needs the headroom). - RTX 5070 VRAM at 83% — role-appropriate for vision/web (smaller batch sizes). @@ -156,7 +158,7 @@ Key notes: - **Detect**: GPU roles misaligned with hardware capabilities - **Target distribution**: - RTX 3090 (gpu-dense, 24GB, 74.9 tok/s) → Heavy reasoning, code gen, long conversations (slowest per-token but largest context capacity). Weight: 0.55 (LiteLLM). - - RTX 5070 (gpu-light, 12GB, 169.6 tok/s) → Vision/image, web search, lightweight tasks (2.3x faster than 3090 per token). Weight: 0.15 (LiteLLM). + - RTX 5070 (gpu-light, 12GB, ~145 tok/s) → Vision (image+text), web search, lightweight tasks. Weight: 0.15 (LiteLLM). - Strix Halo (strix-moe, 64GB, 62.9 tok/s) → Context compression, summarization, long docs (MoE model). Weight: 0.30 (LiteLLM). - **Note**: RTX 5070 is the fastest endpoint per token. Route high-volume, low-complexity work there first. - **Fix**: @@ -166,6 +168,7 @@ Key notes: - **Verify**: Each GPU's request pattern matches its designated role within 24h - **Escalate**: If role mismatch persists >48h → agent alias audit needed +--- --- ## Execution @@ -247,6 +250,7 @@ call update-gpu-health } ``` +--- --- ## Reporting @@ -263,6 +267,7 @@ Pushed to `SyslogSolution/health-logs/gpu/{run_id}.json` — versioned, searchab - Per-GPU tok/s trend over 7 days - Regression alerts if any GPU degrades >10% week-over-week +--- --- ## Design Decisions (Verified 2026-07-12, Reaffirmed 2026-07-18) diff --git a/hermes-agent-baseline.prose.md b/hermes-agent-baseline.prose.md index 5926446..50d0ae3 100644 --- a/hermes-agent-baseline.prose.md +++ b/hermes-agent-baseline.prose.md @@ -168,11 +168,15 @@ pct-run grep -A8 "vision:" /root/.hermes/config.yaml | grep api_key # Must show both api_key: sk-... and api_key_env: LITELLM_API_KEY ``` -### For Koby (CT 111 / tdunna) +### For Koby (CT 111 / tdunna) — **REPORT-ONLY MODE** Koby runs Hermes on CT 111 (tdunna). Config files at `/root/.hermes/config.yaml`. Same Hermes pattern as Tanko/Mumuni/Koonimo — see config sections above. +**⛔ KOBY IS NEVER REPAIRED (2026-08-17, Captain)**: Diagnostic only — detect and report, never fix on .129. +No heal step, no restart, no key rotation, no config edit, no memory rewrite, no disk GC, no service touch, no process kill — ever. +If a health check shows Koby degraded, **DO NOT** execute any repair action. Instead, report to Zulip and let Theo fix it. + **LiteLLM key**: alias `koby` in LiteLLM DB, injected via `infisical run --` wrapper. ### For pi Agents (Abiba) diff --git a/hermes-config-template.prose.md b/hermes-config-template.prose.md index 06e0400..076ff04 100644 --- a/hermes-config-template.prose.md +++ b/hermes-config-template.prose.md @@ -346,6 +346,17 @@ curl -s -o /dev/null -w 'key_health: %{http_code}\n' -H "Authorization: Bearer $ ### Rule 13: API Key Injection — Two Patterns (UPDATED 2026-07-16, WAL #1300) +### Rule 14: Hermes Context Detection Uses `max_model_tokens`, NOT `max_input_tokens` + +**CRITICAL**: Hermes context detection reads `max_model_tokens` (128K), NOT `max_input_tokens` (64K cap). + +- **Abiba and Hermes agents**: `max_model_tokens: 131072` (128K) — unlimited context +- **Crewmates (ops, tune, verify, auth-keys, build)**: `max_input_tokens: 64000` (64K) — capped +- If you see `max_input_tokens: 64000` in an Abiba/Hermes config, that's a mistake +- Using `max_input_tokens` for Hermes agents causes premature context loss +- Check: `grep -n 'max_model_tokens\|max_input_tokens' ~/.hermes/config.yaml` +- Expected output: `max_model_tokens: 131072` (not max_input_tokens) + Agents inject `LITELLM_API_KEY` via ONE of two mechanisms. Both are valid; the contract requirement is that the key is a **valid LiteLLM virtual key** (HTTP 200 on /v1/models). @@ -422,13 +433,3 @@ curl -s -o /dev/null -w '%{http_code}' -H "Authorization: Bearer $K" http://192. 5. **Set model choice** — Per agent's workload 6. **Verify** — curl all shared endpoints, test the model with the new key 7. **Report** — What was changed, preserved, custom -### Rule 16: Koby Configuration (DeepSeek-primary) -(Ref: See Rule 10 for default model behavior, with Koby exception) - -Koby uses a split-model architecture: - -- Primary Model: `deepseek-v4-flash` via `api.deepseek.com` (for reasoning) -- Auxiliary Models: `gpu-light` (vision/web_extract) and `syslog-auto` (compression) -- Key Hygiene: `api_key_env` is strictly `LITELLM_API_KEY` or `DEEPSEEK_API_KEY` -- Constraint: Do NOT touch Koby's primary model/provider/compression settings unless explicitly ruled by the captain. - diff --git a/hermes-zulip-restore.prose.md b/hermes-zulip-restore.prose.md index 8ff525c..bafb5a3 100644 --- a/hermes-zulip-restore.prose.md +++ b/hermes-zulip-restore.prose.md @@ -1,4 +1,6 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 kind: function name: hermes-zulip-restore description: > @@ -10,6 +12,7 @@ version: 1.0.0 status: active runtime_contract: 2 --- +--- # Hermes Zulip Restore — Bring Any Agent Back to Good State @@ -180,6 +183,7 @@ https://git.sysloggh.net/SyslogSolution/zulip-platform-plugins/src/branch/feat/z Commit `55ca15d` — `fix(zulip): add _strip_html for slash command matching` Pull request #33 is the primary integration branch. +--- --- **Last verified good state**: 2026-07-08 — Mumuni, Tanko, Koby all connected with `_strip_html` applied. diff --git a/infrastructure-monitoring.prose.md b/infrastructure-monitoring.prose.md index 62d8d4b..e1f8172 100644 --- a/infrastructure-monitoring.prose.md +++ b/infrastructure-monitoring.prose.md @@ -136,6 +136,79 @@ curl -s http://192.168.68.116:4001/metrics | head -20 **Report format**: Summarize actual results from each probe. If any probe returns non-200 or empty output, flag as alert. +### Phase 1: GPU Exporters + +**NVIDIA (.8 and .110)**: +1. Download `nvidia_gpu_exporter` binary +2. Create systemd service `nvidia-gpu-exporter.service` +3. Start and enable + +**AMD (.15)**: +1. Create Python exporter script at `/opt/amdgpu-exporter/exporter.py` +2. Parses `amdgpu_top --json -d 1000` output +3. Exposes key metrics at `:9400/metrics` via Python http.server +4. Create systemd service +5. Start and enable + +### Phase 2: Prometheus + +1. Create `/opt/monitoring/` directory on CT 116 +2. Write `prometheus.yml` with scrape configs for all targets +3. Add to docker-compose (or separate compose file) +4. Start container + +### Phase 3: Grafana + +1. Create `/opt/monitoring/grafana/` directories +2. Provision Prometheus datasource +3. Provision GPU fleet dashboard JSON +4. Provision LiteLLM dashboard JSON +5. Add to docker-compose +6. Start container + +### Phase 4: Verification + +1. Verify all 3 GPU exporters return 200 at :9400/metrics +2. Verify Prometheus targets all UP at :9090/targets +3. Verify Grafana accessible at :3001 with dashboards +4. Verify LiteLLM metrics flowing to Prometheus +5. ~~Update nginx to proxy `/monitoring/` → Grafana~~ (NOT recommended — nginx sub-path was tried for /grafana/ and reverted per proxmox-monitor; direct :3001 access is the standard) + +## Execution + +### check-health + +**RUN LIVE, NEVER ECHO — every dispatch must execute the probes below with real tool calls; never repeat a prior report unless a live probe fails.** + +```bash +# Zulip API health (POST ping) +curl -s -o /dev/null -w '%{http_code}' -X POST https://chat.sysloggh.net/api/v1/messages -u 'abiba-bot@chat.sysloggh.net:KEY' +# Expected: 200 (HTTP 000 = unreachable/cache) + +# PM2 process health +pm2 jlist +# Expected: 5/5 online (abiba-telegram, abiba-zulip, zulip-watchdog, gitea-runner, spoton-service) + +# GPU exporters (may be down per DEPLOYMENT STATUS) +curl -s http://192.168.68.8:9400/metrics && echo " - OK" || echo " - FAIL" +curl -s http://192.168.68.110:9400/metrics && echo " - OK" || echo " - FAIL" +curl -s http://192.168.68.15:9400/metrics && echo " - OK" || echo " - FAIL" + +# Prometheus targets +curl -s http://192.168.68.116:9090/api/v1/targets | jq '.data.activeTargets' +# Expected: All targets UP (may show some down if exporters not deployed) + +# Grafana health +curl -s http://192.168.68.116:3001/api/health | jq '{status, version}' +# Expected: {"status":"ok","version":"..."} + +# LiteLLM metrics +curl -s http://192.168.68.116:4001/metrics | head -20 +# Expected: Prometheus-formatted metrics output +``` + +**Report format**: Summarize actual results from each probe. If any probe returns non-200 or empty output, flag as alert. + ### Phase 1: GPU Exporters **NVIDIA (.8 and .110)**: diff --git a/litellm-self-heal.prose.md b/litellm-self-heal.prose.md index 2d01c35..c7680a8 100644 --- a/litellm-self-heal.prose.md +++ b/litellm-self-heal.prose.md @@ -1,4 +1,6 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 kind: responsibility name: litellm-self-heal status: deployed @@ -22,6 +24,7 @@ description: > inference, and agent keys. Applies remediation rules for common failures. Reports every action via Zulip DM and Gitea (SyslogSolution/health-logs). --- +--- # LiteLLM Operations — Health Check + Self-Heal @@ -65,7 +68,15 @@ Request → nginx:80 → LiteLLM:4000 → GPU(llama-server, parallel 2) ## LiteLLM Model Surface (ground truth — `/opt/inference-harness/litellm_config.yaml` on CT 116) -`model_name`s served: `qwen3.6-27B-code`, `gemma-4-12b`, `qwen3.6-35B-udq4`, `strix-moe`, `gpu-dense`, `gpu-light`, `syslog-auto`. +`model_name`s served: `qwen3.6-27B-code`, `gemma-4-12b`, `qwen3.6-35B-udq4`, `strix-moe`, `gpu-dense`, `gpu-light`, `syslog-auto`, `crew-auto` (new 2026-08-20). + +### Context Cap Split (2026-08-20) + +- **Abiba (firstmate)**: 128K uncapped — unlimited context for primary workloads +- **Hermes agents** (mumuni, tanko, koby, koonimo): 128K uncapped +- **Crewmates** (ops, tune, verify, auth-keys, build): 64K capped — alias `crew-auto` enforces 64K limit + +Preferred implementation: uncap shared pool, add capped alias for crew-only. - `syslog-auto` is a weighted router model: qwen3.6-27B-code (0.55, rpm 500) + qwen3.6-35B-udq4 (0.30, rpm 60) + gemma-4-12b (0.15, rpm 200). - `gpu-dense` / `gpu-light` are high-rpm aliases (rpm 500) onto qwen3.6-27B-code / gemma-4-12b respectively. @@ -120,6 +131,7 @@ Request → nginx:80 → LiteLLM:4000 → GPU(llama-server, parallel 2) - Also wakes on user request - On failure: re-check after 30s, escalate after 3 consecutive failures +--- --- ## Health Check @@ -162,6 +174,7 @@ Determine overall_status from individual check results: - "degraded" — 1-2 non-critical checks fail - "down" — critical checks fail +--- --- ## Remediation Rules @@ -205,6 +218,7 @@ Escalate → if SSH access unavailable, send Zulip DM Router no longer in path so Redis active counters are unused. Rule retained for reference but inactive. If Redis issues occur, check harness-redis container. +--- --- ## Reporting @@ -227,6 +241,7 @@ top actions, uptime. If a fix requires another agent (e.g., Authentik restart), relay sent to responsible agent with full context. +--- --- ## Execution diff --git a/memory-audit-maintenance.prose.md b/memory-audit-maintenance.prose.md index bbbf0a7..3cf1f1a 100644 --- a/memory-audit-maintenance.prose.md +++ b/memory-audit-maintenance.prose.md @@ -1,9 +1,12 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 name: memory-audit-maintenance kind: responsibility description: Shared memory audit and maintenance contract for Hermes agents (Mumuni, Koby, Koonimo). Tanko is no longer a Hermes agent (now on DSH/DeepSeek Harness since 2026-08-27) and uses DSH-native memory, so it is excluded from this Hermes roster. Each agent runs it against its own isolated memory files — no cross-agent access, no shared state. Detects staleness, enforces writer registry, and rotates canary tokens. id: 067NC4KG01RG50R40M30E20918 --- +--- ### Goal @@ -342,4 +345,3 @@ return { ### Per-Agent Notes -Each Hermes agent (Mumuni, Tdunna/Koby, Baggy/Koonimo) runs this contract against its own `~/.hermes/memories/` directory. The contract is identical across agents, but all data is fully isolated: separate ledgers, separate writer registries, separate canaries. If a new agent is added to the roster, it must be listed in `### Scope` above and given its own isolated memory directory. **Tanko is not covered by this contract — it runs on DSH (DeepSeek Harness) since 2026-08-27 and uses DSH-native memory.** \ No newline at end of file diff --git a/memory-fixer.prose.md b/memory-fixer.prose.md index 2f8f9a3..46da583 100644 --- a/memory-fixer.prose.md +++ b/memory-fixer.prose.md @@ -1,4 +1,6 @@ --- +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 kind: pattern name: memory-fixer description: > @@ -6,6 +8,7 @@ description: > Escalate anything that needs Kwame's input. Executes confirmed Kwame decisions to completion (state + updated_at). version: 2.0.0 --- +--- # Memory Fixer diff --git a/pm2-self-heal.prose.md b/pm2-self-heal.prose.md index c020116..6081167 100644 --- a/pm2-self-heal.prose.md +++ b/pm2-self-heal.prose.md @@ -2,17 +2,6 @@ kind: responsibility name: pm2-self-heal description: > - Monitors critical PM2 processes (abiba-zulip, abiba-telegram, gitea-runner, - spoton-service, zulip-watchdog) and auto-restarts any that are stopped or - errored. Logs every action to Gitea (SyslogSolution/health-logs — not - knowledge graph, hard rule) and alerts the owner via - Zulip DM on failures. - CRITICAL: Never restart abiba-zulip — it runs this contract. - AS-BUILT 2026-08-09 (captain ruling, ecosystem is authoritative): - gpu-monitor is systemd-managed (gpu-monitor.service) — NOT PM2; - gpu-watchdog decommissioned (function folded into gpu-monitor.service); - gitea-runner KEPT (online in PM2); abiba-zulip KEPT (online 4d+, the - 2026-07-04 'removed/decommissioned' note was stale and is removed). --- ## Maintains @@ -24,9 +13,6 @@ description: > - zulip-watchdog: { status: "online", uptime: string, restarts: number } - last_check: timestamp -> **Note (2026-07-04, SUPERSEDED 2026-08-09):** `abiba-zulip` remains ONLINE and -> is monitored — the decommission note was stale (process re-added; do not treat -> it as removed). ## Continuity @@ -63,9 +49,9 @@ description: > - If status is "online" → pass - If status is "stopped" or "errored" → apply Rule 1 - If restarts > 5 → alert owner -3. **Check abiba-zulip** (self-process, read-only): +3. **Check abiba-zulip** (live Zulip bridge, heartbeating): - If status is "online" → pass, log restarts count - - If status is "stopped" or "errored" → **DO NOT RESTART** — alert owner immediately + - If status is "stopped" or "errored" → restart (`pm2 restart abiba-zulip` — fully restored) - If restarts > 5 in last hour → alert owner with full diagnostics 4. **Log results** — Append to `SyslogSolution/health-logs/pm2/{timestamp}.md` in Gitea (not knowledge graph — hard rule) 5. **Alert** — Send Zulip DM to owner if escalation needed (do NOT run pm2 commands during alerting) diff --git a/scripts/agent-health-check.py b/scripts/agent-health-check.py index 895c28f..2c7ff87 100755 --- a/scripts/agent-health-check.py +++ b/scripts/agent-health-check.py @@ -39,10 +39,6 @@ PVE_NODES = { # Agent definitions: ct, host, user, pve_node, vault_key_name AGENTS = { - "tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"}, - "abiba": {"ct": 100, "host": "192.168.68.24", "user": "root", "pve": "minipve", "vault_key": None}, # Pi agent + Mumuni Zulip, no vault key - "koby": {"ct": 111, "host": "192.168.68.129", "user": "root", "pve": "amdpve", "vault_key": "KOBY_LITELLM_API_KEY"}, - "koonimo": {"ct": 113, "host": "192.168.68.114", "user": "root", "pve": "amdpve", "vault_key": "KOONIMO_LITELLM_API_KEY"}, } GPU_HOSTS = { @@ -238,6 +234,7 @@ def check_agents(): host = agent.get("host") user = agent.get("user") ct = agent["ct"] + report_only = agent.get("report_only", False) # Tanko runs on DSH (DeepSeek Harness) since 2026-08-27 — it no longer runs a # Hermes gateway, so skip the Hermes gateway/state/streaming/journal checks. @@ -253,15 +250,20 @@ def check_agents(): print(f" ⬜ {name} (CT {ct}): cannot SSH — skip liveness check") continue - # Gateway process - pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) - if not pid: - # Try alternate binary name - pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) - if not pid: - print(f" ❌ {name}: GATEWAY NOT RUNNING") - FAIL.append(f"gateway-down:{name}") - continue + # ⛔ KOBY IS NEVER REPAIRED — diagnostic only + if report_only: + print(f" 🔍 {name}: REPORT-ONLY mode (diagnostic only, no repairs on .129)") + # Still check gateway status for reporting purposes + pid = ssh(host, "pgrep -f 'hermes_cli.main gateway run' | grep -v infisical | head -1", user=user) + if not pid: + pid = ssh(host, "pgrep -f 'hermes.*gateway' | grep -v infisical | grep -v bash | head -1", user=user) + if not pid: + print(f" ⚠️ {name}: GATEWAY NOT RUNNING (reported only)") + FAIL.append(f"gateway-down:{name}") + continue + else: + print(f" ✅ {name}: gateway running (pid={pid}, report-only mode)") + continue # Skip the rest of the check for Koby # Gateway state file state = ssh(host, "cat ~/.hermes/gateway_state.json 2>/dev/null", user=user) diff --git a/zulip-health.prose.md b/zulip-health.prose.md index f23b33a..64adba8 100644 --- a/zulip-health.prose.md +++ b/zulip-health.prose.md @@ -6,6 +6,8 @@ title: Zulip Mesh Health Monitor — Multi-Platform version: 3.0.0 runtime_contract: 2 agent: abiba +report_only_agents: + - koby # ⛔ KOBY IS NEVER REPAIRED (Rule 17, 2026-08-17) — detect + report, never fix on .129 --- # Zulip Mesh Health Monitor