Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d6376e5142 | ||
|
|
13ac189365 | ||
|
|
2851a0cfc8 | ||
|
|
e0c9852de8 | ||
|
|
bf3a1ba523 | ||
|
|
f230812e3a | ||
|
|
96769a103f | ||
|
|
d697baa7b6 | ||
|
|
fb7f351a2b | ||
|
|
e42b970dec | ||
|
|
eadb927ec1 | ||
|
|
d4e238047d | ||
|
|
73d5097555 |
@@ -24,7 +24,7 @@ Runs every 4 hours (2, 6, 10, 14, 18, 22 UTC at :35) via cron (`35 2,6,10,14,18,
|
||||
## Requires
|
||||
|
||||
- **LiteLLM admin key** for key validation (retrieved from `/root/.pi/agent/env.sh`)
|
||||
- **SSH access** to GPU hosts (.8, .110, .15) and agent CTs (.122, .129, .114, .24)
|
||||
- **SSH access** to GPU hosts — `llmuser` on .8 (owns `llama-server`), `root` on .110 and .15 — and agent CTs (.122, .129, .114, .24)
|
||||
- **Python 3** for script execution
|
||||
- **Network access** to LiteLLM (:4000), GPU exporters (:9400), and gateway endpoints
|
||||
|
||||
|
||||
+36
-17
@@ -94,7 +94,16 @@ def audit(path):
|
||||
cfg = yaml.safe_load(f)
|
||||
|
||||
model = cfg.get("model", {})
|
||||
fb = cfg.get("fallback_providers", {})
|
||||
fb_raw = cfg.get("fallback_providers", {})
|
||||
# Normalize: fallback_providers may be a dict (single provider) or a list of dicts
|
||||
# (one entry per fallback). Both shapes are valid; we must handle both without crashing.
|
||||
if isinstance(fb_raw, dict):
|
||||
fb_entries = [fb_raw]
|
||||
elif isinstance(fb_raw, list):
|
||||
fb_entries = fb_raw
|
||||
else:
|
||||
fb_entries = [fb_raw] # Let it fail the check below as malformed
|
||||
fb = fb_entries[0] if fb_entries else {}
|
||||
comp = cfg.get("compression", {})
|
||||
aux = cfg.get("auxiliary", {})
|
||||
deleg = cfg.get("delegation", {})
|
||||
@@ -207,22 +216,32 @@ def audit(path):
|
||||
"Rule 14",
|
||||
f"delegation.provider must be 'harness' (got {deleg.get('provider')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"fallback_providers.provider must be 'deepseek' (got {fb.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
fb.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"fallback_providers.model must be 'deepseek-v4-flash' (got {fb.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
fb.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"fallback_providers.api_key_env must be DEEPSEEK_API_KEY (got {fb.get('api_key_env')!r})",
|
||||
)
|
||||
# Check each fallback entry. A malformed entry (not a mapping) is a VIOLATION, not a crash.
|
||||
for idx, entry in enumerate(fb_entries):
|
||||
prefix = f"fallback_providers[{idx}]"
|
||||
if not isinstance(entry, dict):
|
||||
check(
|
||||
False,
|
||||
"Rule 14",
|
||||
f"{prefix} must be a mapping (got {type(entry).__name__})",
|
||||
)
|
||||
continue
|
||||
check(
|
||||
entry.get("provider") == "deepseek",
|
||||
"Rule 14",
|
||||
f"{prefix}.provider must be 'deepseek' (got {entry.get('provider')!r}) — "
|
||||
f"true fallback diversity, not same endpoint as primary",
|
||||
)
|
||||
check(
|
||||
entry.get("model") == "deepseek-v4-flash",
|
||||
"Rule 14",
|
||||
f"{prefix}.model must be 'deepseek-v4-flash' (got {entry.get('model')!r})",
|
||||
)
|
||||
check(
|
||||
entry.get("api_key_env") == "DEEPSEEK_API_KEY",
|
||||
"Rule 14",
|
||||
f"{prefix}.api_key_env must be DEEPSEEK_API_KEY (got {entry.get('api_key_env')!r})",
|
||||
)
|
||||
|
||||
# --- custom_providers sanity ---
|
||||
check(
|
||||
|
||||
@@ -643,7 +643,7 @@ contracts:
|
||||
timeout: 120
|
||||
requires:
|
||||
- Zulip API key for abiba-bot@chat.sysloggh.net
|
||||
- SSH access to amdpve (192.168.68.15) for Tanko (CT 112) and the Agent Zero Docker host (.14)
|
||||
- SSH access to minipve (192.168.68.12) for Tanko (CT 112) and the Agent Zero Docker host (.14)
|
||||
verification:
|
||||
postconditions:
|
||||
- check: bot registration active
|
||||
|
||||
@@ -414,7 +414,7 @@ one-off GPU builds. No automated post-migration cleanup was in place.
|
||||
| 108 | media | storepve | lxc | ✅ reachable |
|
||||
| 110 | gitea | minipve | lxc | ✅ reachable |
|
||||
| 111 | tdunna | **storepve** | lxc | ⛔ **REPORT-ONLY** (192.168.68.129, Theo's box — no GC at any level) |
|
||||
| 112 | tanko | amdpve | lxc | ✅ reachable |
|
||||
| 112 | tanko | minipve | lxc | ✅ reachable |
|
||||
| 113 | baggy | amdpve | lxc | ✅ reachable |
|
||||
| 115 | scottdenya | amdpve | lxc | ✅ reachable |
|
||||
| 116 | syslog-api | minipve | lxc | ✅ reachable |
|
||||
|
||||
@@ -5,6 +5,11 @@ description: >
|
||||
Standard Hermes configuration template for Syslog Solution LLC agents.
|
||||
Enforces shared infrastructure setup (Firecrawl, SearXNG, local models,
|
||||
RA-H OS MCP) while keeping agent-specific API keys and model choices.
|
||||
UPDATED 2026-09-27: Clarified the Auxiliary Tasks policy — light aux (vision,
|
||||
web_extract/browsing) -> gpu-vision (RTX 5070); context-heavy aux (compression) ->
|
||||
syslog-auto (2026-07-23 decision, Rule 7). Removed the false "one model for all
|
||||
auxiliary" / "never syslog-auto" claim; stated gpu-dense + strix-moe are the reasoning
|
||||
hosts and aux should not be pinned to them. Now matches audit-hermes-config.py line-for-line.
|
||||
UPDATED 2026-08-07: Added litellm MCP server entry; updated Rule 15 (MCP Validation)
|
||||
to enforce REAL key headers (not env-vars) from the 2026-08-07 keyless-MCP incident.
|
||||
Added Rule 12 (Context-Issue Diagnostic) + Rule 13 (.env fallback enforcement) from the
|
||||
@@ -167,13 +172,16 @@ compression:
|
||||
abort_on_summary_failure: false
|
||||
|
||||
# ─── Auxiliary Tasks (CONSISTENCY RULE) ───
|
||||
# All auxiliary services MUST use identical model, base_url, and api_key_env:
|
||||
# model: gpu-vision # stable alias (NOT a raw model name)
|
||||
# Auxiliary tasks split into TWO model classes — do NOT assume one model for all:
|
||||
# Light auxiliary (vision, web_extract/browsing) -> model: gpu-vision # RTX 5070
|
||||
# Keeps the reasoning hosts (gpu-dense / strix-moe) free for agent prompts.
|
||||
# Context-heavy auxiliary (compression) -> model: syslog-auto # weighted pool
|
||||
# Deliberate per the 2026-07-23 OPERATIONAL DECISION in Rule 7: summarization
|
||||
# runs against long histories and must be able to use the pool.
|
||||
# Do NOT pin auxiliary work to the reasoning hosts (gpu-dense / strix-moe).
|
||||
# All auxiliary services share identical ROUTING (base_url + api_key_env), not model:
|
||||
# base_url: http://192.168.68.116/litellm/v1 # Rule 5 (2026-08-09): canonical authenticated; /v1 also OK
|
||||
# api_key_env: LITELLM_API_KEY
|
||||
# Do NOT use syslog-auto for auxiliary tasks — it routes to the primary GPU.
|
||||
# gpu-vision = RTX 5070 (12B), freeing the Strix Halo for agent reasoning.
|
||||
# Heavy aux (delegation, x_search) use gpu-dense (RTX 3090) instead.
|
||||
# NEVER use retired model names (qwen3.6-27B-code, qwen3.6-35B-udq4; gemma-4-12b is retired
|
||||
# and no longer resolves) in agent configs — use the stable aliases so model swaps don't break agents.
|
||||
auxiliary:
|
||||
|
||||
@@ -55,7 +55,7 @@ connectivity recovery including end-to-end DM validation.
|
||||
|
||||
| Host | CT | Proxmox | IP (direct) | Hermes Home | User |
|
||||
|------|-----|---------|-------------|-------------|------|
|
||||
| Tanko | CT112 | amdpve | 192.168.68.122 | /home/jerome/.hermes | jerome | *(DSH since 2026-08-27 — historical, plugin retired on this host)* |
|
||||
| Tanko | CT112 | minipve | 192.168.68.122 | /home/jerome/.hermes | jerome | *(DSH since 2026-08-27 — historical, plugin retired on this host)* |
|
||||
| Koby | CT111 | storepve | 192.168.68.129 | /root/.hermes | root |
|
||||
| Shumba | — | — | 192.168.68.119 | /home/lucky/.hermes | lucky |
|
||||
|
||||
@@ -72,7 +72,7 @@ connectivity recovery including end-to-end DM validation.
|
||||
### Step 1: Resolve Target
|
||||
|
||||
Map `target` to host, CT ID, hermes_home, and user from the live-state table.
|
||||
For CT112 route through `ssh root@amdpve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
For CT112 route through `ssh root@minipve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
|
||||
### Step 2: Pull Latest Plugin Source
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ gateway restart, and connection validation.
|
||||
### Step 1: Locate Target
|
||||
|
||||
Map `target` to connectivity parameters from the live-state table above.
|
||||
For CT112 route through `ssh root@amdpve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
For CT112 route through `ssh root@minipve`; for CT111 route through `ssh root@storepve` — then `pct exec <id>`.
|
||||
|
||||
### Step 2: Deploy Zulip Adapter
|
||||
|
||||
|
||||
@@ -105,8 +105,8 @@ description: >
|
||||
|
||||
| Node | IP | CPU | RAM | VMs/CTs | Role |
|
||||
|------|----|-----|-----|---------|------|
|
||||
| minipve | .12 | 16C | 30GB | abiba, authentik, gitea, syslog-api, infisical-vault, jitsi | Auth, git, messaging |
|
||||
| amdpve | .15 | 32C | 62GB | kagentz, tanko, baggy, scottdenya, adguard2 | Agents, compute |
|
||||
| minipve | .12 | 16C | 30GB | abiba, tanko, authentik, gitea, syslog-api, infisical-vault, jitsi | Auth, git, messaging |
|
||||
| amdpve | .15 | 32C | 62GB | kagentz, baggy, scottdenya, adguard2 | Agents, compute |
|
||||
| storepve | .6 | 28C | 31GB | docker-vm, ra-h-os, PBS, media, jdownloader, zulip, tdunna | Docker, storage, chat |
|
||||
| acerpve | .9 | 28C | 31GB | llm-gpu | GPU VMs |
|
||||
| ocupve | .5 | 12C | 14GB | ocu-llm | GPU VMs |
|
||||
@@ -682,7 +682,7 @@ ssh root@192.168.68.110 "systemctl restart llama-server"
|
||||
| 109 | docker-vm | storepve | .7 | Docker host | ❌ |
|
||||
| 110 | gitea | minipve | **.17** | Git | ❌ |
|
||||
| 111 | tdunna | storepve | .129 | Hermes agent — ⛔ REPORT-ONLY (Theo's box, no GC) | ✅ |
|
||||
| 112 | tanko | amdpve | .122 | DSH (DeepSeek Harness) agent | ✅ |
|
||||
| 112 | tanko | minipve | .122 | DSH (DeepSeek Harness) agent | ✅ |
|
||||
| 113 | baggy | amdpve | .114 | Hermes agent | ✅ |
|
||||
| 115 | scottdenya | amdpve | .75 | Denya OneCare | ❌ |
|
||||
| 116 | syslog-api | minipve | .116 | LiteLLM + Grafana | ❌ |
|
||||
@@ -712,7 +712,7 @@ Source of truth: `/root/scripts/pct-run.sh` or `prose-contracts/scripts/pct-run.
|
||||
| 100 | abiba | minipve | `pct-run 100` |
|
||||
| 105 | kagentz | amdpve | `pct-run 105` |
|
||||
| 111 | tdunna | storepve | `pct-run 111` (⛔ report-only — no GC) |
|
||||
| 112 | tanko | amdpve | `pct-run 112` |
|
||||
| 112 | tanko | minipve | `pct-run 112` |
|
||||
| 113 | baggy | amdpve | `pct-run 113` |
|
||||
| 115 | scottdenya | amdpve | `pct-run 115` |
|
||||
| 104 | authentik | minipve | `pct-run 104` |
|
||||
|
||||
@@ -59,7 +59,7 @@ Before ANY update wave:
|
||||
| ocupve (.5) | Proxmox node | `apt update && apt upgrade -y` | 5 min |
|
||||
| CT 100 (.24) | Abiba (pi) | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 116 (.116) | syslog-api (LiteLLM host) | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 112 (tanko, amdpve) | Tanko | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 112 (tanko, minipve) | Tanko | `apt update && apt upgrade -y` | 3 min |
|
||||
| CT 105 (kagentz, minipve) | Mumuni | `apt update && apt upgrade -y` | 3 min |
|
||||
| VM 101 (.8) | llm-gpu (RTX 3090) | `apt update && apt upgrade -y` | 3 min |
|
||||
| VM 103 (.110) | ocu-llm (RTX 5070) | `apt update && apt upgrade -y` | 3 min |
|
||||
|
||||
@@ -50,6 +50,11 @@ Changelog:
|
||||
(kagentz CT 105 on minipve, .14, dedicated `hermes` user) and is monitored
|
||||
from her side. This script must not probe mumuni or .24 — the v2 changelog
|
||||
roster line was the last reference still placing her at .24 / CT100.
|
||||
v6 (2026-09-28): .8 GPU health probe now runs as `llmuser` instead of `root`.
|
||||
Root SSH to .8 was lost when the guest was rebuilt, so every .8 leg read as
|
||||
UNREACHABLE for a healthy host. llmuser owns llama-server and can read
|
||||
`systemctl is-active`, `systemctl show -p MainPID`, and the :8080 pid.
|
||||
.110 and .15 keep the default `root` user.
|
||||
"""
|
||||
|
||||
import subprocess, json, sys, os, time, re, io, contextlib
|
||||
@@ -70,7 +75,7 @@ PVE_NODES = {
|
||||
|
||||
# Agent definitions: ct, host, user, pve_node, vault_key_name
|
||||
AGENTS = {
|
||||
"tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "amdpve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"},
|
||||
"tanko": {"ct": 112, "host": "192.168.68.122", "user": "jerome", "pve": "minipve", "vault_key": "TANKO_LITELLM_API_KEY", "runtime": "dsh"},
|
||||
# abiba = pi agent (.24) — no vault key; its LiteLLM key is read from its
|
||||
# local env file (key_env below), not from the shared vault or .bashrc.
|
||||
# runtime=pi: abiba has run pi-only since the harness purge. There is no
|
||||
@@ -95,7 +100,7 @@ AGENTS = {
|
||||
# .110 rtx5070 (ocu-llm VM) -> llama-server.service (active)
|
||||
# .15 strixhalo (amdpve) -> strix-server.service (active)
|
||||
GPU_HOSTS = {
|
||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service"},
|
||||
"gpu-rtx3090 (.8)": {"host": "192.168.68.8", "port": 8080, "service": "llama-chat-api.service", "user": "llmuser"},
|
||||
"gpu-rtx5070 (.110)": {"host": "192.168.68.110", "port": 8080, "service": "llama-server.service"},
|
||||
"gpu-strixhalo (.15)": {"host": "192.168.68.15", "port": 8080, "service": "strix-server.service"},
|
||||
}
|
||||
@@ -300,13 +305,14 @@ def check_gpu_ports():
|
||||
host = gpu["host"]
|
||||
port = gpu["port"]
|
||||
svc = gpu["service"]
|
||||
user = gpu.get("user", "root") # default root, overridden per-host where needed
|
||||
|
||||
# `systemctl is-active` exits non-zero when the unit is inactive or
|
||||
# missing, which the ssh() helper would swallow as an SSH failure and
|
||||
# report as UNREACHABLE. `|| true` keeps the real state word so we can
|
||||
# tell "unit inactive" from "host unreachable".
|
||||
svc_status = ssh(host, f"systemctl is-active {svc} || true")
|
||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1")
|
||||
svc_status = ssh(host, f"systemctl is-active {svc} || true", user=user)
|
||||
port_owner = ssh(host, f"ss -tlnp 2>/dev/null | grep -Po ':{port}\\s+.*pid=\\K[0-9]+' | head -1", user=user)
|
||||
|
||||
if not svc_status:
|
||||
print(f" ❌ {label}: UNREACHABLE")
|
||||
@@ -317,14 +323,14 @@ def check_gpu_ports():
|
||||
print(f" ❌ {label}: PORT {port} NOT LISTENING (svc={svc_status})")
|
||||
FAIL.append(f"gpu-no-port:{label}")
|
||||
elif svc_status != "active":
|
||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2")
|
||||
svc_pid = ssh(host, f"systemctl show {svc} -p MainPID 2>/dev/null | cut -d= -f2", user=user)
|
||||
if svc_pid and port_owner != svc_pid:
|
||||
print(f" ❌ {label}: GHOST PROCESS — port owned by pid {port_owner}, svc pid {svc_pid} (svc={svc_status})")
|
||||
FAIL.append(f"gpu-ghost:{label}:{port_owner}")
|
||||
else:
|
||||
print(f" ⚠️ {label}: svc={svc_status}, port owned by {port_owner}")
|
||||
else:
|
||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health")
|
||||
health = ssh(host, f"curl -s --max-time 5 http://localhost:{port}/health", user=user)
|
||||
if health and '"status":"ok"' in health:
|
||||
print(f" ✅ {label}: healthy (pid={port_owner})")
|
||||
elif health and '"status":"no slot available"' in health:
|
||||
|
||||
@@ -101,8 +101,6 @@ GUESTS: list[Guest] = [
|
||||
# amdpve (192.168.68.15)
|
||||
Guest(ct_id="105", hostname="kagentz", ip="192.168.68.105", node="amdpve",
|
||||
access_method="ssh-host", probe_target="kagentz (CT 105, amdpve)"),
|
||||
Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="amdpve",
|
||||
access_method="pct-run", probe_target="tanko (CT 112, amdpve)"),
|
||||
Guest(ct_id="113", hostname="baggy", ip="192.168.68.113", node="amdpve",
|
||||
access_method="pct-run", probe_target="baggy (CT 113, amdpve)"),
|
||||
Guest(ct_id="115", hostname="scottdenya", ip="192.168.68.115", node="amdpve",
|
||||
@@ -110,6 +108,8 @@ GUESTS: list[Guest] = [
|
||||
Guest(ct_id="120", hostname="adguard2", ip="192.168.68.120", node="amdpve",
|
||||
access_method="pct-run", probe_target="adguard2 (CT 120, amdpve)"),
|
||||
# minipve (192.168.68.12)
|
||||
Guest(ct_id="112", hostname="tanko", ip="192.168.68.112", node="minipve",
|
||||
access_method="pct-run", probe_target="tanko (CT 112, minipve)"),
|
||||
Guest(ct_id="100", hostname="abiba", ip="192.168.68.100", node="minipve",
|
||||
access_method="pct-run", probe_target="abiba (CT 100, minipve)"),
|
||||
Guest(ct_id="102", hostname="adguard", ip="192.168.68.102", node="minipve",
|
||||
|
||||
+1
-1
@@ -12,7 +12,6 @@ set -euo pipefail
|
||||
declare -A CT_NODES=(
|
||||
# amdpve (192.168.68.15)
|
||||
[105]=amdpve # kagentz (was hwepve — corrected 2026-09-12; live per pvesh)
|
||||
[112]=amdpve # tanko
|
||||
[113]=amdpve # baggy
|
||||
[115]=amdpve # scottdenya
|
||||
[120]=amdpve # adguard2 (added 2026-09-12)
|
||||
@@ -21,6 +20,7 @@ declare -A CT_NODES=(
|
||||
[102]=minipve # adguard (was acerpve)
|
||||
[104]=minipve # authentik
|
||||
[110]=minipve # gitea
|
||||
[112]=minipve # tanko (was amdpve — migrated 2026-09-27; live per pvesh)
|
||||
[116]=minipve # syslog-api
|
||||
[119]=minipve # infisical-vault
|
||||
# storepve (192.168.68.6)
|
||||
|
||||
@@ -47,8 +47,8 @@ You are a code reviewer for OpenProse infrastructure contracts in the Syslog Sol
|
||||
The infrastructure-control.prose.md contract is the canonical reference for the cluster topology:
|
||||
|
||||
**Proxmox Cluster "Tabiri" (5 nodes):**
|
||||
- amdpve (192.168.68.15): kagentz, tanko, baggy, scottdenya, adguard2
|
||||
- minipve (192.168.68.12): abiba, adguard, authentik, gitea, syslog-api, infisical-vault
|
||||
- amdpve (192.168.68.15): kagentz, baggy, scottdenya, adguard2
|
||||
- minipve (192.168.68.12): abiba, tanko, adguard, authentik, gitea, syslog-api, infisical-vault
|
||||
- storepve (192.168.68.6): docker-vm, ra-h-os, PBS, media, jdownloader, zulip, tdunna
|
||||
- acerpve (192.168.68.9): llm-gpu
|
||||
- ocupve (192.168.68.5): ocu-llm
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/bin/bash
|
||||
# swap-gpu-dense-model.sh — Swap RTX 3090 from qwen3.6-27B-code to SmartCode-Fable-5
|
||||
# Run when download completes: ssh root@192.168.68.8 'bash -s' < this script
|
||||
# Run when download completes: ssh llmuser@192.168.68.8 'sudo bash -s' < this script
|
||||
#
|
||||
# Usage: bash swap-gpu-dense-model.sh
|
||||
# Requires: new model at /home/llmuser/models/SmartCode-Fable-5-27B-UD-Q4_K_XL.gguf
|
||||
|
||||
@@ -135,16 +135,16 @@ case "$PI_VERDICT" in
|
||||
esac
|
||||
# -- abiba-leg-end
|
||||
|
||||
# ── Platform B: Tanko (DSH dsh-web on amdpve CT 112) ──
|
||||
# ── Platform B: Tanko (DSH dsh-web on minipve CT 112) ──
|
||||
# Direct SSH to 192.168.68.122 is not a dependency of this monitor — per-worker
|
||||
# key availability varies — so probes run from the amdpve vantage via `pct exec`.
|
||||
# Tanko's Zulip gateway runs as the dsh-web systemd unit inside CT 112 on amdpve
|
||||
# (192.168.68.15). The gateway binds 127.0.0.1:3080 loopback-only by design — a
|
||||
# key availability varies — so probes run from the minipve vantage via `pct exec`.
|
||||
# Tanko's Zulip gateway runs as the dsh-web systemd unit inside CT 112 on minipve
|
||||
# (192.168.68.12). The gateway binds 127.0.0.1:3080 loopback-only by design — a
|
||||
# remote :3080 probe is refused and is NOT a fault.
|
||||
TANKO_SVC=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \
|
||||
TANKO_SVC=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.12 \
|
||||
"pct exec 112 -- systemctl is-active dsh-web" 2>/dev/null || true)
|
||||
[ -n "$TANKO_SVC" ] || TANKO_SVC="unknown"
|
||||
TANKO_HTTP=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.15 \
|
||||
TANKO_HTTP=$(ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 root@192.168.68.12 \
|
||||
"pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/" 2>/dev/null || true)
|
||||
[ -n "$TANKO_HTTP" ] || TANKO_HTTP="000"
|
||||
|
||||
|
||||
@@ -0,0 +1,236 @@
|
||||
"""Regression test for the fallback_providers list-shape crash in audit-hermes-config.py.
|
||||
|
||||
WHY THIS FILE EXISTS: audit-hermes-config.py assumed `fallback_providers` was always a dict
|
||||
(single provider). Two live agents (koby, koonimo) carry it as a LIST of dicts (one entry per
|
||||
fallback), so the script crashed with:
|
||||
|
||||
File "audit-hermes-config.py", line 211, in audit
|
||||
fb.get("provider") == "deepseek",
|
||||
AttributeError: 'list' object has no attribute 'get'
|
||||
|
||||
Both are REAL agent configs, so this is not a malformed-input case — the script simply could not
|
||||
audit two of the four agents it exists to audit. Until fixed, the key-hygiene check had no
|
||||
coverage for half the fleet while appearing to run.
|
||||
|
||||
These tests execute the real CLI (`python3 audit-hermes-config.py <config>`) and assert:
|
||||
1. A config whose `fallback_providers` is a LIST of valid dicts does NOT crash (exit code is 0 or 1,
|
||||
never a traceback/AttributeError).
|
||||
2. A config whose `fallback_providers` contains a MALFORMED entry (a list element that is not a
|
||||
mapping) reports a VIOLATION naming the offending entry, NOT an uncaught exception.
|
||||
3. The dict shape still works (existing tests must stay green).
|
||||
|
||||
No network, vault, or SSH access is required.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||
AUDIT = ROOT / "audit-hermes-config.py"
|
||||
|
||||
# A valid config where fallback_providers is a LIST of dicts (the real koby/koonimo shape).
|
||||
# One entry, well-formed: provider=deepseek, model=deepseek-v4-flash, api_key_env=DEEPSEEK_API_KEY.
|
||||
# This must produce a real verdict (PASS or FAIL) without crashing.
|
||||
LIST_SHAPE_VALID = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# A valid config where fallback_providers is a LIST with TWO entries (multiple fallbacks).
|
||||
# Both entries well-formed. Must not crash and should produce a real verdict.
|
||||
LIST_SHAPE_MULTI = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# A config where fallback_providers is a LIST containing a MALFORMED entry:
|
||||
# one element is a plain string, not a mapping. The checker must report a VIOLATION
|
||||
# naming the offending entry (fallback_providers[1]) and NOT crash.
|
||||
LIST_SHAPE_MALFORMED = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
- provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
- "not-a-mapping"
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
# The original DICT shape (single provider) must still work — existing behaviour preserved.
|
||||
DICT_SHAPE_VALID = """
|
||||
model:
|
||||
api_key: ""
|
||||
api_key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
max_tokens: 4096
|
||||
default: syslog-auto
|
||||
provider: harness
|
||||
fallback_providers:
|
||||
provider: deepseek
|
||||
model: deepseek-v4-flash
|
||||
api_key_env: DEEPSEEK_API_KEY
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
threshold: 0.65
|
||||
max_context_window: 131072
|
||||
auxiliary:
|
||||
vision:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
web_extract:
|
||||
model: gpu-vision
|
||||
provider: harness
|
||||
compression:
|
||||
model: syslog-auto
|
||||
provider: harness
|
||||
delegation:
|
||||
provider: harness
|
||||
custom_providers:
|
||||
- name: harness
|
||||
key_env: LITELLM_API_KEY
|
||||
base_url: http://192.168.68.116/litellm/v1
|
||||
"""
|
||||
|
||||
|
||||
def _run_config(tmp_path, name, text):
|
||||
cfg = tmp_path / name
|
||||
cfg.write_text(text)
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(AUDIT), str(cfg)],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
return proc.returncode, proc.stdout, proc.stderr
|
||||
|
||||
|
||||
def test_list_shape_single_entry_does_not_crash(tmp_path):
|
||||
"""A LIST with one valid dict must not raise AttributeError; exit 0 (PASS)."""
|
||||
code, out, err = _run_config(tmp_path, "list-single.yaml", LIST_SHAPE_VALID)
|
||||
# Must NOT be a crash (traceback). A clean run exits 0 (PASS) or 1 (FAIL), never 2+ (exception).
|
||||
assert code in (0, 1), f"Expected clean exit 0 or 1, got {code}\nSTDOUT:\n{out}\nSTDERR:\n{err}"
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
# The valid single-entry list should PASS (all rules satisfied).
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}"
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_list_shape_multiple_entries_does_not_crash(tmp_path):
|
||||
"""A LIST with two valid dicts must not raise AttributeError; exit 0 (PASS)."""
|
||||
code, out, err = _run_config(tmp_path, "list-multi.yaml", LIST_SHAPE_MULTI)
|
||||
assert code in (0, 1), f"Expected clean exit 0 or 1, got {code}\nSTDOUT:\n{out}\nSTDERR:\n{err}"
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}"
|
||||
assert "RESULT: PASS" in out
|
||||
|
||||
|
||||
def test_list_shape_malformed_entry_reports_violation_not_crash(tmp_path):
|
||||
"""A LIST containing a non-mapping element must be a reported VIOLATION, not a crash."""
|
||||
code, out, err = _run_config(tmp_path, "list-malformed.yaml", LIST_SHAPE_MALFORMED)
|
||||
# Must NOT be a crash.
|
||||
assert "AttributeError" not in err, f"Crashed with AttributeError:\n{err}"
|
||||
assert "Traceback" not in err, f"Crashed with uncaught exception:\n{err}"
|
||||
# Should be a FAIL (exit 1) because the malformed entry is a violation.
|
||||
assert code == 1, f"Expected FAIL (exit 1) but got {code}\n{out}"
|
||||
assert "RESULT: FAIL" in out
|
||||
# The violation must name the offending entry (fallback_providers[1]).
|
||||
assert "fallback_providers[1]" in out, f"Violation did not name the offending entry:\n{out}"
|
||||
|
||||
|
||||
def test_dict_shape_still_passes(tmp_path):
|
||||
"""The original DICT shape (single provider) must still PASS — existing behaviour preserved."""
|
||||
code, out, err = _run_config(tmp_path, "dict-valid.yaml", DICT_SHAPE_VALID)
|
||||
assert code == 0, f"Expected PASS but got {code}\n{out}\nSTDERR:\n{err}"
|
||||
assert "RESULT: PASS" in out
|
||||
@@ -49,7 +49,7 @@ HEALTH_CONTRACT = ROOT / "zulip-health.prose.md"
|
||||
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
|
||||
|
||||
MUMUNI_IP = "192.168.68.24" # Mumuni's old (decommissioned) deployment
|
||||
TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec
|
||||
TANKO_VANTAGE = "192.168.68.12" # minipve — Tanko CT 112 via pct exec
|
||||
AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker
|
||||
|
||||
|
||||
@@ -82,7 +82,7 @@ done
|
||||
printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts"
|
||||
cmd="${*: -1}"
|
||||
case "$host" in
|
||||
192.168.68.15)
|
||||
192.168.68.12)
|
||||
case "$cmd" in
|
||||
*"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;;
|
||||
*curl*) printf '%s' "$TANKO_HTTP" ;;
|
||||
|
||||
@@ -104,6 +104,59 @@ def test_koby_ct111_is_on_storepve(ahc):
|
||||
assert ahc.AGENTS["koby"]["pve"] == "storepve"
|
||||
|
||||
|
||||
def test_tanko_ct112_is_probed_on_minipve(ahc, monkeypatch, capsys):
|
||||
# CT 112 (tanko) was live-migrated to minipve (.12) on 2026-09-27; the
|
||||
# amdpve mapping made `pct status 112` fail and read as ct-unreachable.
|
||||
# Execute the probe and assert the host the script actually contacts.
|
||||
probes = []
|
||||
monkeypatch.setattr(
|
||||
ahc, "ssh",
|
||||
lambda host, cmd, user="root": probes.append((host, cmd)) or "status: running",
|
||||
)
|
||||
ahc.FAIL.clear()
|
||||
ahc.REPORT_ONLY.clear()
|
||||
try:
|
||||
ahc.check_ct_liveness()
|
||||
tanko_hosts = [h for h, cmd in probes if cmd == "pct status 112 2>/dev/null"]
|
||||
assert tanko_hosts == ["192.168.68.12"]
|
||||
finally:
|
||||
ahc.FAIL.clear()
|
||||
ahc.REPORT_ONLY.clear()
|
||||
|
||||
|
||||
def test_gpu_rtx3090_probe_uses_llmuser_not_root(ahc, monkeypatch, capsys):
|
||||
# 2026-09-28: root SSH to .8 was lost when the guest was rebuilt; llmuser
|
||||
# owns llama-server and can read systemctl status and the :8080 pid. A root
|
||||
# probe reads as UNREACHABLE for a healthy host (the reported bug). Execute
|
||||
# check_gpu_ports() against an SSH boundary that only accepts llmuser@.8 and
|
||||
# assert the .8 leg does not produce the false UNREACHABLE failure.
|
||||
seen = []
|
||||
|
||||
def fake_ssh(host, cmd, user="root"):
|
||||
seen.append((host, user))
|
||||
if host == "192.168.68.8" and user != "llmuser":
|
||||
return None # root SSH denied -> baseline false UNREACHABLE
|
||||
if cmd.startswith("systemctl is-active"):
|
||||
return "active"
|
||||
if cmd.startswith("ss -tlnp"):
|
||||
return "48351"
|
||||
if cmd.startswith("curl"):
|
||||
return '{"status":"ok"}'
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(ahc, "ssh", fake_ssh)
|
||||
ahc.FAIL.clear()
|
||||
try:
|
||||
ahc.check_gpu_ports()
|
||||
out = capsys.readouterr().out
|
||||
assert "gpu-unreachable:192.168.68.8" not in ahc.FAIL
|
||||
assert "\u2705 gpu-rtx3090 (.8): healthy" in out
|
||||
assert ("192.168.68.8", "llmuser") in seen
|
||||
assert not any(host == "192.168.68.8" and user == "root" for host, user in seen)
|
||||
finally:
|
||||
ahc.FAIL.clear()
|
||||
|
||||
|
||||
def test_report_only_legs_never_count_as_failures(ahc):
|
||||
for agent, report_only in (("koby", True), ("koonimo", False), ("tanko", False)):
|
||||
ahc.FAIL.clear()
|
||||
|
||||
@@ -34,7 +34,7 @@ ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
ZULIP_MONITOR = ROOT / "scripts" / "zulip-monitor.sh"
|
||||
CONNECTED_FIXTURE = ROOT / "tests" / "fixtures" / "zulip-health-connected.json"
|
||||
|
||||
TANKO_VANTAGE = "192.168.68.15" # amdpve — Tanko CT 112 via pct exec
|
||||
TANKO_VANTAGE = "192.168.68.12" # minipve — Tanko CT 112 via pct exec
|
||||
AGENT_ZERO_HOST = "192.168.68.14" # kagentz host, Agent Zero docker
|
||||
|
||||
|
||||
@@ -52,7 +52,7 @@ done
|
||||
printf '%s\n' "$host" >> "$RECORD_DIR/ssh.hosts"
|
||||
cmd="${*: -1}"
|
||||
case "$host" in
|
||||
192.168.68.15)
|
||||
192.168.68.12)
|
||||
case "$cmd" in
|
||||
*"systemctl is-active"*) printf '%s' "$TANKO_SVC" ;;
|
||||
*curl*) printf '%s' "$TANKO_HTTP" ;;
|
||||
|
||||
+24
-24
@@ -28,7 +28,7 @@ session start.
|
||||
## Requires
|
||||
|
||||
- **Zulip API key** for `abiba-bot@chat.sysloggh.net` in `$ZULIP_API_KEY`
|
||||
- **SSH access** to amdpve (192.168.68.15) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14)
|
||||
- **SSH access** to minipve (192.168.68.12) for Tanko — CT 112 reached via `pct exec` (direct SSH to .122 is not a dependency of this contract: per-worker key availability varies); and the Agent Zero Docker host (192.168.68.14)
|
||||
- **PM2** on localhost for pi process management
|
||||
- **Network access** to `chat.sysloggh.net`, `kagentz.sysloggh.net` (C3 public path), `localhost:9200`
|
||||
- **Write access** to `/root/zulip-health-monitor.log` and `/tmp/zulip-monitor-debounce`
|
||||
@@ -228,20 +228,20 @@ grep -a "Finalized\|Failed to finalize" /root/.pm2/logs/abiba-zulip-out.log | ta
|
||||
| Crash loop >10/h | Alert user |
|
||||
|
||||
|
||||
### Step 3: Platform B — Tanko (DSH on amdpve CT 112)
|
||||
### Step 3: Platform B — Tanko (DSH on minipve CT 112)
|
||||
|
||||
Mumuni is out of scope for this host (see the note above): she runs on her own
|
||||
container and is monitored on her side.
|
||||
|
||||
Tanko runs on DSH (DeepSeek Harness) — it no longer runs a Hermes gateway, so
|
||||
there is no `~/.hermes/gateway_state.json` on CT 112. Tanko's Zulip gateway runs
|
||||
as the `dsh-web` systemd unit inside **CT 112**, which resides on the **amdpve**
|
||||
PVE host (**192.168.68.15**). Direct SSH to 192.168.68.122 is not a dependency
|
||||
as the `dsh-web` systemd unit inside **CT 112**, which resides on the **minipve**
|
||||
PVE host (**192.168.68.12**). Direct SSH to 192.168.68.122 is not a dependency
|
||||
of this contract — per-worker key availability varies — so CT 112 probes run
|
||||
from the amdpve vantage via `pct exec`:
|
||||
from the minipve vantage via `pct exec`:
|
||||
|
||||
```bash
|
||||
ssh root@192.168.68.15 "pct exec 112 -- <command>"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- <command>"
|
||||
```
|
||||
|
||||
> **By design (verified 2026-09-08):** the `dsh-web` gateway binds
|
||||
@@ -253,7 +253,7 @@ ssh root@192.168.68.15 "pct exec 112 -- <command>"
|
||||
**B1: Gateway Service State (Tanko)**
|
||||
|
||||
```bash
|
||||
ssh root@192.168.68.15 "pct exec 112 -- systemctl is-active dsh-web"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- systemctl is-active dsh-web"
|
||||
```
|
||||
|
||||
Expected: `active`. Anything else → gateway service down → apply the Tanko heal
|
||||
@@ -262,7 +262,7 @@ Expected: `active`. Anything else → gateway service down → apply the Tanko h
|
||||
**B2: Gateway HTTP Liveness (Tanko — loopback-only :3080)**
|
||||
|
||||
```bash
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s --connect-timeout 5 --max-time 10 -o /dev/null -w '%{http_code}' http://127.0.0.1:3080/"
|
||||
```
|
||||
|
||||
Alive = **ANY** HTTP status response from the endpoint — the expected set is
|
||||
@@ -273,13 +273,13 @@ process answering `503` is running and self-heal must NOT restart-loop it.
|
||||
Down = connection refused (`000`) or timeout only. Statuses outside the
|
||||
expected set are logged/reported as a warning — reported, never healed on.
|
||||
|
||||
**B3: Public-URL Fallback Probe (Tanko — for nodes without pct/ssh access to amdpve)**
|
||||
**B3: Public-URL Fallback Probe (Tanko — for nodes without pct/ssh access to minipve)**
|
||||
|
||||
```bash
|
||||
curl -s --connect-timeout 10 --max-time 15 -o /dev/null -w '%{http_code}' https://tankodhs.sysloggh.net/
|
||||
```
|
||||
|
||||
Fallback only — used when the monitoring node has no pct/SSH path to amdpve.
|
||||
Fallback only — used when the monitoring node has no pct/SSH path to minipve.
|
||||
Alive = **ANY** HTTP status response from the endpoint — healthy signals are
|
||||
`302` (authentik proxy-auth redirect) and `401` (auth-gated), and any other
|
||||
status, including `404`/`5xx`, also counts alive: the endpoint is up and
|
||||
@@ -404,29 +404,29 @@ ExecStartPost=/bin/systemctl --no-block start dsh-web-token.service
|
||||
4. Every later request through `/` presents that cookie; the token is not needed
|
||||
again until the cookie expires or a new browser is used.
|
||||
|
||||
**Verification** (amdpve vantage):
|
||||
**Verification** (minipve vantage):
|
||||
```bash
|
||||
# 1. Login endpoint is Authentik-gated: unauthenticated -> 302 (not 200/303).
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}\n' \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}\n' \
|
||||
-H 'Host: tankodhs.sysloggh.net' http://127.0.0.1/dsh-web-login"
|
||||
# Expected: 302
|
||||
|
||||
# 2. Legacy :8081 endpoint is gone (connection refused -> 000).
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s --max-time 3 -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s --max-time 3 -o /dev/null \
|
||||
-w '%{http_code}\n' http://192.168.68.122:8081/"
|
||||
# Expected: 000
|
||||
|
||||
# 3. Backend cookie mint + reuse (exactly what /dsh-web-login proxies to).
|
||||
TOKEN=$(ssh root@192.168.68.15 "pct exec 112 -- cat /etc/dsh-web/launch-token")
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -c /tmp/dsh.jar -o /dev/null \
|
||||
TOKEN=$(ssh root@192.168.68.12 "pct exec 112 -- cat /etc/dsh-web/launch-token")
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -c /tmp/dsh.jar -o /dev/null \
|
||||
-H 'Host: tankodhs.sysloggh.net' 'http://127.0.0.1:3080/?token=$TOKEN'"
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
-w '%{http_code}\n' -H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/"
|
||||
# Expected: 200 — the minted dsh-auth-... cookie (authority
|
||||
# tankodhs.sysloggh.net) is replayed on the next request and accepted.
|
||||
|
||||
# 4. Token refresh is non-disruptive and idempotent.
|
||||
ssh root@192.168.68.15 "pct exec 112 -- /opt/deepseek-harness/capture-dsh-token.sh"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- /opt/deepseek-harness/capture-dsh-token.sh"
|
||||
# Expected: "token unchanged; nginx not reloaded" when nothing changed
|
||||
```
|
||||
|
||||
@@ -437,32 +437,32 @@ fresh cookie. Both verified live 2026-09-11.
|
||||
|
||||
```bash
|
||||
# 5. Cookie survives a dsh-web restart, and the new token mints a new cookie.
|
||||
ssh root@192.168.68.15 "pct exec 112 -- systemctl restart dsh-web"
|
||||
ssh root@192.168.68.12 "pct exec 112 -- systemctl restart dsh-web"
|
||||
# dsh-web is Type=simple: restart returns before :3080 is listening. Bounded-poll
|
||||
# until the socket answers (any status but 000) before asserting the cookie.
|
||||
for i in $(seq 1 60); do
|
||||
UP=$(ssh root@192.168.68.15 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
UP=$(ssh root@192.168.68.12 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
-H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/")
|
||||
[ "$UP" != "000" ] && break
|
||||
sleep 2
|
||||
done
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -b /tmp/dsh.jar -o /dev/null \
|
||||
-w '%{http_code}\n' -H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/"
|
||||
# Expected: 200 — the pre-restart cookie is still accepted.
|
||||
# The restart's ExecStartPost (or the 2-minute timer) refreshes the include. A
|
||||
# manual run may no-op on the flock, so poll until the include carries a token
|
||||
# the running process accepts (bounded wait) before the mint+reuse check.
|
||||
for i in $(seq 1 60); do
|
||||
TOKEN=$(ssh root@192.168.68.15 "pct exec 112 -- sed -n 's/.*token=//p' /etc/dsh-web/nginx-login.conf | tr -d ';\n'")
|
||||
CODE=$(ssh root@192.168.68.15 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
TOKEN=$(ssh root@192.168.68.12 "pct exec 112 -- sed -n 's/.*token=//p' /etc/dsh-web/nginx-login.conf | tr -d ';\n'")
|
||||
CODE=$(ssh root@192.168.68.12 "pct exec 112 -- curl -s -o /dev/null -w '%{http_code}' \
|
||||
-H 'Host: tankodhs.sysloggh.net' 'http://127.0.0.1:3080/?token=$TOKEN'")
|
||||
[ "$CODE" = "303" ] && break
|
||||
sleep 2
|
||||
done
|
||||
# Expected: 303 — the include now holds the token the running process accepts.
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -c /tmp/dsh-new.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -c /tmp/dsh-new.jar -o /dev/null \
|
||||
-H 'Host: tankodhs.sysloggh.net' 'http://127.0.0.1:3080/?token=$TOKEN'"
|
||||
ssh root@192.168.68.15 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null \
|
||||
ssh root@192.168.68.12 "pct exec 112 -- curl -s -b /tmp/dsh-new.jar -o /dev/null \
|
||||
-w '%{http_code}\n' -H 'Host: tankodhs.sysloggh.net' http://127.0.0.1:3080/"
|
||||
# Expected: 200 — the refreshed token minted a fresh cookie.
|
||||
```
|
||||
|
||||
Reference in New Issue
Block a user