litellm_config.yaml - client-visible model_list reduced to capability names: syslog-auto, gpu-dense, strix-moe, gpu-vision - retired qwen3.6-27B-code, qwen3.8-27B-uncensored, qwen3.6-35B-udq4 (and the already-dead gemma-4-12b) - fallbacks and model_cost re-keyed to the surviving names - upstream litellm_params.model set to the capability names; the backends ignore the model field (verified HTTP 200 on all three hosts), so this needs no llama.cpp relaunch and loses no KV warmth - verified live: /v1/models advertises exactly the four names, each returns 200 through nginx gpu_roster.yaml - launch args replaced with the VERIFIED live command lines, read from the running processes via the PVE guest agent (acerpve VM 101 = .8, ocupve VM 103 = .110) and on amdpve (.15) - .8: model_path corrected to Qwen3.8-27B-Uncensored-Q4_K_M.gguf, max_concurrent 2 -> 1, context 262144 -> 131072, full arg list recorded (incl. --parallel 1 and the new --slot-save-path / --metrics added 2026-09-12) - .15: model_path corrected to Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf, full arg list recorded - keys renamed to the capability names; hosts.current_model aligned README.md / dashboard - README dense entry corrected (1 slot, 131K ctx, actual model file) - dashboard picker now uses capability names; fixed the "Gemma 4 12B" / "12B VLM" labels (the RTX 5070 serves a 9B Qwen3.5) and the gpu-vision -> gpu-light id mapping scripts/ - added the three operational monitors as tracked files (they were untracked): gpu-monitor.py, gpu-self-heal.py, litellm-health-check.sh - cleared their references to retired model names, which were causing failed calls every benchmark cycle (150 failed gemma-4-12b calls in the last 7 days); gpu-monitor.py's .110 entry also wrongly listed .8's model Intentionally NOT changed - LITELLM-MIGRATION-PLAN.md: historical planning document (June 14), not a live-state claim - backups/, graphify-out/, litellm_config.yaml.backup: historical artifacts - unrelated untracked files (router.py, docker-compose.yml.pre-1991-20260911, nginx/default.conf, dashboard/gpu-monitor.html, scripts/gitea-logger.sh): out of scope for this change
153 lines
3.5 KiB
YAML
153 lines
3.5 KiB
YAML
general_settings:
|
|
master_key: os.environ/LITELLM_MASTER_KEY
|
|
store_model_in_db: false
|
|
user_url_allowed_hosts:
|
|
- 192.168.68.14
|
|
- 192.168.68.14:5080
|
|
guardrails:
|
|
- guardrail_name: input-moderation
|
|
litellm_params:
|
|
guardrail: openai_moderation
|
|
mode: pre_call
|
|
- guardrail_name: output-moderation
|
|
litellm_params:
|
|
guardrail: openai_moderation
|
|
mode: post_call
|
|
- guardrail_name: harmful-content-filter
|
|
litellm_params:
|
|
categories:
|
|
- action: BLOCK
|
|
category: harmful_self_harm
|
|
enabled: true
|
|
severity_threshold: medium
|
|
- action: BLOCK
|
|
category: harmful_violence
|
|
enabled: true
|
|
severity_threshold: medium
|
|
- action: BLOCK
|
|
category: harmful_illegal_weapons
|
|
enabled: true
|
|
severity_threshold: medium
|
|
guardrail: litellm_content_filter
|
|
mode: pre_call
|
|
litellm_settings:
|
|
user_url_allowed_hosts:
|
|
- 192.168.68.14
|
|
- 192.168.68.14:5080
|
|
cache: true
|
|
cache_params:
|
|
host: harness-redis
|
|
namespace: litellm
|
|
port: 6379
|
|
ttl: 600
|
|
type: redis
|
|
drop_params: true
|
|
success_callback:
|
|
- prometheus
|
|
failure_callback:
|
|
- prometheus
|
|
model_cost:
|
|
gpu-dense:
|
|
input_cost_per_token: 1.5e-07
|
|
output_cost_per_token: 6.0e-07
|
|
strix-moe:
|
|
input_cost_per_token: 1.5e-07
|
|
output_cost_per_token: 6.0e-07
|
|
syslog-auto:
|
|
input_cost_per_token: 1.5e-07
|
|
output_cost_per_token: 6.0e-07
|
|
gpu-vision:
|
|
input_cost_per_token: 1.5e-07
|
|
output_cost_per_token: 6.0e-07
|
|
num_retries: 2
|
|
request_timeout: 600
|
|
sso_callback: /sso/callback
|
|
model_list:
|
|
- litellm_params:
|
|
api_base: http://192.168.68.15:8080/v1
|
|
api_key: not-needed
|
|
model: openai/strix-moe
|
|
rpm: 40
|
|
timeout: 300
|
|
model_info:
|
|
max_input_tokens: 131072
|
|
max_model_tokens: 131072
|
|
max_tokens: 131072
|
|
model_name: strix-moe
|
|
- litellm_params:
|
|
api_base: http://192.168.68.8:8080/v1
|
|
api_key: not-needed
|
|
model: openai/gpu-dense
|
|
rpm: 500
|
|
timeout: 300
|
|
model_info:
|
|
max_input_tokens: 131072
|
|
max_model_tokens: 131072
|
|
max_tokens: 131072
|
|
model_name: gpu-dense
|
|
- litellm_params:
|
|
api_base: http://192.168.68.110:8080/v1
|
|
api_key: not-needed
|
|
model: openai/gpu-vision
|
|
rpm: 500
|
|
timeout: 300
|
|
model_info:
|
|
max_input_tokens: 131072
|
|
max_model_tokens: 131072
|
|
max_tokens: 131072
|
|
model_name: gpu-vision
|
|
- litellm_params:
|
|
api_base: http://192.168.68.8:8080/v1
|
|
api_key: not-needed
|
|
model: openai/gpu-dense
|
|
rpm: 500
|
|
timeout: 300
|
|
model_info:
|
|
max_input_tokens: 131072
|
|
max_model_tokens: 131072
|
|
max_tokens: 131072
|
|
weight: 0.70
|
|
model_name: syslog-auto
|
|
- litellm_params:
|
|
api_base: http://192.168.68.15:8080/v1
|
|
api_key: not-needed
|
|
model: openai/strix-moe
|
|
rpm: 60
|
|
timeout: 300
|
|
model_info:
|
|
max_input_tokens: 131072
|
|
weight: 0.20
|
|
model_name: syslog-auto
|
|
- litellm_params:
|
|
api_base: http://192.168.68.110:8080/v1
|
|
api_key: not-needed
|
|
model: openai/gpu-vision
|
|
rpm: 200
|
|
timeout: 300
|
|
model_info:
|
|
max_input_tokens: 131072
|
|
weight: 0.10
|
|
model_name: syslog-auto
|
|
router_settings:
|
|
allowed_fails: 100
|
|
enable_loadbalancing_on_proxy: true
|
|
fallbacks:
|
|
- syslog-auto:
|
|
- gpu-dense
|
|
- strix-moe
|
|
- gpu-vision
|
|
- gpu-dense:
|
|
- strix-moe
|
|
- strix-moe:
|
|
- gpu-dense
|
|
- gpu-vision
|
|
request_timeout: 300
|
|
routing_strategy: simple-shuffle
|
|
|
|
agents:
|
|
- agent_name: agent-zero-homelab
|
|
agent_card_params:
|
|
name: Agent Zero HomeLab
|
|
url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab
|
|
protocolVersion: '1.0'
|