fix(routing): restore gpu-dense (.8 qwen3.6-27B-code) as syslog-auto 0.55-weight member

The Aug 24 rename moved the syslog-auto top-weight member to .110/gpu-vision,
bypassing gpu-dense entirely (0 requests/hr while strix-moe and .110 absorbed
everything). Restore the intended 55/30/15 split: gpu-dense, strix-moe, .110.
This commit is contained in:
Abiba
2026-08-28 14:53:51 +00:00
parent ce2b90ed50
commit 7cad063e27
+131 -26
View File
@@ -1,6 +1,9 @@
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
store_model_in_db: true
store_model_in_db: false
user_url_allowed_hosts:
- 192.168.68.14
- 192.168.68.14:5080
guardrails:
- guardrail_name: input-moderation
litellm_params:
@@ -28,57 +31,159 @@ guardrails:
guardrail: litellm_content_filter
mode: pre_call
litellm_settings:
user_url_allowed_hosts:
- 192.168.68.14
- 192.168.68.14:5080
cache: true
cache_params:
host: harness-redis
namespace: litellm
port: 6379
ttl: 600
type: redis
drop_params: true
success_callback:
- prometheus
failure_callback:
- prometheus
model_cost:
gemma-4-12b:
input_cost_per_token: 0.000075
output_cost_per_token: 0.0003
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
qwen3.6-27B-code:
input_cost_per_token: 0.00015
output_cost_per_token: 0.0006
ornith-1.0-35b:
input_cost_per_token: 0.0002
output_cost_per_token: 0.0008
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
strix-moe:
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
syslog-auto:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
num_retries: 0
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
num_retries: 2
request_timeout: 600
set_verbose: true
sso_callback: /sso/callback
model_list:
- litellm_params:
api_base: http://router:9000/v1
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
model: openai/syslog-auto
rpm: 600
model_name: syslog-auto
- litellm_params:
api_base: http://router:9000/v1
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
timeout: 300
model_info:
max_input_tokens: 131072
model_name: qwen3.6-27B-code
- litellm_params:
api_base: http://router:9000/v1
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
model: openai/gemma-4-12b
timeout: 120
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: gemma-4-12b
- litellm_params:
api_base: http://router:9000/v1
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
model: openai/ornith-1.0-35b
model_name: ornith-1.0-35b
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
model: openai/strix-moe
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: qwen3.6-35B-udq4
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
model: openai/strix-moe
rpm: 40
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: strix-moe
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: gpu-dense
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: gpu-vision
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.55
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: syslog-auto
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
model: openai/strix-moe
rpm: 60
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.3
model_name: syslog-auto
- litellm_params:
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
model: openai/gemma-4-12b
rpm: 200
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.15
model_name: syslog-auto
router_settings:
allowed_fails: 100
enable_loadbalancing_on_proxy: false
fallbacks:
- syslog-auto:
- qwen3.6-27B-code
- strix-moe
- gemma-4-12b
- qwen3.6-27B-code:
- gemma-4-12b
- gemma-4-12b:
- qwen3.6-27B-code
- strix-moe:
- qwen3.6-27B-code
- gemma-4-12b
request_timeout: 300
routing_strategy: usage-based-routing
agents:
- agent_name: agent-zero-homelab
agent_card_params:
name: Agent Zero HomeLab
url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab
protocolVersion: '1.0'