fix(routing): restore gpu-dense (.8 qwen3.6-27B-code) as syslog-auto 0.55-weight member
The Aug 24 rename moved the syslog-auto top-weight member to .110/gpu-vision, bypassing gpu-dense entirely (0 requests/hr while strix-moe and .110 absorbed everything). Restore the intended 55/30/15 split: gpu-dense, strix-moe, .110.
This commit is contained in:
+131
-26
@@ -1,6 +1,9 @@
|
||||
general_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
store_model_in_db: true
|
||||
store_model_in_db: false
|
||||
user_url_allowed_hosts:
|
||||
- 192.168.68.14
|
||||
- 192.168.68.14:5080
|
||||
guardrails:
|
||||
- guardrail_name: input-moderation
|
||||
litellm_params:
|
||||
@@ -28,57 +31,159 @@ guardrails:
|
||||
guardrail: litellm_content_filter
|
||||
mode: pre_call
|
||||
litellm_settings:
|
||||
user_url_allowed_hosts:
|
||||
- 192.168.68.14
|
||||
- 192.168.68.14:5080
|
||||
cache: true
|
||||
cache_params:
|
||||
host: harness-redis
|
||||
namespace: litellm
|
||||
port: 6379
|
||||
ttl: 600
|
||||
type: redis
|
||||
drop_params: true
|
||||
success_callback:
|
||||
- prometheus
|
||||
failure_callback:
|
||||
- prometheus
|
||||
model_cost:
|
||||
gemma-4-12b:
|
||||
input_cost_per_token: 0.000075
|
||||
output_cost_per_token: 0.0003
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
qwen3.6-27B-code:
|
||||
input_cost_per_token: 0.00015
|
||||
output_cost_per_token: 0.0006
|
||||
ornith-1.0-35b:
|
||||
input_cost_per_token: 0.0002
|
||||
output_cost_per_token: 0.0008
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
strix-moe:
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
syslog-auto:
|
||||
input_cost_per_token: 0.0
|
||||
output_cost_per_token: 0.0
|
||||
num_retries: 0
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
num_retries: 2
|
||||
request_timeout: 600
|
||||
set_verbose: true
|
||||
sso_callback: /sso/callback
|
||||
model_list:
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
model: openai/syslog-auto
|
||||
rpm: 600
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
model_name: qwen3.6-27B-code
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/gemma-4-12b
|
||||
timeout: 120
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: gemma-4-12b
|
||||
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
model: openai/ornith-1.0-35b
|
||||
model_name: ornith-1.0-35b
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/strix-moe
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: qwen3.6-35B-udq4
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/strix-moe
|
||||
rpm: 40
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: strix-moe
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: gpu-dense
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: gpu-vision
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.55
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/strix-moe
|
||||
rpm: 60
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.3
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/gemma-4-12b
|
||||
rpm: 200
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.15
|
||||
model_name: syslog-auto
|
||||
router_settings:
|
||||
allowed_fails: 100
|
||||
enable_loadbalancing_on_proxy: false
|
||||
fallbacks:
|
||||
- syslog-auto:
|
||||
- qwen3.6-27B-code
|
||||
- strix-moe
|
||||
- gemma-4-12b
|
||||
- qwen3.6-27B-code:
|
||||
- gemma-4-12b
|
||||
- gemma-4-12b:
|
||||
- qwen3.6-27B-code
|
||||
- strix-moe:
|
||||
- qwen3.6-27B-code
|
||||
- gemma-4-12b
|
||||
request_timeout: 300
|
||||
routing_strategy: usage-based-routing
|
||||
|
||||
agents:
|
||||
- agent_name: agent-zero-homelab
|
||||
agent_card_params:
|
||||
name: Agent Zero HomeLab
|
||||
url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab
|
||||
protocolVersion: '1.0'
|
||||
|
||||
Reference in New Issue
Block a user