chore(ct116): capture production state (LiteLLM 1.99.1, nginx /ui //docs, router decommission, trove agent)
This commit is contained in:
+21
-32
@@ -47,9 +47,6 @@ litellm_settings:
|
||||
failure_callback:
|
||||
- prometheus
|
||||
model_cost:
|
||||
gemma-4-12b:
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
qwen3.6-27B-code:
|
||||
input_cost_per_token: 1.5e-07
|
||||
output_cost_per_token: 6.0e-07
|
||||
@@ -61,7 +58,6 @@ litellm_settings:
|
||||
output_cost_per_token: 6.0e-07
|
||||
num_retries: 2
|
||||
request_timeout: 600
|
||||
set_verbose: true
|
||||
sso_callback: /sso/callback
|
||||
model_list:
|
||||
- litellm_params:
|
||||
@@ -72,16 +68,6 @@ model_list:
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
model_name: qwen3.6-27B-code
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/gemma-4-12b
|
||||
timeout: 120
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: gemma-4-12b
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
@@ -102,10 +88,6 @@ model_list:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: strix-moe
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
@@ -119,9 +101,9 @@ model_list:
|
||||
max_tokens: 131072
|
||||
model_name: gpu-dense
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
model: openai/gpu-vision
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
@@ -137,12 +119,21 @@ model_list:
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.55
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
weight: 0.70
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.8:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/qwen3.6-27B-code
|
||||
rpm: 500
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
max_model_tokens: 131072
|
||||
max_tokens: 131072
|
||||
model_name: syslog-auto
|
||||
model_name: qwen3.8-27B-uncensored
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.15:8080/v1
|
||||
api_key: not-needed
|
||||
@@ -151,35 +142,33 @@ model_list:
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.3
|
||||
weight: 0.20
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://192.168.68.110:8080/v1
|
||||
api_key: not-needed
|
||||
model: openai/gemma-4-12b
|
||||
model: openai/gpu-vision
|
||||
rpm: 200
|
||||
timeout: 300
|
||||
model_info:
|
||||
max_input_tokens: 131072
|
||||
weight: 0.15
|
||||
weight: 0.10
|
||||
model_name: syslog-auto
|
||||
router_settings:
|
||||
allowed_fails: 100
|
||||
enable_loadbalancing_on_proxy: false
|
||||
enable_loadbalancing_on_proxy: true
|
||||
fallbacks:
|
||||
- syslog-auto:
|
||||
- qwen3.6-27B-code
|
||||
- strix-moe
|
||||
- gemma-4-12b
|
||||
- gpu-vision
|
||||
- qwen3.6-27B-code:
|
||||
- gemma-4-12b
|
||||
- gemma-4-12b:
|
||||
- qwen3.6-27B-code
|
||||
- strix-moe
|
||||
- strix-moe:
|
||||
- qwen3.6-27B-code
|
||||
- gemma-4-12b
|
||||
- gpu-vision
|
||||
request_timeout: 300
|
||||
routing_strategy: usage-based-routing
|
||||
routing_strategy: simple-shuffle
|
||||
|
||||
agents:
|
||||
- agent_name: agent-zero-homelab
|
||||
|
||||
Reference in New Issue
Block a user