chore(ct116): capture production state (LiteLLM 1.99.1, nginx /ui //docs, router decommission, trove agent)

This commit is contained in:
agent-zero
2026-09-11 19:34:30 +00:00
parent 7cad063e27
commit 93e7605aa2
21 changed files with 196 additions and 1929 deletions
+21 -32
View File
@@ -47,9 +47,6 @@ litellm_settings:
failure_callback:
- prometheus
model_cost:
gemma-4-12b:
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
qwen3.6-27B-code:
input_cost_per_token: 1.5e-07
output_cost_per_token: 6.0e-07
@@ -61,7 +58,6 @@ litellm_settings:
output_cost_per_token: 6.0e-07
num_retries: 2
request_timeout: 600
set_verbose: true
sso_callback: /sso/callback
model_list:
- litellm_params:
@@ -72,16 +68,6 @@ model_list:
model_info:
max_input_tokens: 131072
model_name: qwen3.6-27B-code
- litellm_params:
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
model: openai/gemma-4-12b
timeout: 120
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: gemma-4-12b
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
@@ -102,10 +88,6 @@ model_list:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: strix-moe
- litellm_params:
api_base: http://192.168.68.8:8080/v1
@@ -119,9 +101,9 @@ model_list:
max_tokens: 131072
model_name: gpu-dense
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
model: openai/gpu-vision
rpm: 500
timeout: 300
model_info:
@@ -137,12 +119,21 @@ model_list:
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.55
max_model_tokens: 131072
max_tokens: 131072
weight: 0.70
model_name: syslog-auto
- litellm_params:
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
model: openai/qwen3.6-27B-code
rpm: 500
timeout: 300
model_info:
max_input_tokens: 131072
max_model_tokens: 131072
max_tokens: 131072
model_name: syslog-auto
model_name: qwen3.8-27B-uncensored
- litellm_params:
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
@@ -151,35 +142,33 @@ model_list:
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.3
weight: 0.20
model_name: syslog-auto
- litellm_params:
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
model: openai/gemma-4-12b
model: openai/gpu-vision
rpm: 200
timeout: 300
model_info:
max_input_tokens: 131072
weight: 0.15
weight: 0.10
model_name: syslog-auto
router_settings:
allowed_fails: 100
enable_loadbalancing_on_proxy: false
enable_loadbalancing_on_proxy: true
fallbacks:
- syslog-auto:
- qwen3.6-27B-code
- strix-moe
- gemma-4-12b
- gpu-vision
- qwen3.6-27B-code:
- gemma-4-12b
- gemma-4-12b:
- qwen3.6-27B-code
- strix-moe
- strix-moe:
- qwen3.6-27B-code
- gemma-4-12b
- gpu-vision
request_timeout: 300
routing_strategy: usage-based-routing
routing_strategy: simple-shuffle
agents:
- agent_name: agent-zero-homelab