Archive 688 GPU self-heal logs from shared memory

This commit is contained in:
2026-07-18 00:36:06 -04:00
parent 9acabf7ba6
commit 937ba2c1ce
19 changed files with 7893 additions and 251 deletions
+18 -86
View File
@@ -1,89 +1,21 @@
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
store_model_in_db: true
guardrails:
- guardrail_name: input-moderation
litellm_params:
guardrail: openai_moderation
mode: pre_call
- guardrail_name: output-moderation
litellm_params:
guardrail: openai_moderation
mode: post_call
- guardrail_name: harmful-content-filter
litellm_params:
categories:
- action: BLOCK
category: harmful_self_harm
enabled: true
severity_threshold: medium
- action: BLOCK
category: harmful_violence
enabled: true
severity_threshold: medium
- action: BLOCK
category: harmful_illegal_weapons
enabled: true
severity_threshold: medium
guardrail: litellm_content_filter
mode: pre_call
litellm_settings:
failure_callback:
- prometheus
model_cost:
gemma-4-12b:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
qwen3.6-27B-code:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
qwen3.6-35B-A3B:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
syslog-auto:
input_cost_per_token: 0.0
output_cost_per_token: 0.0
num_retries: 0
request_timeout: 600
set_verbose: true
sso_callback: /sso/callback
model_list:
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/syslog-auto
rpm: 600
model_name: syslog-auto
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
- model_name: qwen3.6-35B-A3B
litellm_params:
model: openai/qwen3.6-35B-A3B
model_name: qwen3.6-35B-A3B
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/qwen3.6-27B-code
model_name: qwen3.6-27B-code
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
api_base: http://192.168.68.15:8080/v1
api_key: not-needed
- model_name: gpu-dense
litellm_params:
model: openai/qwen3.6-27B-code-text
api_base: http://192.168.68.8:8080/v1
api_key: not-needed
- model_name: gpu-light
litellm_params:
model: openai/gemma-4-12b
model_name: gemma-4-12b
router_settings:
allowed_fails: 100
enable_loadbalancing_on_proxy: false
fallbacks:
- syslog-auto:
- qwen3.6-35B-A3B
- qwen3.6-27B-code
- gemma-4-12b
- qwen3.6-35B-A3B:
- qwen3.6-27B-code
- gemma-4-12b
- qwen3.6-27B-code:
- qwen3.6-35B-A3B
- gemma-4-12b
- gemma-4-12b:
- qwen3.6-27B-code
- qwen3.6-35B-A3B
routing_strategy: usage-based-routing
api_base: http://192.168.68.110:8080/v1
api_key: not-needed
general_settings:
master_key: sk-syslog-local-master-key
litellm_settings:
drop_params: true
request_timeout: 120