general_settings: master_key: os.environ/LITELLM_MASTER_KEY store_model_in_db: false user_url_allowed_hosts: - 192.168.68.14 - 192.168.68.14:5080 guardrails: - guardrail_name: input-moderation litellm_params: guardrail: openai_moderation mode: pre_call - guardrail_name: output-moderation litellm_params: guardrail: openai_moderation mode: post_call - guardrail_name: harmful-content-filter litellm_params: categories: - action: BLOCK category: harmful_self_harm enabled: true severity_threshold: medium - action: BLOCK category: harmful_violence enabled: true severity_threshold: medium - action: BLOCK category: harmful_illegal_weapons enabled: true severity_threshold: medium guardrail: litellm_content_filter mode: pre_call litellm_settings: user_url_allowed_hosts: - 192.168.68.14 - 192.168.68.14:5080 cache: true cache_params: host: harness-redis namespace: litellm port: 6379 ttl: 600 type: redis drop_params: true success_callback: - prometheus failure_callback: - prometheus model_cost: qwen3.6-27B-code: input_cost_per_token: 1.5e-07 output_cost_per_token: 6.0e-07 strix-moe: input_cost_per_token: 1.5e-07 output_cost_per_token: 6.0e-07 syslog-auto: input_cost_per_token: 1.5e-07 output_cost_per_token: 6.0e-07 num_retries: 2 request_timeout: 600 sso_callback: /sso/callback model_list: - litellm_params: api_base: http://192.168.68.8:8080/v1 api_key: not-needed model: openai/qwen3.6-27B-code timeout: 300 model_info: max_input_tokens: 131072 model_name: qwen3.6-27B-code - litellm_params: api_base: http://192.168.68.15:8080/v1 api_key: not-needed model: openai/strix-moe timeout: 300 model_info: max_input_tokens: 131072 max_model_tokens: 131072 max_tokens: 131072 model_name: qwen3.6-35B-udq4 - litellm_params: api_base: http://192.168.68.15:8080/v1 api_key: not-needed model: openai/strix-moe rpm: 40 timeout: 300 model_info: max_input_tokens: 131072 max_model_tokens: 131072 max_tokens: 131072 model_name: strix-moe - litellm_params: api_base: http://192.168.68.8:8080/v1 api_key: not-needed model: openai/qwen3.6-27B-code rpm: 500 timeout: 300 model_info: max_input_tokens: 131072 max_model_tokens: 131072 max_tokens: 131072 model_name: gpu-dense - litellm_params: api_base: http://192.168.68.110:8080/v1 api_key: not-needed model: openai/gpu-vision rpm: 500 timeout: 300 model_info: max_input_tokens: 131072 max_model_tokens: 131072 max_tokens: 131072 model_name: gpu-vision - litellm_params: api_base: http://192.168.68.8:8080/v1 api_key: not-needed model: openai/qwen3.6-27B-code rpm: 500 timeout: 300 model_info: max_input_tokens: 131072 max_model_tokens: 131072 max_tokens: 131072 weight: 0.70 model_name: syslog-auto - litellm_params: api_base: http://192.168.68.8:8080/v1 api_key: not-needed model: openai/qwen3.6-27B-code rpm: 500 timeout: 300 model_info: max_input_tokens: 131072 max_model_tokens: 131072 max_tokens: 131072 model_name: qwen3.8-27B-uncensored - litellm_params: api_base: http://192.168.68.15:8080/v1 api_key: not-needed model: openai/strix-moe rpm: 60 timeout: 300 model_info: max_input_tokens: 131072 weight: 0.20 model_name: syslog-auto - litellm_params: api_base: http://192.168.68.110:8080/v1 api_key: not-needed model: openai/gpu-vision rpm: 200 timeout: 300 model_info: max_input_tokens: 131072 weight: 0.10 model_name: syslog-auto router_settings: allowed_fails: 100 enable_loadbalancing_on_proxy: true fallbacks: - syslog-auto: - qwen3.6-27B-code - strix-moe - gpu-vision - qwen3.6-27B-code: - strix-moe - strix-moe: - qwen3.6-27B-code - gpu-vision request_timeout: 300 routing_strategy: simple-shuffle agents: - agent_name: agent-zero-homelab agent_card_params: name: Agent Zero HomeLab url: http://192.168.68.14:5080/a2a/t-8zNgdOEXzYxjQvTl/p-homelab protocolVersion: '1.0'