models: gemma-4-12b: gpu_url: http://192.168.68.110:8080/v1 sidecar_url: http://192.168.68.110:8090 gpu_host: 192.168.68.110 label: Gemma-4 12B (RTX 5070) max_concurrent: 2 context: 262144 tiers: [starter, professional, enterprise] capabilities: [completion, multimodal] model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384 qwen3.6-27B-code: gpu_url: http://192.168.68.8:8080/v1 sidecar_url: http://192.168.68.8:8090 gpu_host: 192.168.68.8 label: Qwen3.6 27B Code (RTX 3090) max_concurrent: 2 context: 262144 tiers: [professional, enterprise] capabilities: [completion] model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8 ornith-1.0-35b: gpu_url: http://192.168.68.15:8080/v1 sidecar_url: http://192.168.68.15:8090 gpu_host: 192.168.68.15 label: Ornith-1.0 35B (Strix Halo) max_concurrent: 1 context: 4096 tiers: [professional, enterprise] capabilities: [completion] model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf args: -c 262144 -ngl 99 --flash-attn on hosts: gpu-light: address: 192.168.68.110 gpu_name: NVIDIA GeForce RTX 5070 vram_gb: 12 current_model: gemma-4-12b gpu-dense: address: 192.168.68.8 gpu_name: NVIDIA GeForce RTX 3090 vram_gb: 24 current_model: qwen3.6-27B-code gpu-moe: address: 192.168.68.15 gpu_name: AMD Strix Halo (iGPU) vram_gb: 64 current_model: ornith-1.0-35b