Files
inference-harness/gpu_roster.yaml
T

55 lines
1.8 KiB
YAML

models:
gpu-vision:
gpu_url: http://192.168.68.110:8080/v1
gpu_host: 192.168.68.110
label: Qwen3.5-9B Vision (RTX 5070)
max_concurrent: 1
context: 131072
tiers: [starter, professional, enterprise]
capabilities: [completion, multimodal]
model_path: /home/llmuser/models/qwen3.5-9b/Qwen3.5-9B-Q5_K_M.gguf
args: --mmproj /home/llmuser/models/qwen3.5-9b/mmproj-F16.gguf --ctx-size 131072 --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn 1 --parallel 1 --alias gpu-light --reasoning off --api-key not-needed --cache-prompt
qwen3.6-27B-code:
gpu_url: http://192.168.68.8:8080/v1
sidecar_url: http://192.168.68.8:8090
gpu_host: 192.168.68.8
label: Qwen3.6 27B Code (RTX 3090)
max_concurrent: 2
context: 262144
tiers: [professional, enterprise]
capabilities: [completion]
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
args: --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
qwen3.6-35B-udq4:
gpu_url: http://192.168.68.15:8080/v1
sidecar_url: http://192.168.68.15:8090
gpu_host: 192.168.68.15
label: Qwen3.6 35B UD-Q4_K_M (Strix Halo)
max_concurrent: 1
context: 262144
tiers: [professional, enterprise]
capabilities: [completion, multimodal]
model_path: /models/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf
args: -c 262144 -ngl 99 --flash-attn on
hosts:
gpu-light:
address: 192.168.68.110
gpu_name: NVIDIA GeForce RTX 5070
vram_gb: 12
current_model: gpu-vision
gpu-dense:
address: 192.168.68.8
gpu_name: NVIDIA GeForce RTX 3090
vram_gb: 24
current_model: qwen3.6-27B-code
gpu-moe:
address: 192.168.68.15
gpu_name: AMD Strix Halo (iGPU)
vram_gb: 64
current_model: qwen3.6-35B-udq4