chore(ct116): capture production state (LiteLLM 1.99.1, nginx /ui //docs, router decommission, trove agent)
This commit is contained in:
+14
-15
@@ -1,15 +1,14 @@
|
||||
models:
|
||||
gemma-4-12b:
|
||||
gpu-vision:
|
||||
gpu_url: http://192.168.68.110:8080/v1
|
||||
sidecar_url: http://192.168.68.110:8090
|
||||
gpu_host: 192.168.68.110
|
||||
label: Gemma-4 12B (RTX 5070)
|
||||
max_concurrent: 2
|
||||
context: 262144
|
||||
label: Qwen3.5-9B Vision (RTX 5070)
|
||||
max_concurrent: 1
|
||||
context: 131072
|
||||
tiers: [starter, professional, enterprise]
|
||||
capabilities: [completion, multimodal]
|
||||
model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf
|
||||
args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384
|
||||
model_path: /home/llmuser/models/qwen3.5-9b/Qwen3.5-9B-Q5_K_M.gguf
|
||||
args: --mmproj /home/llmuser/models/qwen3.5-9b/mmproj-F16.gguf --ctx-size 131072 --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn 1 --parallel 1 --alias gpu-light --reasoning off --api-key not-needed --cache-prompt
|
||||
|
||||
qwen3.6-27B-code:
|
||||
gpu_url: http://192.168.68.8:8080/v1
|
||||
@@ -21,18 +20,18 @@ models:
|
||||
tiers: [professional, enterprise]
|
||||
capabilities: [completion]
|
||||
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
|
||||
args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
|
||||
args: --cache-type-k q4_0 --cache-type-v q4_0 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
|
||||
|
||||
ornith-1.0-35b:
|
||||
qwen3.6-35B-udq4:
|
||||
gpu_url: http://192.168.68.15:8080/v1
|
||||
sidecar_url: http://192.168.68.15:8090
|
||||
gpu_host: 192.168.68.15
|
||||
label: Ornith-1.0 35B (Strix Halo)
|
||||
label: Qwen3.6 35B UD-Q4_K_M (Strix Halo)
|
||||
max_concurrent: 1
|
||||
context: 4096
|
||||
context: 262144
|
||||
tiers: [professional, enterprise]
|
||||
capabilities: [completion]
|
||||
model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf
|
||||
capabilities: [completion, multimodal]
|
||||
model_path: /models/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf
|
||||
args: -c 262144 -ngl 99 --flash-attn on
|
||||
|
||||
hosts:
|
||||
@@ -40,7 +39,7 @@ hosts:
|
||||
address: 192.168.68.110
|
||||
gpu_name: NVIDIA GeForce RTX 5070
|
||||
vram_gb: 12
|
||||
current_model: gemma-4-12b
|
||||
current_model: gpu-vision
|
||||
|
||||
gpu-dense:
|
||||
address: 192.168.68.8
|
||||
@@ -52,4 +51,4 @@ hosts:
|
||||
address: 192.168.68.15
|
||||
gpu_name: AMD Strix Halo (iGPU)
|
||||
vram_gb: 64
|
||||
current_model: ornith-1.0-35b
|
||||
current_model: qwen3.6-35B-udq4
|
||||
|
||||
Reference in New Issue
Block a user