feat(gpu-fleet): roster-driven GPU config + Ornith-1.0-35B

This commit is contained in:
Abiba
2026-06-28 16:24:54 +00:00
parent 8e0f6e407b
commit e9aa220ee8
3 changed files with 140 additions and 15 deletions
+55
View File
@@ -0,0 +1,55 @@
models:
gemma-4-12b:
gpu_url: http://192.168.68.110:8080/v1
sidecar_url: http://192.168.68.110:8090
gpu_host: 192.168.68.110
label: Gemma-4 12B (RTX 5070)
max_concurrent: 2
context: 262144
tiers: [starter, professional, enterprise]
capabilities: [completion, multimodal]
model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf
args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384
qwen3.6-27B-code:
gpu_url: http://192.168.68.8:8080/v1
sidecar_url: http://192.168.68.8:8090
gpu_host: 192.168.68.8
label: Qwen3.6 27B Code (RTX 3090)
max_concurrent: 2
context: 262144
tiers: [professional, enterprise]
capabilities: [completion]
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
ornith-1.0-35b:
gpu_url: http://192.168.68.15:8080/v1
sidecar_url: http://192.168.68.15:8090
gpu_host: 192.168.68.15
label: Ornith-1.0 35B (Strix Halo)
max_concurrent: 1
context: 4096
tiers: [professional, enterprise]
capabilities: [completion]
model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf
args: -c 262144 -ngl 99 --flash-attn on
hosts:
gpu-light:
address: 192.168.68.110
gpu_name: NVIDIA GeForce RTX 5070
vram_gb: 12
current_model: gemma-4-12b
gpu-dense:
address: 192.168.68.8
gpu_name: NVIDIA GeForce RTX 3090
vram_gb: 24
current_model: qwen3.6-27B-code
gpu-moe:
address: 192.168.68.15
gpu_name: AMD Strix Halo (iGPU)
vram_gb: 64
current_model: ornith-1.0-35b
+10 -15
View File
@@ -37,7 +37,7 @@ litellm_settings:
qwen3.6-27B-code: qwen3.6-27B-code:
input_cost_per_token: 0.0 input_cost_per_token: 0.0
output_cost_per_token: 0.0 output_cost_per_token: 0.0
qwen3.6-35B-A3B: ornith-1.0-35b:
input_cost_per_token: 0.0 input_cost_per_token: 0.0
output_cost_per_token: 0.0 output_cost_per_token: 0.0
syslog-auto: syslog-auto:
@@ -50,40 +50,35 @@ litellm_settings:
model_list: model_list:
- litellm_params: - litellm_params:
api_base: http://router:9000/v1 api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
model: openai/syslog-auto model: openai/syslog-auto
rpm: 600 rpm: 600
model_name: syslog-auto model_name: syslog-auto
- litellm_params: - litellm_params:
api_base: http://router:9000/v1 api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
model: openai/qwen3.6-35B-A3B
model_name: qwen3.6-35B-A3B
- litellm_params:
api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY
model: openai/qwen3.6-27B-code model: openai/qwen3.6-27B-code
model_name: qwen3.6-27B-code model_name: qwen3.6-27B-code
- litellm_params: - litellm_params:
api_base: http://router:9000/v1 api_base: http://router:9000/v1
api_key: os.environ/ROUTER_API_KEY api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
model: openai/gemma-4-12b model: openai/gemma-4-12b
model_name: gemma-4-12b model_name: gemma-4-12b
- litellm_params:
api_base: http://router:9000/v1
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
model: openai/ornith-1.0-35b
model_name: ornith-1.0-35b
router_settings: router_settings:
allowed_fails: 100 allowed_fails: 100
enable_loadbalancing_on_proxy: false enable_loadbalancing_on_proxy: false
fallbacks: fallbacks:
- syslog-auto: - syslog-auto:
- qwen3.6-35B-A3B
- qwen3.6-27B-code
- gemma-4-12b
- qwen3.6-35B-A3B:
- qwen3.6-27B-code - qwen3.6-27B-code
- gemma-4-12b - gemma-4-12b
- qwen3.6-27B-code: - qwen3.6-27B-code:
- qwen3.6-35B-A3B
- gemma-4-12b - gemma-4-12b
- gemma-4-12b: - gemma-4-12b:
- qwen3.6-27B-code - qwen3.6-27B-code
- qwen3.6-35B-A3B
routing_strategy: usage-based-routing routing_strategy: usage-based-routing
+75
View File
@@ -0,0 +1,75 @@
# GPU Roster Loader - reads gpu_roster.yaml via PyYAML
# Hot-reloadable via /admin/roster/reload endpoint
import os, json, threading, time
try:
import yaml
except ImportError:
yaml = None
ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml')
_last_mtime = 0
_lock = threading.Lock()
GPU_URLS = {}
GPU_SIDECARS = {}
GPU_LABELS = {}
GPU_MAX_CONCURRENT = {}
GPU_CONTEXT = {}
TIER_MODELS = {}
HOSTS = {}
def load_roster(path=None):
global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT
global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime
if yaml is None:
return False, 'PyYAML not installed - run: pip3 install pyyaml'
path = path or ROSTER_PATH
try:
with open(path) as f:
data = yaml.safe_load(f)
with _lock:
GPU_URLS.clear()
GPU_SIDECARS.clear()
GPU_LABELS.clear()
GPU_MAX_CONCURRENT.clear()
GPU_CONTEXT.clear()
TIER_MODELS.clear()
models = data.get('models', {})
for name, cfg in models.items():
GPU_URLS[name] = cfg.get('gpu_url', '')
GPU_SIDECARS[name] = cfg.get('sidecar_url', '')
GPU_LABELS[name] = cfg.get('label', name)
GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1)
GPU_CONTEXT[name] = cfg.get('context', 65536)
tiers = cfg.get('tiers', ['enterprise'])
for t in tiers:
if t not in TIER_MODELS:
TIER_MODELS[t] = []
if name not in TIER_MODELS[t]:
TIER_MODELS[t].append(name)
HOSTS.clear()
HOSTS.update(data.get('hosts', {}))
_last_mtime = os.path.getmtime(path)
return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS))
except Exception as e:
return False, str(e)
def check_reload():
global _last_mtime
try:
mtime = os.path.getmtime(ROSTER_PATH)
if mtime > _last_mtime:
success, msg = load_roster()
if success:
print('[ROSTER] Auto-reloaded:', msg)
except:
pass
def reload_thread(interval=30):
while True:
time.sleep(interval)
check_reload()
threading.Thread(target=reload_thread, daemon=True).start()