feat(gpu-fleet): roster-driven GPU config + Ornith-1.0-35B
This commit is contained in:
@@ -0,0 +1,55 @@
|
|||||||
|
models:
|
||||||
|
gemma-4-12b:
|
||||||
|
gpu_url: http://192.168.68.110:8080/v1
|
||||||
|
sidecar_url: http://192.168.68.110:8090
|
||||||
|
gpu_host: 192.168.68.110
|
||||||
|
label: Gemma-4 12B (RTX 5070)
|
||||||
|
max_concurrent: 2
|
||||||
|
context: 262144
|
||||||
|
tiers: [starter, professional, enterprise]
|
||||||
|
capabilities: [completion, multimodal]
|
||||||
|
model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf
|
||||||
|
args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384
|
||||||
|
|
||||||
|
qwen3.6-27B-code:
|
||||||
|
gpu_url: http://192.168.68.8:8080/v1
|
||||||
|
sidecar_url: http://192.168.68.8:8090
|
||||||
|
gpu_host: 192.168.68.8
|
||||||
|
label: Qwen3.6 27B Code (RTX 3090)
|
||||||
|
max_concurrent: 2
|
||||||
|
context: 262144
|
||||||
|
tiers: [professional, enterprise]
|
||||||
|
capabilities: [completion]
|
||||||
|
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
|
||||||
|
args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
|
||||||
|
|
||||||
|
ornith-1.0-35b:
|
||||||
|
gpu_url: http://192.168.68.15:8080/v1
|
||||||
|
sidecar_url: http://192.168.68.15:8090
|
||||||
|
gpu_host: 192.168.68.15
|
||||||
|
label: Ornith-1.0 35B (Strix Halo)
|
||||||
|
max_concurrent: 1
|
||||||
|
context: 4096
|
||||||
|
tiers: [professional, enterprise]
|
||||||
|
capabilities: [completion]
|
||||||
|
model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf
|
||||||
|
args: -c 262144 -ngl 99 --flash-attn on
|
||||||
|
|
||||||
|
hosts:
|
||||||
|
gpu-light:
|
||||||
|
address: 192.168.68.110
|
||||||
|
gpu_name: NVIDIA GeForce RTX 5070
|
||||||
|
vram_gb: 12
|
||||||
|
current_model: gemma-4-12b
|
||||||
|
|
||||||
|
gpu-dense:
|
||||||
|
address: 192.168.68.8
|
||||||
|
gpu_name: NVIDIA GeForce RTX 3090
|
||||||
|
vram_gb: 24
|
||||||
|
current_model: qwen3.6-27B-code
|
||||||
|
|
||||||
|
gpu-moe:
|
||||||
|
address: 192.168.68.15
|
||||||
|
gpu_name: AMD Strix Halo (iGPU)
|
||||||
|
vram_gb: 64
|
||||||
|
current_model: ornith-1.0-35b
|
||||||
+10
-15
@@ -37,7 +37,7 @@ litellm_settings:
|
|||||||
qwen3.6-27B-code:
|
qwen3.6-27B-code:
|
||||||
input_cost_per_token: 0.0
|
input_cost_per_token: 0.0
|
||||||
output_cost_per_token: 0.0
|
output_cost_per_token: 0.0
|
||||||
qwen3.6-35B-A3B:
|
ornith-1.0-35b:
|
||||||
input_cost_per_token: 0.0
|
input_cost_per_token: 0.0
|
||||||
output_cost_per_token: 0.0
|
output_cost_per_token: 0.0
|
||||||
syslog-auto:
|
syslog-auto:
|
||||||
@@ -50,40 +50,35 @@ litellm_settings:
|
|||||||
model_list:
|
model_list:
|
||||||
- litellm_params:
|
- litellm_params:
|
||||||
api_base: http://router:9000/v1
|
api_base: http://router:9000/v1
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||||
model: openai/syslog-auto
|
model: openai/syslog-auto
|
||||||
rpm: 600
|
rpm: 600
|
||||||
model_name: syslog-auto
|
model_name: syslog-auto
|
||||||
- litellm_params:
|
- litellm_params:
|
||||||
api_base: http://router:9000/v1
|
api_base: http://router:9000/v1
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||||
model: openai/qwen3.6-35B-A3B
|
|
||||||
model_name: qwen3.6-35B-A3B
|
|
||||||
- litellm_params:
|
|
||||||
api_base: http://router:9000/v1
|
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
|
||||||
model: openai/qwen3.6-27B-code
|
model: openai/qwen3.6-27B-code
|
||||||
model_name: qwen3.6-27B-code
|
model_name: qwen3.6-27B-code
|
||||||
- litellm_params:
|
- litellm_params:
|
||||||
api_base: http://router:9000/v1
|
api_base: http://router:9000/v1
|
||||||
api_key: os.environ/ROUTER_API_KEY
|
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||||
model: openai/gemma-4-12b
|
model: openai/gemma-4-12b
|
||||||
model_name: gemma-4-12b
|
model_name: gemma-4-12b
|
||||||
|
|
||||||
|
- litellm_params:
|
||||||
|
api_base: http://router:9000/v1
|
||||||
|
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||||
|
model: openai/ornith-1.0-35b
|
||||||
|
model_name: ornith-1.0-35b
|
||||||
router_settings:
|
router_settings:
|
||||||
allowed_fails: 100
|
allowed_fails: 100
|
||||||
enable_loadbalancing_on_proxy: false
|
enable_loadbalancing_on_proxy: false
|
||||||
fallbacks:
|
fallbacks:
|
||||||
- syslog-auto:
|
- syslog-auto:
|
||||||
- qwen3.6-35B-A3B
|
|
||||||
- qwen3.6-27B-code
|
|
||||||
- gemma-4-12b
|
|
||||||
- qwen3.6-35B-A3B:
|
|
||||||
- qwen3.6-27B-code
|
- qwen3.6-27B-code
|
||||||
- gemma-4-12b
|
- gemma-4-12b
|
||||||
- qwen3.6-27B-code:
|
- qwen3.6-27B-code:
|
||||||
- qwen3.6-35B-A3B
|
|
||||||
- gemma-4-12b
|
- gemma-4-12b
|
||||||
- gemma-4-12b:
|
- gemma-4-12b:
|
||||||
- qwen3.6-27B-code
|
- qwen3.6-27B-code
|
||||||
- qwen3.6-35B-A3B
|
|
||||||
routing_strategy: usage-based-routing
|
routing_strategy: usage-based-routing
|
||||||
|
|||||||
@@ -0,0 +1,75 @@
|
|||||||
|
# GPU Roster Loader - reads gpu_roster.yaml via PyYAML
|
||||||
|
# Hot-reloadable via /admin/roster/reload endpoint
|
||||||
|
|
||||||
|
import os, json, threading, time
|
||||||
|
|
||||||
|
try:
|
||||||
|
import yaml
|
||||||
|
except ImportError:
|
||||||
|
yaml = None
|
||||||
|
|
||||||
|
ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml')
|
||||||
|
_last_mtime = 0
|
||||||
|
_lock = threading.Lock()
|
||||||
|
|
||||||
|
GPU_URLS = {}
|
||||||
|
GPU_SIDECARS = {}
|
||||||
|
GPU_LABELS = {}
|
||||||
|
GPU_MAX_CONCURRENT = {}
|
||||||
|
GPU_CONTEXT = {}
|
||||||
|
TIER_MODELS = {}
|
||||||
|
HOSTS = {}
|
||||||
|
|
||||||
|
def load_roster(path=None):
|
||||||
|
global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT
|
||||||
|
global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime
|
||||||
|
if yaml is None:
|
||||||
|
return False, 'PyYAML not installed - run: pip3 install pyyaml'
|
||||||
|
path = path or ROSTER_PATH
|
||||||
|
try:
|
||||||
|
with open(path) as f:
|
||||||
|
data = yaml.safe_load(f)
|
||||||
|
with _lock:
|
||||||
|
GPU_URLS.clear()
|
||||||
|
GPU_SIDECARS.clear()
|
||||||
|
GPU_LABELS.clear()
|
||||||
|
GPU_MAX_CONCURRENT.clear()
|
||||||
|
GPU_CONTEXT.clear()
|
||||||
|
TIER_MODELS.clear()
|
||||||
|
models = data.get('models', {})
|
||||||
|
for name, cfg in models.items():
|
||||||
|
GPU_URLS[name] = cfg.get('gpu_url', '')
|
||||||
|
GPU_SIDECARS[name] = cfg.get('sidecar_url', '')
|
||||||
|
GPU_LABELS[name] = cfg.get('label', name)
|
||||||
|
GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1)
|
||||||
|
GPU_CONTEXT[name] = cfg.get('context', 65536)
|
||||||
|
tiers = cfg.get('tiers', ['enterprise'])
|
||||||
|
for t in tiers:
|
||||||
|
if t not in TIER_MODELS:
|
||||||
|
TIER_MODELS[t] = []
|
||||||
|
if name not in TIER_MODELS[t]:
|
||||||
|
TIER_MODELS[t].append(name)
|
||||||
|
HOSTS.clear()
|
||||||
|
HOSTS.update(data.get('hosts', {}))
|
||||||
|
_last_mtime = os.path.getmtime(path)
|
||||||
|
return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS))
|
||||||
|
except Exception as e:
|
||||||
|
return False, str(e)
|
||||||
|
|
||||||
|
def check_reload():
|
||||||
|
global _last_mtime
|
||||||
|
try:
|
||||||
|
mtime = os.path.getmtime(ROSTER_PATH)
|
||||||
|
if mtime > _last_mtime:
|
||||||
|
success, msg = load_roster()
|
||||||
|
if success:
|
||||||
|
print('[ROSTER] Auto-reloaded:', msg)
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def reload_thread(interval=30):
|
||||||
|
while True:
|
||||||
|
time.sleep(interval)
|
||||||
|
check_reload()
|
||||||
|
|
||||||
|
threading.Thread(target=reload_thread, daemon=True).start()
|
||||||
Reference in New Issue
Block a user