Reconcile upstream into CT116 production state #1
@@ -0,0 +1,55 @@
|
||||
models:
|
||||
gemma-4-12b:
|
||||
gpu_url: http://192.168.68.110:8080/v1
|
||||
sidecar_url: http://192.168.68.110:8090
|
||||
gpu_host: 192.168.68.110
|
||||
label: Gemma-4 12B (RTX 5070)
|
||||
max_concurrent: 2
|
||||
context: 262144
|
||||
tiers: [starter, professional, enterprise]
|
||||
capabilities: [completion, multimodal]
|
||||
model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf
|
||||
args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384
|
||||
|
||||
qwen3.6-27B-code:
|
||||
gpu_url: http://192.168.68.8:8080/v1
|
||||
sidecar_url: http://192.168.68.8:8090
|
||||
gpu_host: 192.168.68.8
|
||||
label: Qwen3.6 27B Code (RTX 3090)
|
||||
max_concurrent: 2
|
||||
context: 262144
|
||||
tiers: [professional, enterprise]
|
||||
capabilities: [completion]
|
||||
model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf
|
||||
args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8
|
||||
|
||||
ornith-1.0-35b:
|
||||
gpu_url: http://192.168.68.15:8080/v1
|
||||
sidecar_url: http://192.168.68.15:8090
|
||||
gpu_host: 192.168.68.15
|
||||
label: Ornith-1.0 35B (Strix Halo)
|
||||
max_concurrent: 1
|
||||
context: 4096
|
||||
tiers: [professional, enterprise]
|
||||
capabilities: [completion]
|
||||
model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf
|
||||
args: -c 262144 -ngl 99 --flash-attn on
|
||||
|
||||
hosts:
|
||||
gpu-light:
|
||||
address: 192.168.68.110
|
||||
gpu_name: NVIDIA GeForce RTX 5070
|
||||
vram_gb: 12
|
||||
current_model: gemma-4-12b
|
||||
|
||||
gpu-dense:
|
||||
address: 192.168.68.8
|
||||
gpu_name: NVIDIA GeForce RTX 3090
|
||||
vram_gb: 24
|
||||
current_model: qwen3.6-27B-code
|
||||
|
||||
gpu-moe:
|
||||
address: 192.168.68.15
|
||||
gpu_name: AMD Strix Halo (iGPU)
|
||||
vram_gb: 64
|
||||
current_model: ornith-1.0-35b
|
||||
+10
-15
@@ -37,7 +37,7 @@ litellm_settings:
|
||||
qwen3.6-27B-code:
|
||||
input_cost_per_token: 0.0
|
||||
output_cost_per_token: 0.0
|
||||
qwen3.6-35B-A3B:
|
||||
ornith-1.0-35b:
|
||||
input_cost_per_token: 0.0
|
||||
output_cost_per_token: 0.0
|
||||
syslog-auto:
|
||||
@@ -50,40 +50,35 @@ litellm_settings:
|
||||
model_list:
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: os.environ/ROUTER_API_KEY
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
model: openai/syslog-auto
|
||||
rpm: 600
|
||||
model_name: syslog-auto
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: os.environ/ROUTER_API_KEY
|
||||
model: openai/qwen3.6-35B-A3B
|
||||
model_name: qwen3.6-35B-A3B
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: os.environ/ROUTER_API_KEY
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
model: openai/qwen3.6-27B-code
|
||||
model_name: qwen3.6-27B-code
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: os.environ/ROUTER_API_KEY
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
model: openai/gemma-4-12b
|
||||
model_name: gemma-4-12b
|
||||
|
||||
- litellm_params:
|
||||
api_base: http://router:9000/v1
|
||||
api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64
|
||||
model: openai/ornith-1.0-35b
|
||||
model_name: ornith-1.0-35b
|
||||
router_settings:
|
||||
allowed_fails: 100
|
||||
enable_loadbalancing_on_proxy: false
|
||||
fallbacks:
|
||||
- syslog-auto:
|
||||
- qwen3.6-35B-A3B
|
||||
- qwen3.6-27B-code
|
||||
- gemma-4-12b
|
||||
- qwen3.6-35B-A3B:
|
||||
- qwen3.6-27B-code
|
||||
- gemma-4-12b
|
||||
- qwen3.6-27B-code:
|
||||
- qwen3.6-35B-A3B
|
||||
- gemma-4-12b
|
||||
- gemma-4-12b:
|
||||
- qwen3.6-27B-code
|
||||
- qwen3.6-35B-A3B
|
||||
routing_strategy: usage-based-routing
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
# GPU Roster Loader - reads gpu_roster.yaml via PyYAML
|
||||
# Hot-reloadable via /admin/roster/reload endpoint
|
||||
|
||||
import os, json, threading, time
|
||||
|
||||
try:
|
||||
import yaml
|
||||
except ImportError:
|
||||
yaml = None
|
||||
|
||||
ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml')
|
||||
_last_mtime = 0
|
||||
_lock = threading.Lock()
|
||||
|
||||
GPU_URLS = {}
|
||||
GPU_SIDECARS = {}
|
||||
GPU_LABELS = {}
|
||||
GPU_MAX_CONCURRENT = {}
|
||||
GPU_CONTEXT = {}
|
||||
TIER_MODELS = {}
|
||||
HOSTS = {}
|
||||
|
||||
def load_roster(path=None):
|
||||
global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT
|
||||
global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime
|
||||
if yaml is None:
|
||||
return False, 'PyYAML not installed - run: pip3 install pyyaml'
|
||||
path = path or ROSTER_PATH
|
||||
try:
|
||||
with open(path) as f:
|
||||
data = yaml.safe_load(f)
|
||||
with _lock:
|
||||
GPU_URLS.clear()
|
||||
GPU_SIDECARS.clear()
|
||||
GPU_LABELS.clear()
|
||||
GPU_MAX_CONCURRENT.clear()
|
||||
GPU_CONTEXT.clear()
|
||||
TIER_MODELS.clear()
|
||||
models = data.get('models', {})
|
||||
for name, cfg in models.items():
|
||||
GPU_URLS[name] = cfg.get('gpu_url', '')
|
||||
GPU_SIDECARS[name] = cfg.get('sidecar_url', '')
|
||||
GPU_LABELS[name] = cfg.get('label', name)
|
||||
GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1)
|
||||
GPU_CONTEXT[name] = cfg.get('context', 65536)
|
||||
tiers = cfg.get('tiers', ['enterprise'])
|
||||
for t in tiers:
|
||||
if t not in TIER_MODELS:
|
||||
TIER_MODELS[t] = []
|
||||
if name not in TIER_MODELS[t]:
|
||||
TIER_MODELS[t].append(name)
|
||||
HOSTS.clear()
|
||||
HOSTS.update(data.get('hosts', {}))
|
||||
_last_mtime = os.path.getmtime(path)
|
||||
return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS))
|
||||
except Exception as e:
|
||||
return False, str(e)
|
||||
|
||||
def check_reload():
|
||||
global _last_mtime
|
||||
try:
|
||||
mtime = os.path.getmtime(ROSTER_PATH)
|
||||
if mtime > _last_mtime:
|
||||
success, msg = load_roster()
|
||||
if success:
|
||||
print('[ROSTER] Auto-reloaded:', msg)
|
||||
except:
|
||||
pass
|
||||
|
||||
def reload_thread(interval=30):
|
||||
while True:
|
||||
time.sleep(interval)
|
||||
check_reload()
|
||||
|
||||
threading.Thread(target=reload_thread, daemon=True).start()
|
||||
Reference in New Issue
Block a user