diff --git a/gpu_roster.yaml b/gpu_roster.yaml new file mode 100644 index 0000000..4b76fc4 --- /dev/null +++ b/gpu_roster.yaml @@ -0,0 +1,55 @@ +models: + gemma-4-12b: + gpu_url: http://192.168.68.110:8080/v1 + sidecar_url: http://192.168.68.110:8090 + gpu_host: 192.168.68.110 + label: Gemma-4 12B (RTX 5070) + max_concurrent: 2 + context: 262144 + tiers: [starter, professional, enterprise] + capabilities: [completion, multimodal] + model_path: /home/llmuser/models/gemma-4/gemma-4-12b-it-Q4_K_M.gguf + args: --mmproj /home/llmuser/models/gemma-4/mmproj-F16.gguf --spec-draft-model /home/llmuser/models/gemma-4/gemma-4-12b-it-Q8_0-MTP.gguf --spec-type draft-mtp --spec-draft-n-max 4 --batch-size 2048 --ubatch-size 4096 --n-gpu-layers 99 --flash-attn 1 --image-min-tokens 2048 --image-max-tokens 16384 + + qwen3.6-27B-code: + gpu_url: http://192.168.68.8:8080/v1 + sidecar_url: http://192.168.68.8:8090 + gpu_host: 192.168.68.8 + label: Qwen3.6 27B Code (RTX 3090) + max_concurrent: 2 + context: 262144 + tiers: [professional, enterprise] + capabilities: [completion] + model_path: /root/models/Qwen3.6-27B-NEO-CODE-2T-OT-IQ4_NL.gguf + args: -ctk turbo4 -ctv turbo4 --flash-attn on --reasoning off --spec-type draft-mtp --spec-draft-n-max 2 -t 8 + + ornith-1.0-35b: + gpu_url: http://192.168.68.15:8080/v1 + sidecar_url: http://192.168.68.15:8090 + gpu_host: 192.168.68.15 + label: Ornith-1.0 35B (Strix Halo) + max_concurrent: 1 + context: 4096 + tiers: [professional, enterprise] + capabilities: [completion] + model_path: /data/llamaccp-models/Qwen3.6-35B-A3B-Opus-IQ4_NL.gguf + args: -c 262144 -ngl 99 --flash-attn on + +hosts: + gpu-light: + address: 192.168.68.110 + gpu_name: NVIDIA GeForce RTX 5070 + vram_gb: 12 + current_model: gemma-4-12b + + gpu-dense: + address: 192.168.68.8 + gpu_name: NVIDIA GeForce RTX 3090 + vram_gb: 24 + current_model: qwen3.6-27B-code + + gpu-moe: + address: 192.168.68.15 + gpu_name: AMD Strix Halo (iGPU) + vram_gb: 64 + current_model: ornith-1.0-35b diff --git a/litellm_config.yaml b/litellm_config.yaml index 361b9c6..bcab179 100644 --- a/litellm_config.yaml +++ b/litellm_config.yaml @@ -37,7 +37,7 @@ litellm_settings: qwen3.6-27B-code: input_cost_per_token: 0.0 output_cost_per_token: 0.0 - qwen3.6-35B-A3B: + ornith-1.0-35b: input_cost_per_token: 0.0 output_cost_per_token: 0.0 syslog-auto: @@ -50,40 +50,35 @@ litellm_settings: model_list: - litellm_params: api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 model: openai/syslog-auto rpm: 600 model_name: syslog-auto - litellm_params: api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY - model: openai/qwen3.6-35B-A3B - model_name: qwen3.6-35B-A3B -- litellm_params: - api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 model: openai/qwen3.6-27B-code model_name: qwen3.6-27B-code - litellm_params: api_base: http://router:9000/v1 - api_key: os.environ/ROUTER_API_KEY + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 model: openai/gemma-4-12b model_name: gemma-4-12b + +- litellm_params: + api_base: http://router:9000/v1 + api_key: sk-9e65b69a67-e54af421c1b09fb8bd4f75dacb38cb64 + model: openai/ornith-1.0-35b + model_name: ornith-1.0-35b router_settings: allowed_fails: 100 enable_loadbalancing_on_proxy: false fallbacks: - syslog-auto: - - qwen3.6-35B-A3B - - qwen3.6-27B-code - - gemma-4-12b - - qwen3.6-35B-A3B: - qwen3.6-27B-code - gemma-4-12b - qwen3.6-27B-code: - - qwen3.6-35B-A3B - gemma-4-12b - gemma-4-12b: - qwen3.6-27B-code - - qwen3.6-35B-A3B routing_strategy: usage-based-routing diff --git a/roster_loader.py b/roster_loader.py new file mode 100644 index 0000000..f500e66 --- /dev/null +++ b/roster_loader.py @@ -0,0 +1,75 @@ +# GPU Roster Loader - reads gpu_roster.yaml via PyYAML +# Hot-reloadable via /admin/roster/reload endpoint + +import os, json, threading, time + +try: + import yaml +except ImportError: + yaml = None + +ROSTER_PATH = os.environ.get('ROSTER_PATH', '/app/gpu_roster.yaml') +_last_mtime = 0 +_lock = threading.Lock() + +GPU_URLS = {} +GPU_SIDECARS = {} +GPU_LABELS = {} +GPU_MAX_CONCURRENT = {} +GPU_CONTEXT = {} +TIER_MODELS = {} +HOSTS = {} + +def load_roster(path=None): + global GPU_URLS, GPU_SIDECARS, GPU_LABELS, GPU_MAX_CONCURRENT + global GPU_CONTEXT, TIER_MODELS, HOSTS, _last_mtime + if yaml is None: + return False, 'PyYAML not installed - run: pip3 install pyyaml' + path = path or ROSTER_PATH + try: + with open(path) as f: + data = yaml.safe_load(f) + with _lock: + GPU_URLS.clear() + GPU_SIDECARS.clear() + GPU_LABELS.clear() + GPU_MAX_CONCURRENT.clear() + GPU_CONTEXT.clear() + TIER_MODELS.clear() + models = data.get('models', {}) + for name, cfg in models.items(): + GPU_URLS[name] = cfg.get('gpu_url', '') + GPU_SIDECARS[name] = cfg.get('sidecar_url', '') + GPU_LABELS[name] = cfg.get('label', name) + GPU_MAX_CONCURRENT[name] = cfg.get('max_concurrent', 1) + GPU_CONTEXT[name] = cfg.get('context', 65536) + tiers = cfg.get('tiers', ['enterprise']) + for t in tiers: + if t not in TIER_MODELS: + TIER_MODELS[t] = [] + if name not in TIER_MODELS[t]: + TIER_MODELS[t].append(name) + HOSTS.clear() + HOSTS.update(data.get('hosts', {})) + _last_mtime = os.path.getmtime(path) + return True, 'Loaded {} models, {} hosts'.format(len(GPU_URLS), len(HOSTS)) + except Exception as e: + return False, str(e) + +def check_reload(): + global _last_mtime + try: + mtime = os.path.getmtime(ROSTER_PATH) + if mtime > _last_mtime: + success, msg = load_roster() + if success: + print('[ROSTER] Auto-reloaded:', msg) + except: + pass + +def reload_thread(interval=30): + while True: + time.sleep(interval) + check_reload() + +threading.Thread(target=reload_thread, daemon=True).start()