diff --git a/scripts/gpu-self-heal.py b/scripts/gpu-self-heal.py index 3229ca7..9598463 100755 --- a/scripts/gpu-self-heal.py +++ b/scripts/gpu-self-heal.py @@ -141,6 +141,7 @@ def evaluate_rules(data, history): host_key = k break if not host_key: + print(f\"[warn] no GPU_HOSTS mapping for '{name}' — remote rules (incl. Rule 7) skipped for this GPU\") host_key = name.replace(" ", "-").lower() temp = gpu.get("temp_c", 0) @@ -384,7 +385,7 @@ def run_inference_test(model="syslog-auto"): req = Request("http://192.168.68.116:4000/v1/chat/completions", data=body, headers={ "Content-Type": "application/json", - "Authorization": "Bearer sk-litellm-7f96080dd99b15c36bd4b333b58a6796" + "Authorization": "Bearer " + os.environ.get("LITELLM_API_KEY", "") }) with urlopen(req, timeout=30) as resp: return resp.status == 200