#!/usr/bin/env python3 """ power-governor.py — Resilient Dynamic GPU Power Governor for CMP 50HX. Tracks llama-server slot state + GPU utilization with fail-safe active retention: - Never downclocks while llama-server is processing a request or busy in CUDA. - Holds full 1900 MHz boost clocks for 45s of silence before entering low-power idle. - Safe 150W power cap enforced on boot. """ import time import subprocess import urllib.request import json import sys CMP_GPUS = "1,2" IDLE_TIMEOUT_SEC = 45 LOW_POWER_CLOCK = 300 def run_cmd(*args): try: subprocess.run(args, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False) except Exception: pass def get_gpu_utilization(): try: out = subprocess.check_output( ["nvidia-smi", f"-i={CMP_GPUS}", "--query-gpu=utilization.gpu", "--format=csv,noheader,nounits"], universal_newlines=True, timeout=2.0 ) utils = [int(line.strip()) for line in out.strip().split("\n") if line.strip().isdigit()] return max(utils) if utils else 0 except Exception: return 0 def is_server_processing(): try: req = urllib.request.Request("http://127.0.0.1:8080/slots", headers={"User-Agent": "power-governor"}) with urllib.request.urlopen(req, timeout=4.0) as r: slots = json.loads(r.read().decode()) return any(bool(s.get("is_processing")) for s in slots) except Exception: # If server is busy or timing out, assume it is ACTIVE to prevent downclocking return True def main(): print("[power-governor] Initializing GPU persistence and 150W power cap...", flush=True) run_cmd("nvidia-smi", "-pm", "1") run_cmd("nvidia-smi", f"-i={CMP_GPUS}", "-pl", "150") run_cmd("nvidia-smi", f"-i={CMP_GPUS}", "-rgc") is_low_power = False last_active = time.time() print("[power-governor] Monitoring llama-server state & GPU compute...", flush=True) while True: try: active = is_server_processing() if not active: # Double-check GPU compute utilization before declaring idle if get_gpu_utilization() > 0: active = True now = time.time() if active: last_active = now if is_low_power: run_cmd("nvidia-smi", f"-i={CMP_GPUS}", "-rgc") is_low_power = False print(f"[{time.strftime('%X')}] Inference Active -> Boost clocks engaged (1900 MHz)", flush=True) else: if not is_low_power and (now - last_active) > IDLE_TIMEOUT_SEC: run_cmd("nvidia-smi", f"-i={CMP_GPUS}", f"-lgc={LOW_POWER_CLOCK},{LOW_POWER_CLOCK}") is_low_power = True print(f"[{time.strftime('%X')}] Inference Idle (> {IDLE_TIMEOUT_SEC}s) -> Low-power state engaged (300 MHz, ~32W/card)", flush=True) time.sleep(0.5 if is_low_power else 1.5) except Exception as e: time.sleep(2.0) if __name__ == "__main__": main()