91 lines
3.6 KiB
Python
Executable File
91 lines
3.6 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
power-governor.py — Resilient Dynamic GPU Power Governor for CMP 50HX.
|
|
|
|
Tracks llama-server slot state + GPU utilization with fail-safe active retention:
|
|
- Never downclocks while llama-server is processing a request or busy in CUDA.
|
|
- Holds full 1900 MHz boost clocks for 45s of silence before entering low-power idle.
|
|
- Safe 150W power cap enforced on boot.
|
|
"""
|
|
|
|
import time
|
|
import subprocess
|
|
import urllib.request
|
|
import json
|
|
import sys
|
|
|
|
CMP_GPUS = "1,2"
|
|
IDLE_TIMEOUT_SEC = 45
|
|
LOW_POWER_CLOCK = 300
|
|
|
|
def run_cmd(*args):
|
|
try:
|
|
subprocess.run(args, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False)
|
|
except Exception:
|
|
pass
|
|
|
|
def get_gpu_utilization():
|
|
try:
|
|
out = subprocess.check_output(
|
|
["nvidia-smi", f"-i={CMP_GPUS}", "--query-gpu=utilization.gpu", "--format=csv,noheader,nounits"],
|
|
universal_newlines=True,
|
|
timeout=2.0
|
|
)
|
|
utils = [int(line.strip()) for line in out.strip().split("\n") if line.strip().isdigit()]
|
|
return max(utils) if utils else 0
|
|
except Exception:
|
|
return 0
|
|
|
|
def is_server_processing():
|
|
try:
|
|
req = urllib.request.Request("http://127.0.0.1:8080/slots", headers={"User-Agent": "power-governor"})
|
|
with urllib.request.urlopen(req, timeout=4.0) as r:
|
|
slots = json.loads(r.read().decode())
|
|
return any(bool(s.get("is_processing")) for s in slots)
|
|
except Exception:
|
|
# If server is busy or timing out, assume it is ACTIVE to prevent downclocking
|
|
return True
|
|
|
|
def main():
|
|
print("[power-governor] Initializing GPU persistence and 150W power cap...", flush=True)
|
|
run_cmd("nvidia-smi", "-pm", "1")
|
|
run_cmd("nvidia-smi", f"-i={CMP_GPUS}", "-pl", "150")
|
|
run_cmd("nvidia-smi", f"-i={CMP_GPUS}", "-rgc")
|
|
|
|
is_low_power = False
|
|
last_active = time.time()
|
|
|
|
print("[power-governor] Monitoring llama-server state & GPU compute...", flush=True)
|
|
while True:
|
|
try:
|
|
active = is_server_processing()
|
|
if not active:
|
|
# Double-check GPU compute utilization before declaring idle
|
|
if get_gpu_utilization() > 0:
|
|
active = True
|
|
|
|
now = time.time()
|
|
|
|
if active:
|
|
last_active = now
|
|
if is_low_power:
|
|
run_cmd("python3", "/opt/turing-multi-gpu-llm-server/tools/cmp-pstate.py", "--bus-id", "00000000:07:00.0", "--pstate", "16")
|
|
run_cmd("python3", "/opt/turing-multi-gpu-llm-server/tools/cmp-pstate.py", "--bus-id", "00000000:21:00.0", "--pstate", "16")
|
|
run_cmd("nvidia-smi", f"-i={CMP_GPUS}", "-rgc")
|
|
is_low_power = False
|
|
print(f"[{time.strftime('%X')}] Inference Active -> Boost clocks engaged (P0/P16, 1900 MHz)", flush=True)
|
|
else:
|
|
if not is_low_power and (now - last_active) > IDLE_TIMEOUT_SEC:
|
|
run_cmd("python3", "/opt/turing-multi-gpu-llm-server/tools/cmp-pstate.py", "--bus-id", "00000000:07:00.0", "--pstate", "8")
|
|
run_cmd("python3", "/opt/turing-multi-gpu-llm-server/tools/cmp-pstate.py", "--bus-id", "00000000:21:00.0", "--pstate", "8")
|
|
run_cmd("nvidia-smi", f"-i={CMP_GPUS}", f"-lgc={LOW_POWER_CLOCK},{LOW_POWER_CLOCK}")
|
|
is_low_power = True
|
|
print(f"[{time.strftime('%X')}] Inference Idle (> {IDLE_TIMEOUT_SEC}s) -> Deep low-power state engaged (NVAPI P8 + 300 MHz)", flush=True)
|
|
|
|
time.sleep(0.5 if is_low_power else 1.5)
|
|
except Exception as e:
|
|
time.sleep(2.0)
|
|
|
|
if __name__ == "__main__":
|
|
main()
|