Fix power governor active retention during long prompt evaluations and remove redundant proxy governor thread

This commit is contained in:
wmantly
2026-09-01 14:40:19 +00:00
parent aeb72cecfa
commit c30f4772f9
2 changed files with 30 additions and 61 deletions
-50
View File
@@ -26,54 +26,6 @@ import threading
import re
import subprocess
# ----------------------------------------------------------------------------
# GPU Power & Clock Governor (Automatic idle power reduction for CMP 50HX)
# ----------------------------------------------------------------------------
class GPUGovernor:
"""Manages GPU clocks dynamically: drops CMP cards to low-power state when idle (saves ~90W),
and instantly unconstrains to full boost clocks during inference."""
def __init__(self, idle_timeout=30):
self.idle_timeout = idle_timeout
self.last_active = time.time()
self.is_low_power = False
self.lock = threading.Lock()
self._init_limits()
t = threading.Thread(target=self._monitor_loop, daemon=True)
t.start()
def _run_cmd(self, *args):
try:
subprocess.run(args, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False)
except Exception:
pass
def _init_limits(self):
# Ensure persistence mode and 150W safe power cap
self._run_cmd("nvidia-smi", "-pm", "1")
self._run_cmd("nvidia-smi", "-i", "1,2", "-pl", "150")
def wake(self):
with self.lock:
self.last_active = time.time()
if self.is_low_power:
self._run_cmd("nvidia-smi", "-rgc")
self.is_low_power = False
def touch(self):
with self.lock:
self.last_active = time.time()
def _monitor_loop(self):
while True:
time.sleep(5)
with self.lock:
if not self.is_low_power and (time.time() - self.last_active) > self.idle_timeout:
# Drop CMP cards to low power idle clock
self._run_cmd("nvidia-smi", "-i", "1,2", "-lgc", "600,600")
self.is_low_power = True
GOVERNOR = GPUGovernor(idle_timeout=30)
# ----------------------------------------------------------------------------
# Config
# ----------------------------------------------------------------------------
@@ -125,7 +77,6 @@ def est_tokens(text):
return max(1, (len(text) + 3) // 4)
def _post(path, payload, stream=False):
GOVERNOR.wake()
req = urllib.request.Request(
BACKEND + path,
data=json.dumps(payload).encode(),
@@ -153,7 +104,6 @@ def _backend_stream(messages, **kw):
body.update(kw)
with _post("/chat/completions", body) as r:
for raw in r:
GOVERNOR.touch()
line = raw.decode(errors="replace").strip()
if not line:
continue