Fix power governor active retention during long prompt evaluations and remove redundant proxy governor thread
This commit is contained in:
@@ -26,54 +26,6 @@ import threading
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# GPU Power & Clock Governor (Automatic idle power reduction for CMP 50HX)
|
||||
# ----------------------------------------------------------------------------
|
||||
class GPUGovernor:
|
||||
"""Manages GPU clocks dynamically: drops CMP cards to low-power state when idle (saves ~90W),
|
||||
and instantly unconstrains to full boost clocks during inference."""
|
||||
def __init__(self, idle_timeout=30):
|
||||
self.idle_timeout = idle_timeout
|
||||
self.last_active = time.time()
|
||||
self.is_low_power = False
|
||||
self.lock = threading.Lock()
|
||||
self._init_limits()
|
||||
t = threading.Thread(target=self._monitor_loop, daemon=True)
|
||||
t.start()
|
||||
|
||||
def _run_cmd(self, *args):
|
||||
try:
|
||||
subprocess.run(args, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _init_limits(self):
|
||||
# Ensure persistence mode and 150W safe power cap
|
||||
self._run_cmd("nvidia-smi", "-pm", "1")
|
||||
self._run_cmd("nvidia-smi", "-i", "1,2", "-pl", "150")
|
||||
|
||||
def wake(self):
|
||||
with self.lock:
|
||||
self.last_active = time.time()
|
||||
if self.is_low_power:
|
||||
self._run_cmd("nvidia-smi", "-rgc")
|
||||
self.is_low_power = False
|
||||
|
||||
def touch(self):
|
||||
with self.lock:
|
||||
self.last_active = time.time()
|
||||
|
||||
def _monitor_loop(self):
|
||||
while True:
|
||||
time.sleep(5)
|
||||
with self.lock:
|
||||
if not self.is_low_power and (time.time() - self.last_active) > self.idle_timeout:
|
||||
# Drop CMP cards to low power idle clock
|
||||
self._run_cmd("nvidia-smi", "-i", "1,2", "-lgc", "600,600")
|
||||
self.is_low_power = True
|
||||
|
||||
GOVERNOR = GPUGovernor(idle_timeout=30)
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Config
|
||||
# ----------------------------------------------------------------------------
|
||||
@@ -125,7 +77,6 @@ def est_tokens(text):
|
||||
return max(1, (len(text) + 3) // 4)
|
||||
|
||||
def _post(path, payload, stream=False):
|
||||
GOVERNOR.wake()
|
||||
req = urllib.request.Request(
|
||||
BACKEND + path,
|
||||
data=json.dumps(payload).encode(),
|
||||
@@ -153,7 +104,6 @@ def _backend_stream(messages, **kw):
|
||||
body.update(kw)
|
||||
with _post("/chat/completions", body) as r:
|
||||
for raw in r:
|
||||
GOVERNOR.touch()
|
||||
line = raw.decode(errors="replace").strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user