feat: experimental NVIDIA power control via RM ioctl interface
Adds an experimental power-cap mode using the undocumented RM ioctl interface (based on panchovix's LACT PR #1205) to set power limits below the VBIOS minimum (down to 30 W). - hal/rm_power.py: RM ioctl power-cap read/write/reset + runtime probe - limits.py: power_cap_mode (nvml/ioctl) with support detection - config.py: persist power_cap_mode per GPU - profiles: record/apply power_cap_mode - server.py: POST /api/limits validates ioctl support (409 on failure) - cli.py: profile save falls back to persisted mode - client.py: power_cap_mode in Limits - frontend: toggle + warning with panchovix attribution (LACT #1205) - tests: test_rm_power.py (unit) + integration coverage - Makefile: add test_rm_power.py to make test Also includes automated linter reformatting (prettier, ruff, shellcheck, isort, markdownlint) that the linter would apply anyway.
This commit is contained in:
1 parent
a8462e696c
commit
e148c83622
21 files changed
+1641
-271
No files matched your search
+56
-6
@@ -569,6 +569,8 @@ class SnapshotRestoreRequest(BaseModel):
|
||||
class LimitsRequest(BaseModel):
|
||||
power_limit_w: int | None = None
|
||||
mem_offset_mhz: int | None = None
|
||||
# "nvml" (default) or "ioctl" (experimental RM power control).
|
||||
power_cap_mode: str | None = None
|
||||
|
||||
|
||||
class ProfileSaveRequest(BaseModel):
|
||||
@@ -914,6 +916,17 @@ def _persist_fan_curves(fan_curves: dict) -> None:
|
||||
_persist_config_field("fan_curves", fan_curves if fan_curves else None)
|
||||
|
||||
|
||||
def _persist_power_cap_modes(modes: dict[str, str]) -> None:
|
||||
"""Persist the per-GPU experimental power-cap mode dict to config.json."""
|
||||
_persist_config_field("power_cap_modes", modes if modes else None)
|
||||
|
||||
|
||||
def _power_cap_mode(cfg: Config, gpu_index: int) -> str:
|
||||
"""Return the effective power-cap mode for a GPU ("nvml" or "ioctl")."""
|
||||
mode = cfg.power_cap_modes.get(_gpu_stable_key(gpu_index), "nvml")
|
||||
return mode if mode in ("nvml", "ioctl") else "nvml"
|
||||
|
||||
|
||||
@app.get("/api/profiles")
|
||||
async def api_profiles(gpu_index: int = 0):
|
||||
"""List saved native profiles, the active profile name, and the auto-load profile name."""
|
||||
@@ -941,13 +954,15 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
|
||||
curve_deltas = {str(p.index): p.delta_khz for p in state.points if p.delta_khz != 0}
|
||||
|
||||
try:
|
||||
power_info = await _run(get_power_limit, gpu_index)
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
power_info = await _run(get_power_limit, gpu_index, mode)
|
||||
offsets = await _run(get_clock_offsets, gpu_index)
|
||||
power_limit_w = power_info.get("power_limit_w")
|
||||
mem_offset_mhz = offsets.get("mem_offset_mhz")
|
||||
except Exception:
|
||||
power_limit_w = None
|
||||
mem_offset_mhz = None
|
||||
mode = "nvml"
|
||||
|
||||
data = ProfileData(
|
||||
name=req.name,
|
||||
@@ -955,6 +970,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
|
||||
curve_deltas=curve_deltas,
|
||||
mem_offset_mhz=mem_offset_mhz,
|
||||
power_limit_w=power_limit_w,
|
||||
power_cap_mode=mode,
|
||||
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
|
||||
fan_targets=g_state.get("fan_targets")
|
||||
if g_state.get("fan_curve_active")
|
||||
@@ -1075,7 +1091,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
errs.append(f"Mem offset: {msg}")
|
||||
|
||||
if profile.power_limit_w is not None:
|
||||
ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index)
|
||||
mode = profile.power_cap_mode or "nvml"
|
||||
ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power limit: {msg}")
|
||||
|
||||
@@ -1204,7 +1221,9 @@ async def api_config_update(req: ConfigUpdateRequest):
|
||||
@app.get("/api/limits")
|
||||
async def api_limits(gpu_index: int = 0):
|
||||
"""Current performance limits: power and clock offsets."""
|
||||
power = await _run(get_power_limit, gpu_index)
|
||||
cfg: Config = _state["config"]
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
power = await _run(get_power_limit, gpu_index, mode)
|
||||
offsets = await _run(get_clock_offsets, gpu_index)
|
||||
mem_off_range = await _run(get_mem_offset_range, gpu_index)
|
||||
return {
|
||||
@@ -1218,10 +1237,36 @@ async def api_limits(gpu_index: int = 0):
|
||||
async def api_limits_update(req: LimitsRequest, gpu_index: int = 0):
|
||||
"""Update performance limits."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
cfg: Config = _state["config"]
|
||||
errs = []
|
||||
|
||||
if req.power_cap_mode is not None:
|
||||
if req.power_cap_mode not in ("nvml", "ioctl"):
|
||||
raise HTTPException(
|
||||
status_code=400, detail="power_cap_mode must be 'nvml' or 'ioctl'"
|
||||
)
|
||||
if req.power_cap_mode == "ioctl":
|
||||
# Verify the GPU actually exposes the RM interface before enabling,
|
||||
# so a client can't lock a GPU into a mode where every power
|
||||
# operation fails (ioctl mode has no NVML fallback by design).
|
||||
info = await _run(get_power_limit, gpu_index, "ioctl")
|
||||
if not info.get("rm_power_supported"):
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail="Experimental RM power control is not supported "
|
||||
"on this GPU/driver",
|
||||
)
|
||||
key = _gpu_stable_key(gpu_index)
|
||||
if req.power_cap_mode == "nvml":
|
||||
cfg.power_cap_modes.pop(key, None)
|
||||
else:
|
||||
cfg.power_cap_modes[key] = "ioctl"
|
||||
_persist_power_cap_modes(cfg.power_cap_modes)
|
||||
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
|
||||
if req.power_limit_w is not None:
|
||||
ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index)
|
||||
ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power Limit: {msg}")
|
||||
|
||||
@@ -1284,12 +1329,17 @@ async def _update_offsets_and_broadcast(gpu_index: int) -> None:
|
||||
async def api_limits_reset(gpu_index: int = 0):
|
||||
"""Reset power limit to hardware default and memory clock offset to 0."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
cfg: Config = _state["config"]
|
||||
errs = []
|
||||
|
||||
power = await _run(get_power_limit, gpu_index)
|
||||
# Reset uses the GPU's current mode: in ioctl mode the default is
|
||||
# restored through the RM route (which can also restore a previous
|
||||
# below-VBIOS-minimum cap).
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
power = await _run(get_power_limit, gpu_index, mode)
|
||||
default_w = power.get("default_power_limit_w")
|
||||
if default_w is not None:
|
||||
ok, msg = await _run(set_power_limit, default_w, gpu_index)
|
||||
ok, msg = await _run(set_power_limit, default_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power Limit: {msg}")
|
||||
|
||||
|
||||
Reference in new issue
Block a user