feat: experimental NVIDIA power control via RM ioctl interface

Adds an experimental power-cap mode using the undocumented RM ioctl
interface (based on panchovix's LACT PR #1205) to set power limits
below the VBIOS minimum (down to 30 W).

- hal/rm_power.py: RM ioctl power-cap read/write/reset + runtime probe
- limits.py: power_cap_mode (nvml/ioctl) with support detection
- config.py: persist power_cap_mode per GPU
- profiles: record/apply power_cap_mode
- server.py: POST /api/limits validates ioctl support (409 on failure)
- cli.py: profile save falls back to persisted mode
- client.py: power_cap_mode in Limits
- frontend: toggle + warning with panchovix attribution (LACT #1205)
- tests: test_rm_power.py (unit) + integration coverage
- Makefile: add test_rm_power.py to make test

Also includes automated linter reformatting (prettier, ruff, shellcheck,
isort, markdownlint) that the linter would apply anyway.
This commit is contained in:
ARIA committed 2026-09-17 22:44:27 +02:00
1 parent a8462e696c
commit e148c83622
21 files changed
+1641 -271

No files matched your search

+56 -6
View File
@@ -569,6 +569,8 @@ class SnapshotRestoreRequest(BaseModel):
class LimitsRequest(BaseModel):
power_limit_w: int | None = None
mem_offset_mhz: int | None = None
# "nvml" (default) or "ioctl" (experimental RM power control).
power_cap_mode: str | None = None
class ProfileSaveRequest(BaseModel):
@@ -914,6 +916,17 @@ def _persist_fan_curves(fan_curves: dict) -> None:
_persist_config_field("fan_curves", fan_curves if fan_curves else None)
def _persist_power_cap_modes(modes: dict[str, str]) -> None:
"""Persist the per-GPU experimental power-cap mode dict to config.json."""
_persist_config_field("power_cap_modes", modes if modes else None)
def _power_cap_mode(cfg: Config, gpu_index: int) -> str:
"""Return the effective power-cap mode for a GPU ("nvml" or "ioctl")."""
mode = cfg.power_cap_modes.get(_gpu_stable_key(gpu_index), "nvml")
return mode if mode in ("nvml", "ioctl") else "nvml"
@app.get("/api/profiles")
async def api_profiles(gpu_index: int = 0):
"""List saved native profiles, the active profile name, and the auto-load profile name."""
@@ -941,13 +954,15 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
curve_deltas = {str(p.index): p.delta_khz for p in state.points if p.delta_khz != 0}
try:
power_info = await _run(get_power_limit, gpu_index)
mode = _power_cap_mode(cfg, gpu_index)
power_info = await _run(get_power_limit, gpu_index, mode)
offsets = await _run(get_clock_offsets, gpu_index)
power_limit_w = power_info.get("power_limit_w")
mem_offset_mhz = offsets.get("mem_offset_mhz")
except Exception:
power_limit_w = None
mem_offset_mhz = None
mode = "nvml"
data = ProfileData(
name=req.name,
@@ -955,6 +970,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
curve_deltas=curve_deltas,
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
power_cap_mode=mode,
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
fan_targets=g_state.get("fan_targets")
if g_state.get("fan_curve_active")
@@ -1075,7 +1091,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
errs.append(f"Mem offset: {msg}")
if profile.power_limit_w is not None:
ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index)
mode = profile.power_cap_mode or "nvml"
ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index, mode)
if not ok:
errs.append(f"Power limit: {msg}")
@@ -1204,7 +1221,9 @@ async def api_config_update(req: ConfigUpdateRequest):
@app.get("/api/limits")
async def api_limits(gpu_index: int = 0):
"""Current performance limits: power and clock offsets."""
power = await _run(get_power_limit, gpu_index)
cfg: Config = _state["config"]
mode = _power_cap_mode(cfg, gpu_index)
power = await _run(get_power_limit, gpu_index, mode)
offsets = await _run(get_clock_offsets, gpu_index)
mem_off_range = await _run(get_mem_offset_range, gpu_index)
return {
@@ -1218,10 +1237,36 @@ async def api_limits(gpu_index: int = 0):
async def api_limits_update(req: LimitsRequest, gpu_index: int = 0):
"""Update performance limits."""
g_state = _get_gpu_state(gpu_index)
cfg: Config = _state["config"]
errs = []
if req.power_cap_mode is not None:
if req.power_cap_mode not in ("nvml", "ioctl"):
raise HTTPException(
status_code=400, detail="power_cap_mode must be 'nvml' or 'ioctl'"
)
if req.power_cap_mode == "ioctl":
# Verify the GPU actually exposes the RM interface before enabling,
# so a client can't lock a GPU into a mode where every power
# operation fails (ioctl mode has no NVML fallback by design).
info = await _run(get_power_limit, gpu_index, "ioctl")
if not info.get("rm_power_supported"):
raise HTTPException(
status_code=409,
detail="Experimental RM power control is not supported "
"on this GPU/driver",
)
key = _gpu_stable_key(gpu_index)
if req.power_cap_mode == "nvml":
cfg.power_cap_modes.pop(key, None)
else:
cfg.power_cap_modes[key] = "ioctl"
_persist_power_cap_modes(cfg.power_cap_modes)
mode = _power_cap_mode(cfg, gpu_index)
if req.power_limit_w is not None:
ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index)
ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index, mode)
if not ok:
errs.append(f"Power Limit: {msg}")
@@ -1284,12 +1329,17 @@ async def _update_offsets_and_broadcast(gpu_index: int) -> None:
async def api_limits_reset(gpu_index: int = 0):
"""Reset power limit to hardware default and memory clock offset to 0."""
g_state = _get_gpu_state(gpu_index)
cfg: Config = _state["config"]
errs = []
power = await _run(get_power_limit, gpu_index)
# Reset uses the GPU's current mode: in ioctl mode the default is
# restored through the RM route (which can also restore a previous
# below-VBIOS-minimum cap).
mode = _power_cap_mode(cfg, gpu_index)
power = await _run(get_power_limit, gpu_index, mode)
default_w = power.get("default_power_limit_w")
if default_w is not None:
ok, msg = await _run(set_power_limit, default_w, gpu_index)
ok, msg = await _run(set_power_limit, default_w, gpu_index, mode)
if not ok:
errs.append(f"Power Limit: {msg}")