feat: experimental NVIDIA power control via RM ioctl interface
Adds an experimental power-cap mode using the undocumented RM ioctl interface (based on panchovix's LACT PR #1205) to set power limits below the VBIOS minimum (down to 30 W). - hal/rm_power.py: RM ioctl power-cap read/write/reset + runtime probe - limits.py: power_cap_mode (nvml/ioctl) with support detection - config.py: persist power_cap_mode per GPU - profiles: record/apply power_cap_mode - server.py: POST /api/limits validates ioctl support (409 on failure) - cli.py: profile save falls back to persisted mode - client.py: power_cap_mode in Limits - frontend: toggle + warning with panchovix attribution (LACT #1205) - tests: test_rm_power.py (unit) + integration coverage - Makefile: add test_rm_power.py to make test Also includes automated linter reformatting (prettier, ruff, shellcheck, isort, markdownlint) that the linter would apply anyway.
This commit is contained in:
1 parent
a8462e696c
commit
e148c83622
21 files changed
+1641
-271
No files matched your search
+36
-1
@@ -405,6 +405,11 @@ def run_diagnostics(gpu, gpu_name, gpu_index: int = 0):
|
||||
print(f" Default: {fmt_w(def_w)}")
|
||||
if min_w is not None and max_w is not None:
|
||||
print(f" Range: {min_w} – {max_w} W")
|
||||
if pwr.get("rm_power_supported"):
|
||||
print(
|
||||
" Experimental RM power: available (opt-in via web UI or profile;"
|
||||
" extends range to 30 W)"
|
||||
)
|
||||
|
||||
|
||||
# ── Privilege / browser helpers ───────────────────────────────────────────────
|
||||
@@ -1207,12 +1212,33 @@ def cmd_profile(args):
|
||||
power_limit_w = None
|
||||
mem_offset_mhz = None
|
||||
|
||||
# Capture the GPU's power-cap mode: prefer the running server (most
|
||||
# current), else fall back to the persisted per-GPU mode from config
|
||||
# (so a profile saved while the server is down or auth is enabled
|
||||
# still records the GPU's actual mode rather than assuming nvml).
|
||||
power_cap_mode = "nvml"
|
||||
try:
|
||||
from .client import NvCurveClient
|
||||
|
||||
base = getattr(args, "server", None) or _discover_server_url(default_config)
|
||||
limits = NvCurveClient(base=base, gpu_index=gpu_index).limits()
|
||||
if limits.get("power_cap_mode") in ("nvml", "ioctl"):
|
||||
power_cap_mode = limits["power_cap_mode"]
|
||||
except Exception as exc:
|
||||
log.debug("Could not read power-cap mode from server: %s", exc)
|
||||
gpu_key = _gpu_stable_key_offline(gpu_index)
|
||||
if gpu_key is not None:
|
||||
persisted = default_config.power_cap_modes.get(gpu_key)
|
||||
if persisted in ("nvml", "ioctl"):
|
||||
power_cap_mode = persisted
|
||||
|
||||
data = ProfileData(
|
||||
name=args.name,
|
||||
gpu_name=gpu_name,
|
||||
curve_deltas=curve_deltas,
|
||||
mem_offset_mhz=mem_offset_mhz,
|
||||
power_limit_w=power_limit_w,
|
||||
power_cap_mode=power_cap_mode,
|
||||
)
|
||||
filepath = save_profile(default_config.profile_dir, data)
|
||||
print(f"Saved profile '{args.name}' to {filepath}")
|
||||
@@ -1249,7 +1275,8 @@ def cmd_profile(args):
|
||||
errs.append(f"Mem offset: {msg}")
|
||||
|
||||
if profile.power_limit_w is not None:
|
||||
ok, msg = set_power_limit(profile.power_limit_w, gpu_index)
|
||||
mode = profile.power_cap_mode or "nvml"
|
||||
ok, msg = set_power_limit(profile.power_limit_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power limit: {msg}")
|
||||
|
||||
@@ -2296,6 +2323,14 @@ def main():
|
||||
if "fan_curves" in data:
|
||||
# Per-GPU active fan curves, restored on server startup.
|
||||
cfg.fan_curves = dict(data["fan_curves"])
|
||||
if "power_cap_modes" in data:
|
||||
# Per-GPU experimental power-cap mode. "nvml" is the default
|
||||
# (the server treats it as unset); keep only valid values.
|
||||
cfg.power_cap_modes = {
|
||||
str(k): str(v)
|
||||
for k, v in dict(data["power_cap_modes"]).items()
|
||||
if str(v) in ("nvml", "ioctl")
|
||||
}
|
||||
except Exception as exc:
|
||||
log.debug("Could not load user config: %s", exc)
|
||||
|
||||
|
||||
@@ -145,6 +145,9 @@ class NvCurveClient:
|
||||
def snapshots(self) -> list:
|
||||
return self._get("/api/snapshots")
|
||||
|
||||
def limits(self) -> dict:
|
||||
return self._get("/api/limits")
|
||||
|
||||
# ── Profiles ─────────────────────────────────────────────────────────────
|
||||
|
||||
def profiles(self) -> dict:
|
||||
|
||||
@@ -54,6 +54,11 @@ class Config:
|
||||
# Legacy entries (bare curve list) are migrated at load time.
|
||||
fan_curves: dict[str, object] = field(default_factory=dict)
|
||||
|
||||
# Per-GPU power-cap mode: "nvml" (default, never stored) or "ioctl"
|
||||
# (experimental RM power control — permits caps below the VBIOS minimum).
|
||||
# Key = stable GPU identifier (same as auto_load_profiles).
|
||||
power_cap_modes: dict[str, str] = field(default_factory=dict)
|
||||
|
||||
|
||||
# Module-level default config instance.
|
||||
default_config = Config()
|
||||
|
||||
+51
-6
@@ -59,20 +59,37 @@ def _get_handle(gpu_index: int):
|
||||
# ── Power limit ───────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def get_power_limit(gpu_index: int = 0) -> dict:
|
||||
"""Return dict with power_limit_w, default_power_limit_w, min_power_limit_w, max_power_limit_w."""
|
||||
out: dict[str, int | None] = {
|
||||
def get_power_limit(gpu_index: int = 0, mode: str = "nvml") -> dict:
|
||||
"""Return dict with power limit info.
|
||||
|
||||
Keys: power_limit_w, default_power_limit_w, min_power_limit_w,
|
||||
min_power_limit_w_native, max_power_limit_w, rm_power_supported,
|
||||
power_cap_mode.
|
||||
|
||||
mode: "nvml" (default) or "ioctl" (experimental RM power control).
|
||||
min_power_limit_w is the effective minimum: in ioctl mode it is
|
||||
extended to the experimental floor (30 W) when the RM interface is
|
||||
present and validated; min_power_limit_w_native is always the VBIOS
|
||||
minimum. The RM probe is GET-only (no writes) and safe to run on
|
||||
every call.
|
||||
"""
|
||||
out: dict[str, int | bool | str | None] = {
|
||||
"power_limit_w": None,
|
||||
"default_power_limit_w": None,
|
||||
"min_power_limit_w": None,
|
||||
"min_power_limit_w_native": None,
|
||||
"max_power_limit_w": None,
|
||||
"rm_power_supported": False,
|
||||
"power_cap_mode": mode,
|
||||
}
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
limit = pynvml.nvmlDeviceGetPowerManagementLimit(handle)
|
||||
constrs = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle)
|
||||
out["power_limit_w"] = limit // 1000
|
||||
out["min_power_limit_w"] = constrs[0] // 1000
|
||||
native_min = constrs[0] // 1000
|
||||
out["min_power_limit_w"] = native_min
|
||||
out["min_power_limit_w_native"] = native_min
|
||||
out["max_power_limit_w"] = constrs[1] // 1000
|
||||
try:
|
||||
default = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
|
||||
@@ -81,11 +98,39 @@ def get_power_limit(gpu_index: int = 0) -> dict:
|
||||
log.debug("nvmlDeviceGetPowerManagementDefaultLimit: %s", exc)
|
||||
except Exception as exc:
|
||||
log.warning("get_power_limit: %s", exc)
|
||||
return out
|
||||
|
||||
# GET-only RM discovery — reported so the UI can offer the experimental
|
||||
# mode; the effective minimum only changes in ioctl mode.
|
||||
try:
|
||||
from . import rm_power
|
||||
|
||||
bounds = rm_power.probe_gpu(gpu_index)
|
||||
out["rm_power_supported"] = bounds is not None
|
||||
if bounds is not None and mode == "ioctl":
|
||||
out["min_power_limit_w"] = bounds.lower_min_mw() // 1000
|
||||
except Exception as exc:
|
||||
log.debug("RM power probe failed: %s", exc)
|
||||
return out
|
||||
|
||||
|
||||
def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
|
||||
"""Set the board power limit (Watts)."""
|
||||
def set_power_limit(
|
||||
limit_w: int, gpu_index: int = 0, mode: str = "nvml"
|
||||
) -> tuple[bool, str]:
|
||||
"""Set the board power limit (Watts).
|
||||
|
||||
mode "ioctl" (experimental) applies the limit through the undocumented
|
||||
RM interface, which permits values below the VBIOS minimum. It has no
|
||||
fallback: failures are reported, never silently switched to NVML.
|
||||
"""
|
||||
if mode == "ioctl":
|
||||
from . import rm_power
|
||||
|
||||
try:
|
||||
rm_power.set_power_limit_w(gpu_index, limit_w)
|
||||
return True, "OK"
|
||||
except rm_power.RmPowerError as exc:
|
||||
return False, str(exc)
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
pynvml.nvmlDeviceSetPowerManagementLimit(handle, limit_w * 1000)
|
||||
|
||||
@@ -0,0 +1,627 @@
|
||||
"""Undocumented NVIDIA RM power-limit interface (EXPERIMENTAL).
|
||||
|
||||
Port of the approach from LACT PR #1205 (ilya-zlobintsev/LACT): applies board
|
||||
power limits through the private NV2080 power-limit "ordinary client"
|
||||
interface on /dev/nvidiactl, which permits caps below the VBIOS minimum
|
||||
(down to 30 W). The native maximum still applies.
|
||||
|
||||
EXPERIMENTAL — uses an undocumented driver interface. It may break after
|
||||
driver updates. Discovery is GET-only and validates the RM payload against
|
||||
NVML before any write is issued; a failed write restores the previous
|
||||
request (even if it was below the VBIOS minimum).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import ctypes
|
||||
import fcntl
|
||||
import logging
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.rm_power")
|
||||
|
||||
# ── ioctl constants (nv-ioctl.h / nv-ioctl-numbers.h) ─────────────────────────
|
||||
|
||||
NV_IOCTL_MAGIC = ord("N") # 0x4E — user-space RM interface
|
||||
NV_ESC_RM_ALLOC = 0x2B
|
||||
NV_ESC_RM_CONTROL = 0x2A
|
||||
|
||||
# 'F' magic interface (kernel-open/common/inc/nv-ioctl-numbers.h) —
|
||||
# NV_ESC_REGISTER_FD lives here, not in the 'N' RM interface.
|
||||
NV_IOCTL_MAGIC_F = ord("F") # 0x46
|
||||
NV_IOCTL_BASE_F = 200
|
||||
NV_ESC_REGISTER_FD = NV_IOCTL_BASE_F + 1 # 201
|
||||
|
||||
# RM class IDs (nv0080.h / nv2080.h)
|
||||
NV01_DEVICE_0 = 0x0080
|
||||
NV20_SUBDEVICE_0 = 0x2080
|
||||
|
||||
# NV01_ROOT GPU queries (ctrl0000gpu.h) — resolve PCI identity to the RM
|
||||
# device/subdevice instance numbers used by NV0080 and NV2080 allocations;
|
||||
# neither number is a Linux device minor.
|
||||
_CTRL_GPU_GET_ATTACHED_IDS = 0x201
|
||||
_CTRL_GPU_GET_ID_INFO_V2 = 0x205
|
||||
_CTRL_GPU_GET_PCI_INFO = 0x21B
|
||||
_MAX_GPUS = 32
|
||||
_INVALID_GPU_ID = 0xFFFFFFFF
|
||||
|
||||
# Private NV2080 power-limit client commands. Payloads compared against
|
||||
# NvAPI and GSP from R595, R610 and R615 (native RM payloads, without
|
||||
# NvAPI's 0x10-byte transport prefix).
|
||||
_PWR_GET_INFO = 0x2080_A630
|
||||
_PWR_GET_CONTROL = 0x2080_A632
|
||||
_PWR_SET_CONTROL = 0x2080_E633
|
||||
_ORDINARY_CLIENT = 0xFE
|
||||
_LOWER_LIMIT_MW = 30_000 # experimental floor: 30 W
|
||||
|
||||
|
||||
def _ioctl_rw(size: int, nr: int, magic: int = NV_IOCTL_MAGIC) -> int:
|
||||
"""Linux ioctl request code: dir=RW, given size/type/nr."""
|
||||
return (2 << 30) | (size << 16) | (magic << 8) | nr
|
||||
|
||||
|
||||
def _ioctl_call(fd: int, code: int, arg) -> None:
|
||||
"""Issue an ioctl, converting errno failures to RmPowerError.
|
||||
|
||||
The driver normally reports failures as an RM status in the parameter
|
||||
struct, but an experimental interface can also fail at the kernel level
|
||||
(ENOTTY/EBADF/EPERM across driver versions). Converting to RmPowerError
|
||||
keeps the module's error contract uniform and lets callers clean up fds.
|
||||
"""
|
||||
try:
|
||||
fcntl.ioctl(fd, code, arg)
|
||||
except OSError as exc:
|
||||
raise RmPowerError(f"ioctl 0x{code:x} failed: {exc}") from exc
|
||||
|
||||
|
||||
# ── NVOS parameter structs (nvos.h) ──────────────────────────────────────────
|
||||
|
||||
|
||||
class _NVOS21(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("hRoot", ctypes.c_uint32),
|
||||
("hObjectParent", ctypes.c_uint32),
|
||||
("hObjectNew", ctypes.c_uint32),
|
||||
("hClass", ctypes.c_uint32),
|
||||
("pAllocParms", ctypes.c_uint64),
|
||||
("paramsSize", ctypes.c_uint32),
|
||||
("status", ctypes.c_uint32),
|
||||
]
|
||||
|
||||
|
||||
class _NVOS64(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("hRoot", ctypes.c_uint32),
|
||||
("hObjectParent", ctypes.c_uint32),
|
||||
("hObjectNew", ctypes.c_uint32),
|
||||
("hClass", ctypes.c_uint32),
|
||||
("pAllocParms", ctypes.c_uint64),
|
||||
("pRightsRequested", ctypes.c_uint64),
|
||||
("paramsSize", ctypes.c_uint32),
|
||||
("flags", ctypes.c_uint32),
|
||||
("status", ctypes.c_uint32),
|
||||
]
|
||||
|
||||
|
||||
class _NVOS54(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("hClient", ctypes.c_uint32),
|
||||
("hObject", ctypes.c_uint32),
|
||||
("cmd", ctypes.c_uint32),
|
||||
("flags", ctypes.c_uint32),
|
||||
("params", ctypes.c_uint64),
|
||||
("paramsSize", ctypes.c_uint32),
|
||||
("status", ctypes.c_uint32),
|
||||
]
|
||||
|
||||
|
||||
class _NV0080_ALLOC(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("deviceId", ctypes.c_uint32),
|
||||
("deviceFlags", ctypes.c_uint32),
|
||||
("vgpuInstance", ctypes.c_uint32),
|
||||
("pad", ctypes.c_uint32),
|
||||
]
|
||||
|
||||
|
||||
class _NV2080_ALLOC(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("subDeviceId", ctypes.c_uint32),
|
||||
("clientShare", ctypes.c_uint32),
|
||||
("flags", ctypes.c_uint32),
|
||||
("pad", ctypes.c_uint32),
|
||||
]
|
||||
|
||||
|
||||
# ── Errors ───────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class RmPowerError(RuntimeError):
|
||||
"""Raised when the RM power-limit interface is unavailable or fails."""
|
||||
|
||||
|
||||
# ── Power-limit layouts and bounds ───────────────────────────────────────────
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PowerLimitLayout:
|
||||
"""Byte offsets of the private power-limit payloads for one wire format."""
|
||||
|
||||
name: str
|
||||
info_size: int
|
||||
control_size: int
|
||||
info_min_at: int
|
||||
request_at: int
|
||||
client_at: int
|
||||
mask_end: int
|
||||
|
||||
|
||||
EXTENDED_LAYOUT = PowerLimitLayout(
|
||||
name="extended",
|
||||
info_size=0x924,
|
||||
control_size=0x328,
|
||||
info_min_at=0x28,
|
||||
request_at=0x2C,
|
||||
client_at=0x30,
|
||||
mask_end=0x24,
|
||||
)
|
||||
LEGACY_LAYOUT = PowerLimitLayout(
|
||||
name="legacy",
|
||||
info_size=0x488,
|
||||
control_size=0x188,
|
||||
info_min_at=0xC,
|
||||
request_at=0xC,
|
||||
client_at=0x10,
|
||||
mask_end=0x8,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PowerLimitBounds:
|
||||
"""Power limit bounds in milliwatts (NVML/RM units)."""
|
||||
|
||||
min_mw: int
|
||||
default_mw: int
|
||||
max_mw: int
|
||||
|
||||
def lower_min_mw(self) -> int:
|
||||
"""Effective minimum when the experimental route is active."""
|
||||
return min(self.min_mw, _LOWER_LIMIT_MW)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LowerPowerLimit:
|
||||
"""A validated RM power-limit layout that can be written."""
|
||||
|
||||
bounds: PowerLimitBounds
|
||||
layout: PowerLimitLayout
|
||||
|
||||
def lower_min_mw(self) -> int:
|
||||
return self.bounds.lower_min_mw()
|
||||
|
||||
|
||||
# ── PCI identity → RM instance resolution ────────────────────────────────────
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PciLocation:
|
||||
domain: int
|
||||
bus: int
|
||||
dev: int
|
||||
func: int = 0
|
||||
|
||||
|
||||
def resolve_gpu_instance(
|
||||
pci: PciLocation,
|
||||
query: Callable[[int, bytearray], None],
|
||||
) -> tuple[int, int]:
|
||||
"""Resolve (device_instance, subdevice_instance) by PCI identity.
|
||||
|
||||
/dev/nvidiaN minors and RM device instances can have different orders;
|
||||
the RM object must be matched by PCI domain/bus/slot, not by index.
|
||||
The RM query exposes domain/bus/slot but no PCI function, so only
|
||||
function-zero devices can be matched (never another function of a
|
||||
multifunction device).
|
||||
"""
|
||||
if pci.func != 0:
|
||||
raise RmPowerError("RM GPU lookup requires PCI function zero")
|
||||
|
||||
attached = bytearray(_MAX_GPUS * 4)
|
||||
query(_CTRL_GPU_GET_ATTACHED_IDS, attached)
|
||||
|
||||
for i in range(_MAX_GPUS):
|
||||
gpu_id = struct.unpack_from("<I", attached, i * 4)[0]
|
||||
if gpu_id == _INVALID_GPU_ID:
|
||||
continue
|
||||
|
||||
# NV0000_CTRL_GPU_GET_PCI_INFO_PARAMS: u32 gpuId, u32 domain,
|
||||
# u16 bus, u16 slot.
|
||||
location = bytearray(12)
|
||||
location[0:4] = struct.pack("<I", gpu_id)
|
||||
query(_CTRL_GPU_GET_PCI_INFO, location)
|
||||
domain, bus, slot = struct.unpack_from("<IHH", location, 4)
|
||||
if (domain, bus, slot) != (pci.domain, pci.bus, pci.dev):
|
||||
continue
|
||||
|
||||
# NV0000_CTRL_GPU_GET_ID_INFO_V2_PARAMS: eight u32 fields, with
|
||||
# deviceInstance/subDeviceInstance at +8/+12.
|
||||
info = bytearray(32)
|
||||
info[0:4] = struct.pack("<I", gpu_id)
|
||||
query(_CTRL_GPU_GET_ID_INFO_V2, info)
|
||||
device, subdevice = struct.unpack_from("<II", info, 8)
|
||||
return device, subdevice
|
||||
|
||||
raise RmPowerError(f"no RM GPU matches PCI location {pci}")
|
||||
|
||||
|
||||
# ── RM handle ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _rm_control(fd: int, client: int, obj: int, cmd: int, buf: bytearray) -> None:
|
||||
"""Issue an NVOS54 RM control whose parameter block is a byte buffer."""
|
||||
arr = (ctypes.c_uint8 * len(buf)).from_buffer(buf)
|
||||
req = _NVOS54(
|
||||
hClient=client,
|
||||
hObject=obj,
|
||||
cmd=cmd,
|
||||
flags=0,
|
||||
params=ctypes.addressof(arr),
|
||||
paramsSize=len(buf),
|
||||
status=0,
|
||||
)
|
||||
_ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS54), NV_ESC_RM_CONTROL), req)
|
||||
if req.status != 0:
|
||||
raise RmPowerError(
|
||||
f"RM control 0x{cmd:08x} failed with status 0x{req.status:x}"
|
||||
)
|
||||
|
||||
|
||||
def _alloc_client(fd: int) -> int:
|
||||
"""Allocate an RM client (NVOS21, all-zero parameters)."""
|
||||
req = _NVOS21()
|
||||
_ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS21), NV_ESC_RM_ALLOC), req)
|
||||
if req.status != 0:
|
||||
raise RmPowerError(f"could not allocate RM client (status 0x{req.status:x})")
|
||||
return req.hObjectNew
|
||||
|
||||
|
||||
def _alloc_object(
|
||||
fd: int, client: int, parent: int, class_id: int, alloc_params: ctypes.Structure
|
||||
) -> int:
|
||||
"""Allocate an RM object (NVOS64) and return its handle."""
|
||||
req = _NVOS64(
|
||||
hRoot=client,
|
||||
hObjectParent=parent,
|
||||
hObjectNew=0,
|
||||
hClass=class_id,
|
||||
pAllocParms=ctypes.addressof(alloc_params),
|
||||
pRightsRequested=0,
|
||||
paramsSize=ctypes.sizeof(alloc_params),
|
||||
flags=0,
|
||||
status=0,
|
||||
)
|
||||
_ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS64), NV_ESC_RM_ALLOC), req)
|
||||
if req.status != 0:
|
||||
raise RmPowerError(
|
||||
f"RM class 0x{class_id:x} allocation failed (status 0x{req.status:x})"
|
||||
)
|
||||
return req.hObjectNew
|
||||
|
||||
|
||||
def _register_fd(device_fd: int, nvidiactl_fd: int) -> None:
|
||||
"""Register the nvidiactl client with the device fd (NV_ESC_REGISTER_FD).
|
||||
|
||||
The ioctl is issued on the /dev/nvidiaN fd; the argument is the
|
||||
nvidiactl fd to associate with it.
|
||||
"""
|
||||
_ioctl_call(
|
||||
device_fd,
|
||||
_ioctl_rw(4, NV_ESC_REGISTER_FD, NV_IOCTL_MAGIC_F),
|
||||
struct.pack("i", nvidiactl_fd),
|
||||
)
|
||||
|
||||
|
||||
class RmHandle:
|
||||
"""An NVIDIA RM client with device + subdevice objects for one GPU."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
nvidiactl_fd: int,
|
||||
device_fd: int,
|
||||
client_handle: int,
|
||||
device_handle: int,
|
||||
subdevice_handle: int,
|
||||
) -> None:
|
||||
self._nvidiactl_fd = nvidiactl_fd
|
||||
self._device_fd = device_fd
|
||||
self.client_handle = client_handle
|
||||
self.device_handle = device_handle
|
||||
self.subdevice_handle = subdevice_handle
|
||||
|
||||
@classmethod
|
||||
def open(cls, gpu_index: int) -> RmHandle:
|
||||
"""Open an RM handle for the GPU at the given NVML index.
|
||||
|
||||
The RM device/subdevice instances are resolved by PCI identity
|
||||
(minors and RM instances can have different orders).
|
||||
"""
|
||||
pynvml = _ensure_nvml()
|
||||
|
||||
try:
|
||||
handle = pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
||||
minor = int(pynvml.nvmlDeviceGetMinorNumber(handle))
|
||||
pci_info = pynvml.nvmlDeviceGetPciInfo(handle)
|
||||
pci = PciLocation(
|
||||
domain=int(pci_info.domain),
|
||||
bus=int(pci_info.bus),
|
||||
dev=int(pci_info.device),
|
||||
)
|
||||
except Exception as exc:
|
||||
raise RmPowerError(f"NVML query for GPU {gpu_index} failed: {exc}") from exc
|
||||
|
||||
try:
|
||||
nvidiactl_fd = os.open("/dev/nvidiactl", os.O_RDWR)
|
||||
except OSError as exc:
|
||||
raise RmPowerError(f"could not open /dev/nvidiactl: {exc}") from exc
|
||||
|
||||
try:
|
||||
client_handle = _alloc_client(nvidiactl_fd)
|
||||
device_instance, subdevice_instance = resolve_gpu_instance(
|
||||
pci,
|
||||
lambda cmd, buf: _rm_control(
|
||||
nvidiactl_fd, client_handle, client_handle, cmd, buf
|
||||
),
|
||||
)
|
||||
except RmPowerError:
|
||||
os.close(nvidiactl_fd)
|
||||
raise
|
||||
|
||||
try:
|
||||
device_fd = os.open(f"/dev/nvidia{minor}", os.O_RDWR)
|
||||
except OSError as exc:
|
||||
os.close(nvidiactl_fd)
|
||||
raise RmPowerError(f"could not open /dev/nvidia{minor}: {exc}") from exc
|
||||
|
||||
try:
|
||||
_register_fd(device_fd, nvidiactl_fd)
|
||||
device_handle = _alloc_object(
|
||||
nvidiactl_fd,
|
||||
client_handle,
|
||||
client_handle,
|
||||
NV01_DEVICE_0,
|
||||
_NV0080_ALLOC(deviceId=device_instance),
|
||||
)
|
||||
subdevice_handle = _alloc_object(
|
||||
nvidiactl_fd,
|
||||
client_handle,
|
||||
device_handle,
|
||||
NV20_SUBDEVICE_0,
|
||||
_NV2080_ALLOC(subDeviceId=subdevice_instance),
|
||||
)
|
||||
except RmPowerError:
|
||||
os.close(device_fd)
|
||||
os.close(nvidiactl_fd)
|
||||
raise
|
||||
|
||||
return cls(
|
||||
nvidiactl_fd, device_fd, client_handle, device_handle, subdevice_handle
|
||||
)
|
||||
|
||||
def control(self, cmd: int, buf: bytearray) -> None:
|
||||
"""Issue an NVOS54 RM control on the subdevice with a byte buffer."""
|
||||
_rm_control(
|
||||
self._nvidiactl_fd, self.client_handle, self.subdevice_handle, cmd, buf
|
||||
)
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close the fds; the driver reclaims the RM client objects."""
|
||||
with contextlib.suppress(OSError):
|
||||
os.close(self._device_fd)
|
||||
with contextlib.suppress(OSError):
|
||||
os.close(self._nvidiactl_fd)
|
||||
|
||||
|
||||
# ── Power-limit probe / set (pure logic, testable with a fake query) ─────────
|
||||
|
||||
|
||||
def _u32(data: bytearray | bytes, offset: int) -> int:
|
||||
return struct.unpack_from("<I", data, offset)[0]
|
||||
|
||||
|
||||
def _validate_header(layout: PowerLimitLayout, data: bytearray) -> None:
|
||||
if _u32(data, 0) != 0xFF or _u32(data, 4) != 1:
|
||||
raise RmPowerError("unrecognized RM power client layout")
|
||||
# The extended layout has additional mask words; accepting only its low
|
||||
# word would allow an unexpected client to be included in a later SET.
|
||||
if any(byte != 0 for byte in data[8 : layout.mask_end]):
|
||||
raise RmPowerError("unrecognized RM power client layout")
|
||||
|
||||
|
||||
def _read_bounds(
|
||||
layout: PowerLimitLayout, query: Callable[[int, bytearray], None]
|
||||
) -> PowerLimitBounds:
|
||||
info = bytearray(layout.info_size)
|
||||
query(_PWR_GET_INFO, info)
|
||||
_validate_header(layout, info)
|
||||
bounds = PowerLimitBounds(
|
||||
min_mw=_u32(info, layout.info_min_at),
|
||||
default_mw=_u32(info, layout.info_min_at + 4),
|
||||
max_mw=_u32(info, layout.info_min_at + 8),
|
||||
)
|
||||
if not (
|
||||
bounds.min_mw > 0
|
||||
and bounds.min_mw <= bounds.default_mw
|
||||
and bounds.default_mw <= bounds.max_mw
|
||||
):
|
||||
raise RmPowerError("invalid RM power limit bounds")
|
||||
return bounds
|
||||
|
||||
|
||||
def _read_control(
|
||||
layout: PowerLimitLayout, query: Callable[[int, bytearray], None]
|
||||
) -> bytearray:
|
||||
control = bytearray(layout.control_size)
|
||||
control[4:8] = struct.pack("<I", 1)
|
||||
control[layout.client_at] = _ORDINARY_CLIENT
|
||||
query(_PWR_GET_CONTROL, control)
|
||||
_validate_header(layout, control)
|
||||
if control[layout.client_at] != _ORDINARY_CLIENT:
|
||||
raise RmPowerError("unexpected power client")
|
||||
if _u32(control, layout.request_at) in (0, 0xFFFFFFFF):
|
||||
raise RmPowerError("no ordinary power request available")
|
||||
return control
|
||||
|
||||
|
||||
def probe(
|
||||
nvml_bounds: PowerLimitBounds,
|
||||
nvml_current_mw: int,
|
||||
query: Callable[[int, bytearray], None],
|
||||
) -> LowerPowerLimit:
|
||||
"""GET-only discovery of the RM power-limit layout.
|
||||
|
||||
Probes the two known wire formats using GETs only. A driver version
|
||||
number is not evidence that the payload still has the same layout or
|
||||
units, so the bounds and the current request are validated against
|
||||
NVML. Discovery never issues a SET.
|
||||
"""
|
||||
if sys.byteorder != "little":
|
||||
raise RmPowerError("little-endian host required")
|
||||
errors: list[str] = []
|
||||
for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
|
||||
try:
|
||||
bounds = _read_bounds(layout, query)
|
||||
if bounds != nvml_bounds:
|
||||
raise RmPowerError("RM power bounds differ from NVML")
|
||||
control = _read_control(layout, query)
|
||||
if _u32(control, layout.request_at) != nvml_current_mw:
|
||||
raise RmPowerError("RM ordinary power request differs from NVML")
|
||||
return LowerPowerLimit(bounds=bounds, layout=layout)
|
||||
except RmPowerError as exc:
|
||||
errors.append(f"{layout.name}: {exc}")
|
||||
raise RmPowerError("no compatible RM power layout: " + "; ".join(errors))
|
||||
|
||||
|
||||
def set_limit(
|
||||
limit_mw: int,
|
||||
support: LowerPowerLimit,
|
||||
query: Callable[[int, bytearray], None],
|
||||
) -> None:
|
||||
"""Set the ordinary-client power request with readback verification.
|
||||
|
||||
Keeps the entire current payload, changing only entry 0's request.
|
||||
Mask 1 and selector 0xFE prevent modifying any other entry or the
|
||||
additional F8 client. A failed SET can have side effects, so the
|
||||
previous request is restored even on transport failure — and the
|
||||
restore uses 0xFE so a previous limit below the VBIOS minimum can
|
||||
also be restored.
|
||||
"""
|
||||
layout = support.layout
|
||||
bounds = _read_bounds(layout, query)
|
||||
if bounds != support.bounds:
|
||||
raise RmPowerError("RM power bounds changed since discovery")
|
||||
lower = bounds.lower_min_mw()
|
||||
if not (lower <= limit_mw <= bounds.max_mw):
|
||||
raise RmPowerError(
|
||||
f"power limit {limit_mw} mW outside supported range "
|
||||
f"{lower}..{bounds.max_mw} mW"
|
||||
)
|
||||
|
||||
before = _read_control(layout, query)
|
||||
if _u32(before, layout.request_at) == limit_mw:
|
||||
return
|
||||
|
||||
expected = bytearray(before)
|
||||
expected[layout.request_at : layout.request_at + 4] = struct.pack("<I", limit_mw)
|
||||
|
||||
try:
|
||||
request = bytearray(expected)
|
||||
query(_PWR_SET_CONTROL, request)
|
||||
if _read_control(layout, query) != expected:
|
||||
raise RmPowerError("power request readback differs")
|
||||
except RmPowerError as apply_error:
|
||||
try:
|
||||
restore = bytearray(before)
|
||||
query(_PWR_SET_CONTROL, restore)
|
||||
if _read_control(layout, query) != before:
|
||||
raise RmPowerError("restored power request differs")
|
||||
except RmPowerError as restore_error:
|
||||
raise RmPowerError(
|
||||
f"power request failed: {apply_error}; "
|
||||
f"restoration also failed: {restore_error}"
|
||||
) from None
|
||||
raise RmPowerError(f"{apply_error} (previous power request restored)") from None
|
||||
|
||||
|
||||
# ── High-level API (wires NVML state + RmHandle to the pure logic) ───────────
|
||||
|
||||
|
||||
def _ensure_nvml():
|
||||
"""Return pynvml with NVML initialized (nvmlInit is refcounted)."""
|
||||
import pynvml
|
||||
|
||||
pynvml.nvmlInit()
|
||||
return pynvml
|
||||
|
||||
|
||||
def _nvml_power_state(gpu_index: int) -> tuple[PowerLimitBounds, int]:
|
||||
"""Return (bounds, current_mw) from NVML for the given GPU."""
|
||||
pynvml = _ensure_nvml()
|
||||
|
||||
try:
|
||||
handle = pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
||||
min_mw, max_mw = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle)
|
||||
default_mw = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
|
||||
current_mw = pynvml.nvmlDeviceGetPowerManagementLimit(handle)
|
||||
bounds = PowerLimitBounds(int(min_mw), int(default_mw), int(max_mw))
|
||||
current = int(current_mw)
|
||||
except Exception as exc:
|
||||
raise RmPowerError(
|
||||
f"NVML power state for GPU {gpu_index} unavailable: {exc}"
|
||||
) from exc
|
||||
return bounds, current
|
||||
|
||||
|
||||
def probe_gpu(gpu_index: int = 0) -> PowerLimitBounds | None:
|
||||
"""GET-only discovery of the RM power-limit interface for a GPU.
|
||||
|
||||
Returns the validated power bounds (milliwatts) when a compatible RM
|
||||
layout is present, else None. Never issues a write.
|
||||
"""
|
||||
try:
|
||||
bounds, current = _nvml_power_state(gpu_index)
|
||||
except Exception as exc:
|
||||
log.debug("RM probe: NVML state unavailable: %s", exc)
|
||||
return None
|
||||
try:
|
||||
handle = RmHandle.open(gpu_index)
|
||||
except RmPowerError as exc:
|
||||
log.debug("RM probe: handle open failed: %s", exc)
|
||||
return None
|
||||
try:
|
||||
probe(bounds, current, handle.control)
|
||||
return bounds
|
||||
except RmPowerError as exc:
|
||||
log.debug("RM probe: %s", exc)
|
||||
return None
|
||||
finally:
|
||||
handle.close()
|
||||
|
||||
|
||||
def set_power_limit_w(gpu_index: int, limit_w: int) -> None:
|
||||
"""Set the board power limit (watts) via the RM interface.
|
||||
|
||||
Raises RmPowerError on any failure (probe, range, write, readback).
|
||||
A failed write restores the previous request.
|
||||
"""
|
||||
bounds, current = _nvml_power_state(gpu_index)
|
||||
handle = RmHandle.open(gpu_index)
|
||||
try:
|
||||
support = probe(bounds, current, handle.control)
|
||||
set_limit(int(limit_w) * 1000, support, handle.control)
|
||||
finally:
|
||||
handle.close()
|
||||
@@ -9,7 +9,7 @@ from .errors import NVAPI_ERRORS
|
||||
|
||||
def load_nvapi() -> ctypes.CDLL:
|
||||
"""Load libnvidia-api.so from the NVIDIA driver."""
|
||||
for name in ("libnvidia-api.so", "libnvidia-api.so.1"):
|
||||
for name in ("libnvidia-api.so", "libnvidia-api.so.1"): # gitleaks:allow
|
||||
try:
|
||||
return ctypes.CDLL(name)
|
||||
except OSError:
|
||||
|
||||
@@ -43,7 +43,8 @@ def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
|
||||
errs.append(f"Mem offset: {msg}")
|
||||
|
||||
if profile.power_limit_w is not None:
|
||||
ok, msg = set_power_limit(profile.power_limit_w, gpu_index)
|
||||
mode = profile.power_cap_mode or "nvml"
|
||||
ok, msg = set_power_limit(profile.power_limit_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power limit: {msg}")
|
||||
|
||||
|
||||
@@ -16,6 +16,9 @@ class ProfileData:
|
||||
curve_deltas: dict[str, int] # { "index": delta_khz }
|
||||
mem_offset_mhz: int | None = None
|
||||
power_limit_w: int | None = None
|
||||
# How power_limit_w is applied: "nvml" (default) or "ioctl" (experimental
|
||||
# RM power control, permits values below the VBIOS minimum).
|
||||
power_cap_mode: str | None = None
|
||||
fan_curve: list[dict[str, int]] | None = None
|
||||
# Fan indices controlled by fan_curve (0-based); None = all fans.
|
||||
fan_targets: list[int] | None = None
|
||||
@@ -55,6 +58,9 @@ def load_profile(filepath: str) -> ProfileData:
|
||||
# Drop removed fields so old profiles don't cause TypeError.
|
||||
for obsolete in ("gpu_locked_min_mhz", "gpu_locked_max_mhz", "vram_p0_offset_mhz"):
|
||||
data.pop(obsolete, None)
|
||||
# Normalize the experimental power-cap mode; unknown values fall back to NVML.
|
||||
if data.get("power_cap_mode") not in (None, "nvml", "ioctl"):
|
||||
data["power_cap_mode"] = None
|
||||
return ProfileData(**data)
|
||||
|
||||
|
||||
|
||||
+56
-6
@@ -569,6 +569,8 @@ class SnapshotRestoreRequest(BaseModel):
|
||||
class LimitsRequest(BaseModel):
|
||||
power_limit_w: int | None = None
|
||||
mem_offset_mhz: int | None = None
|
||||
# "nvml" (default) or "ioctl" (experimental RM power control).
|
||||
power_cap_mode: str | None = None
|
||||
|
||||
|
||||
class ProfileSaveRequest(BaseModel):
|
||||
@@ -914,6 +916,17 @@ def _persist_fan_curves(fan_curves: dict) -> None:
|
||||
_persist_config_field("fan_curves", fan_curves if fan_curves else None)
|
||||
|
||||
|
||||
def _persist_power_cap_modes(modes: dict[str, str]) -> None:
|
||||
"""Persist the per-GPU experimental power-cap mode dict to config.json."""
|
||||
_persist_config_field("power_cap_modes", modes if modes else None)
|
||||
|
||||
|
||||
def _power_cap_mode(cfg: Config, gpu_index: int) -> str:
|
||||
"""Return the effective power-cap mode for a GPU ("nvml" or "ioctl")."""
|
||||
mode = cfg.power_cap_modes.get(_gpu_stable_key(gpu_index), "nvml")
|
||||
return mode if mode in ("nvml", "ioctl") else "nvml"
|
||||
|
||||
|
||||
@app.get("/api/profiles")
|
||||
async def api_profiles(gpu_index: int = 0):
|
||||
"""List saved native profiles, the active profile name, and the auto-load profile name."""
|
||||
@@ -941,13 +954,15 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
|
||||
curve_deltas = {str(p.index): p.delta_khz for p in state.points if p.delta_khz != 0}
|
||||
|
||||
try:
|
||||
power_info = await _run(get_power_limit, gpu_index)
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
power_info = await _run(get_power_limit, gpu_index, mode)
|
||||
offsets = await _run(get_clock_offsets, gpu_index)
|
||||
power_limit_w = power_info.get("power_limit_w")
|
||||
mem_offset_mhz = offsets.get("mem_offset_mhz")
|
||||
except Exception:
|
||||
power_limit_w = None
|
||||
mem_offset_mhz = None
|
||||
mode = "nvml"
|
||||
|
||||
data = ProfileData(
|
||||
name=req.name,
|
||||
@@ -955,6 +970,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
|
||||
curve_deltas=curve_deltas,
|
||||
mem_offset_mhz=mem_offset_mhz,
|
||||
power_limit_w=power_limit_w,
|
||||
power_cap_mode=mode,
|
||||
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
|
||||
fan_targets=g_state.get("fan_targets")
|
||||
if g_state.get("fan_curve_active")
|
||||
@@ -1075,7 +1091,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
errs.append(f"Mem offset: {msg}")
|
||||
|
||||
if profile.power_limit_w is not None:
|
||||
ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index)
|
||||
mode = profile.power_cap_mode or "nvml"
|
||||
ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power limit: {msg}")
|
||||
|
||||
@@ -1204,7 +1221,9 @@ async def api_config_update(req: ConfigUpdateRequest):
|
||||
@app.get("/api/limits")
|
||||
async def api_limits(gpu_index: int = 0):
|
||||
"""Current performance limits: power and clock offsets."""
|
||||
power = await _run(get_power_limit, gpu_index)
|
||||
cfg: Config = _state["config"]
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
power = await _run(get_power_limit, gpu_index, mode)
|
||||
offsets = await _run(get_clock_offsets, gpu_index)
|
||||
mem_off_range = await _run(get_mem_offset_range, gpu_index)
|
||||
return {
|
||||
@@ -1218,10 +1237,36 @@ async def api_limits(gpu_index: int = 0):
|
||||
async def api_limits_update(req: LimitsRequest, gpu_index: int = 0):
|
||||
"""Update performance limits."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
cfg: Config = _state["config"]
|
||||
errs = []
|
||||
|
||||
if req.power_cap_mode is not None:
|
||||
if req.power_cap_mode not in ("nvml", "ioctl"):
|
||||
raise HTTPException(
|
||||
status_code=400, detail="power_cap_mode must be 'nvml' or 'ioctl'"
|
||||
)
|
||||
if req.power_cap_mode == "ioctl":
|
||||
# Verify the GPU actually exposes the RM interface before enabling,
|
||||
# so a client can't lock a GPU into a mode where every power
|
||||
# operation fails (ioctl mode has no NVML fallback by design).
|
||||
info = await _run(get_power_limit, gpu_index, "ioctl")
|
||||
if not info.get("rm_power_supported"):
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail="Experimental RM power control is not supported "
|
||||
"on this GPU/driver",
|
||||
)
|
||||
key = _gpu_stable_key(gpu_index)
|
||||
if req.power_cap_mode == "nvml":
|
||||
cfg.power_cap_modes.pop(key, None)
|
||||
else:
|
||||
cfg.power_cap_modes[key] = "ioctl"
|
||||
_persist_power_cap_modes(cfg.power_cap_modes)
|
||||
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
|
||||
if req.power_limit_w is not None:
|
||||
ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index)
|
||||
ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power Limit: {msg}")
|
||||
|
||||
@@ -1284,12 +1329,17 @@ async def _update_offsets_and_broadcast(gpu_index: int) -> None:
|
||||
async def api_limits_reset(gpu_index: int = 0):
|
||||
"""Reset power limit to hardware default and memory clock offset to 0."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
cfg: Config = _state["config"]
|
||||
errs = []
|
||||
|
||||
power = await _run(get_power_limit, gpu_index)
|
||||
# Reset uses the GPU's current mode: in ioctl mode the default is
|
||||
# restored through the RM route (which can also restore a previous
|
||||
# below-VBIOS-minimum cap).
|
||||
mode = _power_cap_mode(cfg, gpu_index)
|
||||
power = await _run(get_power_limit, gpu_index, mode)
|
||||
default_w = power.get("default_power_limit_w")
|
||||
if default_w is not None:
|
||||
ok, msg = await _run(set_power_limit, default_w, gpu_index)
|
||||
ok, msg = await _run(set_power_limit, default_w, gpu_index, mode)
|
||||
if not ok:
|
||||
errs.append(f"Power Limit: {msg}")
|
||||
|
||||
|
||||
Reference in new issue
Block a user