"""Undocumented NVIDIA RM power-limit interface (EXPERIMENTAL). Port of the approach from LACT PR #1205 (ilya-zlobintsev/LACT): applies board power limits through the private NV2080 power-limit "ordinary client" interface on /dev/nvidiactl, which permits caps below the VBIOS minimum (down to 30 W). The native maximum still applies. EXPERIMENTAL — uses an undocumented driver interface. It may break after driver updates. Discovery is GET-only and validates the RM payload against NVML before any write is issued; a failed write restores the previous request (even if it was below the VBIOS minimum). """ from __future__ import annotations import contextlib import ctypes import fcntl import logging import os import struct import sys from collections.abc import Callable from dataclasses import dataclass log = logging.getLogger("nvcurve.hal.rm_power") # ── ioctl constants (nv-ioctl.h / nv-ioctl-numbers.h) ───────────────────────── NV_IOCTL_MAGIC = ord("N") # 0x4E — user-space RM interface NV_ESC_RM_ALLOC = 0x2B NV_ESC_RM_CONTROL = 0x2A # 'F' magic interface (kernel-open/common/inc/nv-ioctl-numbers.h) — # NV_ESC_REGISTER_FD lives here, not in the 'N' RM interface. NV_IOCTL_MAGIC_F = ord("F") # 0x46 NV_IOCTL_BASE_F = 200 NV_ESC_REGISTER_FD = NV_IOCTL_BASE_F + 1 # 201 # RM class IDs (nv0080.h / nv2080.h) NV01_DEVICE_0 = 0x0080 NV20_SUBDEVICE_0 = 0x2080 # NV01_ROOT GPU queries (ctrl0000gpu.h) — resolve PCI identity to the RM # device/subdevice instance numbers used by NV0080 and NV2080 allocations; # neither number is a Linux device minor. _CTRL_GPU_GET_ATTACHED_IDS = 0x201 _CTRL_GPU_GET_ID_INFO_V2 = 0x205 _CTRL_GPU_GET_PCI_INFO = 0x21B _MAX_GPUS = 32 _INVALID_GPU_ID = 0xFFFFFFFF # Private NV2080 power-limit client commands. Payloads compared against # NvAPI and GSP from R595, R610 and R615 (native RM payloads, without # NvAPI's 0x10-byte transport prefix). _PWR_GET_INFO = 0x2080_A630 _PWR_GET_CONTROL = 0x2080_A632 _PWR_SET_CONTROL = 0x2080_E633 _ORDINARY_CLIENT = 0xFE _LOWER_LIMIT_MW = 30_000 # experimental floor: 30 W def _ioctl_rw(size: int, nr: int, magic: int = NV_IOCTL_MAGIC) -> int: """Linux ioctl request code: dir=RW, given size/type/nr.""" return (2 << 30) | (size << 16) | (magic << 8) | nr def _ioctl_call(fd: int, code: int, arg) -> None: """Issue an ioctl, converting errno failures to RmPowerError. The driver normally reports failures as an RM status in the parameter struct, but an experimental interface can also fail at the kernel level (ENOTTY/EBADF/EPERM across driver versions). Converting to RmPowerError keeps the module's error contract uniform and lets callers clean up fds. """ try: fcntl.ioctl(fd, code, arg) except OSError as exc: raise RmPowerError(f"ioctl 0x{code:x} failed: {exc}") from exc # ── NVOS parameter structs (nvos.h) ────────────────────────────────────────── class _NVOS21(ctypes.Structure): _fields_ = [ ("hRoot", ctypes.c_uint32), ("hObjectParent", ctypes.c_uint32), ("hObjectNew", ctypes.c_uint32), ("hClass", ctypes.c_uint32), ("pAllocParms", ctypes.c_uint64), ("paramsSize", ctypes.c_uint32), ("status", ctypes.c_uint32), ] class _NVOS64(ctypes.Structure): _fields_ = [ ("hRoot", ctypes.c_uint32), ("hObjectParent", ctypes.c_uint32), ("hObjectNew", ctypes.c_uint32), ("hClass", ctypes.c_uint32), ("pAllocParms", ctypes.c_uint64), ("pRightsRequested", ctypes.c_uint64), ("paramsSize", ctypes.c_uint32), ("flags", ctypes.c_uint32), ("status", ctypes.c_uint32), ] class _NVOS54(ctypes.Structure): _fields_ = [ ("hClient", ctypes.c_uint32), ("hObject", ctypes.c_uint32), ("cmd", ctypes.c_uint32), ("flags", ctypes.c_uint32), ("params", ctypes.c_uint64), ("paramsSize", ctypes.c_uint32), ("status", ctypes.c_uint32), ] class _NV0080_ALLOC(ctypes.Structure): _fields_ = [ ("deviceId", ctypes.c_uint32), ("deviceFlags", ctypes.c_uint32), ("vgpuInstance", ctypes.c_uint32), ("pad", ctypes.c_uint32), ] class _NV2080_ALLOC(ctypes.Structure): _fields_ = [ ("subDeviceId", ctypes.c_uint32), ("clientShare", ctypes.c_uint32), ("flags", ctypes.c_uint32), ("pad", ctypes.c_uint32), ] # ── Errors ─────────────────────────────────────────────────────────────────── class RmPowerError(RuntimeError): """Raised when the RM power-limit interface is unavailable or fails.""" # ── Power-limit layouts and bounds ─────────────────────────────────────────── @dataclass(frozen=True) class PowerLimitLayout: """Byte offsets of the private power-limit payloads for one wire format.""" name: str info_size: int control_size: int info_min_at: int request_at: int client_at: int mask_end: int EXTENDED_LAYOUT = PowerLimitLayout( name="extended", info_size=0x924, control_size=0x328, info_min_at=0x28, request_at=0x2C, client_at=0x30, mask_end=0x24, ) LEGACY_LAYOUT = PowerLimitLayout( name="legacy", info_size=0x488, control_size=0x188, info_min_at=0xC, request_at=0xC, client_at=0x10, mask_end=0x8, ) @dataclass(frozen=True) class PowerLimitBounds: """Power limit bounds in milliwatts (NVML/RM units).""" min_mw: int default_mw: int max_mw: int def lower_min_mw(self) -> int: """Effective minimum when the experimental route is active.""" return min(self.min_mw, _LOWER_LIMIT_MW) @dataclass(frozen=True) class LowerPowerLimit: """A validated RM power-limit layout that can be written.""" bounds: PowerLimitBounds layout: PowerLimitLayout def lower_min_mw(self) -> int: return self.bounds.lower_min_mw() # ── PCI identity → RM instance resolution ──────────────────────────────────── @dataclass(frozen=True) class PciLocation: domain: int bus: int dev: int func: int = 0 def resolve_gpu_instance( pci: PciLocation, query: Callable[[int, bytearray], None], ) -> tuple[int, int]: """Resolve (device_instance, subdevice_instance) by PCI identity. /dev/nvidiaN minors and RM device instances can have different orders; the RM object must be matched by PCI domain/bus/slot, not by index. The RM query exposes domain/bus/slot but no PCI function, so only function-zero devices can be matched (never another function of a multifunction device). """ if pci.func != 0: raise RmPowerError("RM GPU lookup requires PCI function zero") attached = bytearray(_MAX_GPUS * 4) query(_CTRL_GPU_GET_ATTACHED_IDS, attached) for i in range(_MAX_GPUS): gpu_id = struct.unpack_from(" None: """Issue an NVOS54 RM control whose parameter block is a byte buffer.""" arr = (ctypes.c_uint8 * len(buf)).from_buffer(buf) req = _NVOS54( hClient=client, hObject=obj, cmd=cmd, flags=0, params=ctypes.addressof(arr), paramsSize=len(buf), status=0, ) _ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS54), NV_ESC_RM_CONTROL), req) if req.status != 0: raise RmPowerError( f"RM control 0x{cmd:08x} failed with status 0x{req.status:x}" ) def _alloc_client(fd: int) -> int: """Allocate an RM client (NVOS21, all-zero parameters).""" req = _NVOS21() _ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS21), NV_ESC_RM_ALLOC), req) if req.status != 0: raise RmPowerError(f"could not allocate RM client (status 0x{req.status:x})") return req.hObjectNew def _alloc_object( fd: int, client: int, parent: int, class_id: int, alloc_params: ctypes.Structure ) -> int: """Allocate an RM object (NVOS64) and return its handle.""" req = _NVOS64( hRoot=client, hObjectParent=parent, hObjectNew=0, hClass=class_id, pAllocParms=ctypes.addressof(alloc_params), pRightsRequested=0, paramsSize=ctypes.sizeof(alloc_params), flags=0, status=0, ) _ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS64), NV_ESC_RM_ALLOC), req) if req.status != 0: raise RmPowerError( f"RM class 0x{class_id:x} allocation failed (status 0x{req.status:x})" ) return req.hObjectNew def _register_fd(device_fd: int, nvidiactl_fd: int) -> None: """Register the nvidiactl client with the device fd (NV_ESC_REGISTER_FD). The ioctl is issued on the /dev/nvidiaN fd; the argument is the nvidiactl fd to associate with it. """ _ioctl_call( device_fd, _ioctl_rw(4, NV_ESC_REGISTER_FD, NV_IOCTL_MAGIC_F), struct.pack("i", nvidiactl_fd), ) class RmHandle: """An NVIDIA RM client with device + subdevice objects for one GPU.""" def __init__( self, nvidiactl_fd: int, device_fd: int, client_handle: int, device_handle: int, subdevice_handle: int, ) -> None: self._nvidiactl_fd = nvidiactl_fd self._device_fd = device_fd self.client_handle = client_handle self.device_handle = device_handle self.subdevice_handle = subdevice_handle @classmethod def open(cls, gpu_index: int) -> RmHandle: """Open an RM handle for the GPU at the given NVML index. The RM device/subdevice instances are resolved by PCI identity (minors and RM instances can have different orders). """ pynvml = _ensure_nvml() try: handle = pynvml.nvmlDeviceGetHandleByIndex(gpu_index) minor = int(pynvml.nvmlDeviceGetMinorNumber(handle)) pci_info = pynvml.nvmlDeviceGetPciInfo(handle) pci = PciLocation( domain=int(pci_info.domain), bus=int(pci_info.bus), dev=int(pci_info.device), ) except Exception as exc: raise RmPowerError(f"NVML query for GPU {gpu_index} failed: {exc}") from exc try: nvidiactl_fd = os.open("/dev/nvidiactl", os.O_RDWR) except OSError as exc: raise RmPowerError(f"could not open /dev/nvidiactl: {exc}") from exc try: client_handle = _alloc_client(nvidiactl_fd) device_instance, subdevice_instance = resolve_gpu_instance( pci, lambda cmd, buf: _rm_control( nvidiactl_fd, client_handle, client_handle, cmd, buf ), ) except RmPowerError: os.close(nvidiactl_fd) raise try: device_fd = os.open(f"/dev/nvidia{minor}", os.O_RDWR) except OSError as exc: os.close(nvidiactl_fd) raise RmPowerError(f"could not open /dev/nvidia{minor}: {exc}") from exc try: _register_fd(device_fd, nvidiactl_fd) device_handle = _alloc_object( nvidiactl_fd, client_handle, client_handle, NV01_DEVICE_0, _NV0080_ALLOC(deviceId=device_instance), ) subdevice_handle = _alloc_object( nvidiactl_fd, client_handle, device_handle, NV20_SUBDEVICE_0, _NV2080_ALLOC(subDeviceId=subdevice_instance), ) except RmPowerError: os.close(device_fd) os.close(nvidiactl_fd) raise return cls( nvidiactl_fd, device_fd, client_handle, device_handle, subdevice_handle ) def control(self, cmd: int, buf: bytearray) -> None: """Issue an NVOS54 RM control on the subdevice with a byte buffer.""" _rm_control( self._nvidiactl_fd, self.client_handle, self.subdevice_handle, cmd, buf ) def close(self) -> None: """Close the fds; the driver reclaims the RM client objects.""" with contextlib.suppress(OSError): os.close(self._device_fd) with contextlib.suppress(OSError): os.close(self._nvidiactl_fd) # ── Power-limit probe / set (pure logic, testable with a fake query) ───────── def _u32(data: bytearray | bytes, offset: int) -> int: return struct.unpack_from(" None: if _u32(data, 0) != 0xFF or _u32(data, 4) != 1: raise RmPowerError("unrecognized RM power client layout") # The extended layout has additional mask words; accepting only its low # word would allow an unexpected client to be included in a later SET. if any(byte != 0 for byte in data[8 : layout.mask_end]): raise RmPowerError("unrecognized RM power client layout") def _read_bounds( layout: PowerLimitLayout, query: Callable[[int, bytearray], None] ) -> PowerLimitBounds: info = bytearray(layout.info_size) query(_PWR_GET_INFO, info) _validate_header(layout, info) bounds = PowerLimitBounds( min_mw=_u32(info, layout.info_min_at), default_mw=_u32(info, layout.info_min_at + 4), max_mw=_u32(info, layout.info_min_at + 8), ) if not ( bounds.min_mw > 0 and bounds.min_mw <= bounds.default_mw and bounds.default_mw <= bounds.max_mw ): raise RmPowerError("invalid RM power limit bounds") return bounds def _read_control( layout: PowerLimitLayout, query: Callable[[int, bytearray], None] ) -> bytearray: control = bytearray(layout.control_size) control[4:8] = struct.pack(" LowerPowerLimit: """GET-only discovery of the RM power-limit layout. Probes the two known wire formats using GETs only. A driver version number is not evidence that the payload still has the same layout or units, so the bounds and the current request are validated against NVML. Discovery never issues a SET. """ if sys.byteorder != "little": raise RmPowerError("little-endian host required") errors: list[str] = [] for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT): try: bounds = _read_bounds(layout, query) if bounds != nvml_bounds: raise RmPowerError("RM power bounds differ from NVML") control = _read_control(layout, query) if _u32(control, layout.request_at) != nvml_current_mw: raise RmPowerError("RM ordinary power request differs from NVML") return LowerPowerLimit(bounds=bounds, layout=layout) except RmPowerError as exc: errors.append(f"{layout.name}: {exc}") raise RmPowerError("no compatible RM power layout: " + "; ".join(errors)) def set_limit( limit_mw: int, support: LowerPowerLimit, query: Callable[[int, bytearray], None], ) -> None: """Set the ordinary-client power request with readback verification. Keeps the entire current payload, changing only entry 0's request. Mask 1 and selector 0xFE prevent modifying any other entry or the additional F8 client. A failed SET can have side effects, so the previous request is restored even on transport failure — and the restore uses 0xFE so a previous limit below the VBIOS minimum can also be restored. """ layout = support.layout bounds = _read_bounds(layout, query) if bounds != support.bounds: raise RmPowerError("RM power bounds changed since discovery") lower = bounds.lower_min_mw() if not (lower <= limit_mw <= bounds.max_mw): raise RmPowerError( f"power limit {limit_mw} mW outside supported range " f"{lower}..{bounds.max_mw} mW" ) before = _read_control(layout, query) if _u32(before, layout.request_at) == limit_mw: return expected = bytearray(before) expected[layout.request_at : layout.request_at + 4] = struct.pack(" tuple[PowerLimitBounds, int]: """Return (bounds, current_mw) from NVML for the given GPU.""" pynvml = _ensure_nvml() try: handle = pynvml.nvmlDeviceGetHandleByIndex(gpu_index) min_mw, max_mw = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle) default_mw = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle) current_mw = pynvml.nvmlDeviceGetPowerManagementLimit(handle) bounds = PowerLimitBounds(int(min_mw), int(default_mw), int(max_mw)) current = int(current_mw) except Exception as exc: raise RmPowerError( f"NVML power state for GPU {gpu_index} unavailable: {exc}" ) from exc return bounds, current def probe_gpu(gpu_index: int = 0) -> PowerLimitBounds | None: """GET-only discovery of the RM power-limit interface for a GPU. Returns the validated power bounds (milliwatts) when a compatible RM layout is present, else None. Never issues a write. """ try: bounds, current = _nvml_power_state(gpu_index) except Exception as exc: log.debug("RM probe: NVML state unavailable: %s", exc) return None try: handle = RmHandle.open(gpu_index) except RmPowerError as exc: log.debug("RM probe: handle open failed: %s", exc) return None try: probe(bounds, current, handle.control) return bounds except RmPowerError as exc: log.debug("RM probe: %s", exc) return None finally: handle.close() def set_power_limit_w(gpu_index: int, limit_w: int) -> None: """Set the board power limit (watts) via the RM interface. Raises RmPowerError on any failure (probe, range, write, readback). A failed write restores the previous request. """ bounds, current = _nvml_power_state(gpu_index) handle = RmHandle.open(gpu_index) try: support = probe(bounds, current, handle.control) set_limit(int(limit_w) * 1000, support, handle.control) finally: handle.close()